diff --git a/.playwright-mcp/console-2026-04-04T08-03-37-016Z.log b/.playwright-mcp/console-2026-04-04T08-03-37-016Z.log
deleted file mode 100644
index 1fcd29f636..0000000000
--- a/.playwright-mcp/console-2026-04-04T08-03-37-016Z.log
+++ /dev/null
@@ -1 +0,0 @@
-[ 32698ms] [ERROR] Failed to load resource: the server responded with a status of 404 (Not Found) @ http://localhost:20130/dashboard/usage?_rsc=18t7j:0
diff --git a/.playwright-mcp/page-2026-04-04T08-03-38-145Z.yml b/.playwright-mcp/page-2026-04-04T08-03-38-145Z.yml
deleted file mode 100644
index dc09664b78..0000000000
--- a/.playwright-mcp/page-2026-04-04T08-03-38-145Z.yml
+++ /dev/null
@@ -1,158 +0,0 @@
-- generic [active] [ref=e1]:
- - link "Skip to content" [ref=e2] [cursor=pointer]:
- - /url: "#main-content"
- - generic [ref=e3]:
- - complementary [ref=e5]:
- - link "Skip to content" [ref=e6] [cursor=pointer]:
- - /url: "#main-content"
- - link "OmniRoute v3.5.0" [ref=e12] [cursor=pointer]:
- - /url: /dashboard
- - img [ref=e14]
- - generic [ref=e26]:
- - heading "OmniRoute" [level=1] [ref=e27]
- - generic [ref=e28]: v3.5.0
- - navigation "Main navigation" [ref=e29]:
- - generic [ref=e30]:
- - link "home Home" [ref=e31] [cursor=pointer]:
- - /url: /dashboard
- - generic [ref=e32]: home
- - generic [ref=e33]: Home
- - link "api Endpoints" [ref=e34] [cursor=pointer]:
- - /url: /dashboard/endpoint
- - generic [ref=e35]: api
- - generic [ref=e36]: Endpoints
- - link "vpn_key API Manager" [ref=e37] [cursor=pointer]:
- - /url: /dashboard/api-manager
- - generic [ref=e38]: vpn_key
- - generic [ref=e39]: API Manager
- - link "dns Providers" [ref=e40] [cursor=pointer]:
- - /url: /dashboard/providers
- - generic [ref=e41]: dns
- - generic [ref=e42]: Providers
- - link "layers Combos" [ref=e43] [cursor=pointer]:
- - /url: /dashboard/combos
- - generic [ref=e44]: layers
- - generic [ref=e45]: Combos
- - link "auto_awesome Auto Combo" [ref=e46] [cursor=pointer]:
- - /url: /dashboard/auto-combo
- - generic [ref=e47]: auto_awesome
- - generic [ref=e48]: Auto Combo
- - link "account_balance_wallet Costs" [ref=e49] [cursor=pointer]:
- - /url: /dashboard/costs
- - generic [ref=e50]: account_balance_wallet
- - generic [ref=e51]: Costs
- - link "analytics Analytics" [ref=e52] [cursor=pointer]:
- - /url: /dashboard/analytics
- - generic [ref=e53]: analytics
- - generic [ref=e54]: Analytics
- - link "tune Limits & Quotas" [ref=e55] [cursor=pointer]:
- - /url: /dashboard/limits
- - generic [ref=e56]: tune
- - generic [ref=e57]: Limits & Quotas
- - link "cached Cache" [ref=e58] [cursor=pointer]:
- - /url: /dashboard/cache
- - generic [ref=e59]: cached
- - generic [ref=e60]: Cache
- - link "perm_media Media" [ref=e61] [cursor=pointer]:
- - /url: /dashboard/cache/media
- - generic [ref=e62]: perm_media
- - generic [ref=e63]: Media
- - generic [ref=e64]:
- - paragraph [ref=e65]: CLI
- - link "terminal Tools" [ref=e66] [cursor=pointer]:
- - /url: /dashboard/cli-tools
- - generic [ref=e67]: terminal
- - generic [ref=e68]: Tools
- - link "smart_toy Agents" [ref=e69] [cursor=pointer]:
- - /url: /dashboard/agents
- - generic [ref=e70]: smart_toy
- - generic [ref=e71]: Agents
- - link "psychology Memory" [ref=e72] [cursor=pointer]:
- - /url: /dashboard/memory
- - generic [ref=e73]: psychology
- - generic [ref=e74]: Memory
- - link "auto_fix_high Skills" [ref=e75] [cursor=pointer]:
- - /url: /dashboard/skills
- - generic [ref=e76]: auto_fix_high
- - generic [ref=e77]: Skills
- - generic [ref=e78]:
- - paragraph [ref=e79]: System
- - link "health_and_safety Health" [ref=e80] [cursor=pointer]:
- - /url: /dashboard/health
- - generic [ref=e81]: health_and_safety
- - generic [ref=e82]: Health
- - link "description Logs" [ref=e83] [cursor=pointer]:
- - /url: /dashboard/logs
- - generic [ref=e84]: description
- - generic [ref=e85]: Logs
- - link "history Audit Log" [ref=e86] [cursor=pointer]:
- - /url: /dashboard/audit
- - generic [ref=e87]: history
- - generic [ref=e88]: Audit Log
- - link "settings Settings" [ref=e89] [cursor=pointer]:
- - /url: /dashboard/settings
- - generic [ref=e90]: settings
- - generic [ref=e91]: Settings
- - generic [ref=e92]:
- - paragraph [ref=e93]: Help
- - link "menu_book Docs" [ref=e94] [cursor=pointer]:
- - /url: /docs
- - generic [ref=e95]: menu_book
- - generic [ref=e96]: Docs
- - link "bug_report Issues" [ref=e97] [cursor=pointer]:
- - /url: https://github.com/diegosouzapw/OmniRoute/issues
- - generic [ref=e98]: bug_report
- - generic [ref=e99]: Issues
- - generic [ref=e100]:
- - button "restart_alt Restart" [ref=e101]:
- - generic: restart_alt
- - text: Restart
- - button "power_settings_new Shutdown" [ref=e102]:
- - generic: power_settings_new
- - text: Shutdown
- - main [ref=e103]:
- - generic [ref=e104]:
- - button "menu" [ref=e106]:
- - generic: menu
- - generic [ref=e107]:
- - button "🇺🇸 EN expand_more" [ref=e109]:
- - generic [ref=e110]: 🇺🇸
- - generic [ref=e111]: EN
- - generic: expand_more
- - button "Switch to dark mode" [ref=e112]:
- - generic: dark_mode
- - button "logout" [ref=e113]:
- - generic: logout
- - generic [ref=e115]:
- - navigation "Breadcrumb" [ref=e116]:
- - link "Dashboard" [ref=e118] [cursor=pointer]:
- - /url: /dashboard
- - generic [ref=e119]:
- - generic [ref=e120]: ›
- - generic [ref=e121]: Onboarding
- - generic [ref=e123]:
- - generic [ref=e124]:
- - generic [ref=e126]: "1"
- - generic [ref=e129]: "2"
- - generic [ref=e132]: "3"
- - generic [ref=e135]: "4"
- - generic [ref=e138]: "5"
- - generic [ref=e139]:
- - generic [ref=e140]:
- - generic [ref=e141]: waving_hand
- - heading "Welcome" [level=2] [ref=e142]
- - generic [ref=e144]:
- - paragraph [ref=e145]: OmniRoute is your local AI API proxy. It routes requests to multiple AI providers with load balancing, failover, and usage tracking.
- - generic [ref=e146]:
- - generic [ref=e147]:
- - generic [ref=e148]: swap_horiz
- - text: Multi-Provider
- - generic [ref=e149]:
- - generic [ref=e150]: monitoring
- - text: Usage Tracking
- - generic [ref=e151]:
- - generic [ref=e152]: shield
- - text: API Key Mgmt
- - button "Get Started" [ref=e155] [cursor=pointer]
- - button "Skip wizard entirely" [ref=e157] [cursor=pointer]
- - alert [ref=e158]
diff --git a/.playwright-mcp/page-2026-04-04T08-04-15-839Z.yml b/.playwright-mcp/page-2026-04-04T08-04-15-839Z.yml
deleted file mode 100644
index 88cb680c33..0000000000
--- a/.playwright-mcp/page-2026-04-04T08-04-15-839Z.yml
+++ /dev/null
@@ -1,1011 +0,0 @@
-- generic [active] [ref=e1]:
- - link "Skip to content" [ref=e2] [cursor=pointer]:
- - /url: "#main-content"
- - generic [ref=e3]:
- - complementary [ref=e5]:
- - link "Skip to content" [ref=e6] [cursor=pointer]:
- - /url: "#main-content"
- - link "OmniRoute v3.5.0" [ref=e12] [cursor=pointer]:
- - /url: /dashboard
- - img [ref=e14]
- - generic [ref=e26]:
- - heading "OmniRoute" [level=1] [ref=e27]
- - generic [ref=e28]: v3.5.0
- - navigation "Main navigation" [ref=e29]:
- - generic [ref=e30]:
- - link "home Home" [ref=e31] [cursor=pointer]:
- - /url: /dashboard
- - generic [ref=e32]: home
- - generic [ref=e33]: Home
- - link "api Endpoints" [ref=e34] [cursor=pointer]:
- - /url: /dashboard/endpoint
- - generic [ref=e35]: api
- - generic [ref=e36]: Endpoints
- - link "vpn_key API Manager" [ref=e37] [cursor=pointer]:
- - /url: /dashboard/api-manager
- - generic [ref=e38]: vpn_key
- - generic [ref=e39]: API Manager
- - link "dns Providers" [ref=e40] [cursor=pointer]:
- - /url: /dashboard/providers
- - generic [ref=e41]: dns
- - generic [ref=e42]: Providers
- - link "layers Combos" [ref=e43] [cursor=pointer]:
- - /url: /dashboard/combos
- - generic [ref=e44]: layers
- - generic [ref=e45]: Combos
- - link "auto_awesome Auto Combo" [ref=e46] [cursor=pointer]:
- - /url: /dashboard/auto-combo
- - generic [ref=e47]: auto_awesome
- - generic [ref=e48]: Auto Combo
- - link "account_balance_wallet Costs" [ref=e49] [cursor=pointer]:
- - /url: /dashboard/costs
- - generic [ref=e50]: account_balance_wallet
- - generic [ref=e51]: Costs
- - link "analytics Analytics" [ref=e52] [cursor=pointer]:
- - /url: /dashboard/analytics
- - generic [ref=e53]: analytics
- - generic [ref=e54]: Analytics
- - link "tune Limits & Quotas" [ref=e55] [cursor=pointer]:
- - /url: /dashboard/limits
- - generic [ref=e56]: tune
- - generic [ref=e57]: Limits & Quotas
- - link "cached Cache" [ref=e58] [cursor=pointer]:
- - /url: /dashboard/cache
- - generic [ref=e59]: cached
- - generic [ref=e60]: Cache
- - link "perm_media Media" [ref=e61] [cursor=pointer]:
- - /url: /dashboard/cache/media
- - generic [ref=e62]: perm_media
- - generic [ref=e63]: Media
- - generic [ref=e64]:
- - paragraph [ref=e65]: CLI
- - link "terminal Tools" [ref=e66] [cursor=pointer]:
- - /url: /dashboard/cli-tools
- - generic [ref=e67]: terminal
- - generic [ref=e68]: Tools
- - link "smart_toy Agents" [ref=e69] [cursor=pointer]:
- - /url: /dashboard/agents
- - generic [ref=e70]: smart_toy
- - generic [ref=e71]: Agents
- - link "psychology Memory" [ref=e72] [cursor=pointer]:
- - /url: /dashboard/memory
- - generic [ref=e73]: psychology
- - generic [ref=e74]: Memory
- - link "auto_fix_high Skills" [ref=e75] [cursor=pointer]:
- - /url: /dashboard/skills
- - generic [ref=e76]: auto_fix_high
- - generic [ref=e77]: Skills
- - generic [ref=e78]:
- - paragraph [ref=e79]: System
- - link "health_and_safety Health" [ref=e80] [cursor=pointer]:
- - /url: /dashboard/health
- - generic [ref=e81]: health_and_safety
- - generic [ref=e82]: Health
- - link "description Logs" [ref=e83] [cursor=pointer]:
- - /url: /dashboard/logs
- - generic [ref=e84]: description
- - generic [ref=e85]: Logs
- - link "history Audit Log" [ref=e86] [cursor=pointer]:
- - /url: /dashboard/audit
- - generic [ref=e87]: history
- - generic [ref=e88]: Audit Log
- - link "settings Settings" [ref=e89] [cursor=pointer]:
- - /url: /dashboard/settings
- - generic [ref=e90]: settings
- - generic [ref=e91]: Settings
- - generic [ref=e92]:
- - paragraph [ref=e93]: Help
- - link "menu_book Docs" [ref=e94] [cursor=pointer]:
- - /url: /docs
- - generic [ref=e95]: menu_book
- - generic [ref=e96]: Docs
- - link "bug_report Issues" [ref=e97] [cursor=pointer]:
- - /url: https://github.com/diegosouzapw/OmniRoute/issues
- - generic [ref=e98]: bug_report
- - generic [ref=e99]: Issues
- - generic [ref=e100]:
- - button "restart_alt Restart" [ref=e101]:
- - generic: restart_alt
- - text: Restart
- - button "power_settings_new Shutdown" [ref=e102]:
- - generic: power_settings_new
- - text: Shutdown
- - main [ref=e103]:
- - generic [ref=e104]:
- - button "menu" [ref=e106]:
- - generic: menu
- - generic [ref=e107]:
- - button "🇺🇸 EN expand_more" [ref=e109]:
- - generic [ref=e110]: 🇺🇸
- - generic [ref=e111]: EN
- - generic: expand_more
- - button "Switch to dark mode" [ref=e112]:
- - generic: dark_mode
- - button "logout" [ref=e113]:
- - generic: logout
- - generic [ref=e159]:
- - generic [ref=e161]:
- - generic [ref=e162]:
- - generic [ref=e163]:
- - heading "Quick Start" [level=2] [ref=e164]
- - paragraph [ref=e165]: Get up and running in 4 steps. Connect providers, route models, monitor everything.
- - link "menu_book Full Docs" [ref=e166] [cursor=pointer]:
- - /url: /docs
- - generic [ref=e167]: menu_book
- - text: Full Docs
- - list [ref=e168]:
- - listitem [ref=e169]:
- - generic [ref=e171]: key
- - generic [ref=e172]:
- - text: 1. Create API key
- - paragraph [ref=e173]:
- - text: Go to
- - link "Endpoint" [ref=e174] [cursor=pointer]:
- - /url: /dashboard/endpoint
- - text: "-> Registered Keys. Generate one key per environment."
- - listitem [ref=e175]:
- - generic [ref=e177]: dns
- - generic [ref=e178]:
- - text: 2. Connect providers
- - paragraph [ref=e179]:
- - text: Add accounts in
- - link "Providers" [ref=e180] [cursor=pointer]:
- - /url: /dashboard/providers
- - text: . Supports OAuth, API Key, and free tiers.
- - listitem [ref=e181]:
- - generic [ref=e183]: link
- - generic [ref=e184]:
- - text: 3. Point your client
- - paragraph [ref=e185]: Set base URL to http://localhost:20130/v1 in your IDE or API client.
- - listitem [ref=e186]:
- - generic [ref=e188]: analytics
- - generic [ref=e189]:
- - text: 4. Monitor & optimize
- - paragraph [ref=e190]:
- - text: Track tokens, cost and errors in
- - link "Request Logs" [ref=e191] [cursor=pointer]:
- - /url: /dashboard/usage
- - text: and
- - link "Analytics" [ref=e192] [cursor=pointer]:
- - /url: /dashboard/analytics
- - text: .
- - generic [ref=e193]:
- - link "menu_book Documentation" [ref=e194] [cursor=pointer]:
- - /url: /docs
- - generic [ref=e195]: menu_book
- - text: Documentation
- - link "dns Providers" [ref=e196] [cursor=pointer]:
- - /url: /dashboard/providers
- - generic [ref=e197]: dns
- - text: Providers
- - link "layers Combos" [ref=e198] [cursor=pointer]:
- - /url: /dashboard/combos
- - generic [ref=e199]: layers
- - text: Combos
- - link "analytics Analytics" [ref=e200] [cursor=pointer]:
- - /url: /dashboard/analytics
- - generic [ref=e201]: analytics
- - text: Analytics
- - link "health_and_safety Health Monitor" [ref=e202] [cursor=pointer]:
- - /url: /dashboard/health
- - generic [ref=e203]: health_and_safety
- - text: Health Monitor
- - link "terminal CLI Tools" [ref=e204] [cursor=pointer]:
- - /url: /dashboard/cli-tools
- - generic [ref=e205]: terminal
- - text: CLI Tools
- - link "bug_report Report issue" [ref=e206] [cursor=pointer]:
- - /url: https://github.com/diegosouzapw/OmniRoute/issues
- - generic [ref=e207]: bug_report
- - text: Report issue
- - generic [ref=e208]:
- - generic [ref=e209]:
- - generic [ref=e210]:
- - heading "Providers Overview" [level=2] [ref=e211]
- - paragraph [ref=e212]: 0 configured of 72 available providers
- - generic [ref=e213]:
- - generic [ref=e214]:
- - generic [ref=e215]: Free
- - generic [ref=e217]: OAuth
- - generic [ref=e219]: API Key
- - link "settings Manage" [ref=e221] [cursor=pointer]:
- - /url: /dashboard/providers
- - generic [ref=e222]: settings
- - text: Manage
- - generic [ref=e223]:
- - button "Qoder AI Free Not configured 0 models" [ref=e224] [cursor=pointer]:
- - generic [ref=e225]:
- - img [ref=e228]
- - generic [ref=e231]:
- - generic [ref=e232]:
- - paragraph [ref=e233]: Qoder AI
- - generic "Free" [ref=e234]
- - paragraph [ref=e235]: Not configured
- - generic [ref=e236]:
- - paragraph [ref=e237]: "0"
- - paragraph [ref=e238]: models
- - button "AlibabaCloud Qwen Code Free Not configured 0 models" [ref=e239] [cursor=pointer]:
- - generic [ref=e240]:
- - img "AlibabaCloud" [ref=e243]
- - generic [ref=e246]:
- - generic [ref=e247]:
- - paragraph [ref=e248]: Qwen Code
- - generic "Free" [ref=e249]
- - paragraph [ref=e250]: Not configured
- - generic [ref=e251]:
- - paragraph [ref=e252]: "0"
- - paragraph [ref=e253]: models
- - button "gemini-cli Gemini CLI Free Not configured 0 models" [ref=e254] [cursor=pointer]:
- - generic [ref=e255]:
- - img "gemini-cli" [ref=e258]
- - generic [ref=e259]:
- - generic [ref=e260]:
- - paragraph [ref=e261]: Gemini CLI
- - generic "Free" [ref=e262]
- - paragraph [ref=e263]: Not configured
- - generic [ref=e264]:
- - paragraph [ref=e265]: "0"
- - paragraph [ref=e266]: models
- - button "kiro Kiro AI Free Not configured 0 models" [ref=e267] [cursor=pointer]:
- - generic [ref=e268]:
- - img "kiro" [ref=e271]
- - generic [ref=e272]:
- - generic [ref=e273]:
- - paragraph [ref=e274]: Kiro AI
- - generic "Free" [ref=e275]
- - paragraph [ref=e276]: Not configured
- - generic [ref=e277]:
- - paragraph [ref=e278]: "0"
- - paragraph [ref=e279]: models
- - button "Anthropic Claude Code OAuth Not configured 0 models" [ref=e280] [cursor=pointer]:
- - generic [ref=e281]:
- - img "Anthropic" [ref=e284]
- - generic [ref=e286]:
- - generic [ref=e287]:
- - paragraph [ref=e288]: Claude Code
- - generic "OAuth" [ref=e289]
- - paragraph [ref=e290]: Not configured
- - generic [ref=e291]:
- - paragraph [ref=e292]: "0"
- - paragraph [ref=e293]: models
- - button "antigravity Antigravity OAuth Not configured 0 models" [ref=e294] [cursor=pointer]:
- - generic [ref=e295]:
- - img "antigravity" [ref=e298]
- - generic [ref=e299]:
- - generic [ref=e300]:
- - paragraph [ref=e301]: Antigravity
- - generic "OAuth" [ref=e302]
- - paragraph [ref=e303]: Not configured
- - generic [ref=e304]:
- - paragraph [ref=e305]: "0"
- - paragraph [ref=e306]: models
- - button "OpenAI OpenAI Codex OAuth Not configured 0 models" [ref=e307] [cursor=pointer]:
- - generic [ref=e308]:
- - img "OpenAI" [ref=e311]
- - generic [ref=e313]:
- - generic [ref=e314]:
- - paragraph [ref=e315]: OpenAI Codex
- - generic "OAuth" [ref=e316]
- - paragraph [ref=e317]: Not configured
- - generic [ref=e318]:
- - paragraph [ref=e319]: "0"
- - paragraph [ref=e320]: models
- - button "github GitHub Copilot OAuth Not configured 0 models" [ref=e321] [cursor=pointer]:
- - generic [ref=e322]:
- - img "github" [ref=e325]
- - generic [ref=e326]:
- - generic [ref=e327]:
- - paragraph [ref=e328]: GitHub Copilot
- - generic "OAuth" [ref=e329]
- - paragraph [ref=e330]: Not configured
- - generic [ref=e331]:
- - paragraph [ref=e332]: "0"
- - paragraph [ref=e333]: models
- - button "cursor Cursor IDE OAuth Not configured 0 models" [ref=e334] [cursor=pointer]:
- - generic [ref=e335]:
- - img "cursor" [ref=e338]
- - generic [ref=e339]:
- - generic [ref=e340]:
- - paragraph [ref=e341]: Cursor IDE
- - generic "OAuth" [ref=e342]
- - paragraph [ref=e343]: Not configured
- - generic [ref=e344]:
- - paragraph [ref=e345]: "0"
- - paragraph [ref=e346]: models
- - button "kimi-coding Kimi Coding OAuth Not configured 0 models" [ref=e347] [cursor=pointer]:
- - generic [ref=e348]:
- - img "kimi-coding" [ref=e351]
- - generic [ref=e352]:
- - generic [ref=e353]:
- - paragraph [ref=e354]: Kimi Coding
- - generic "OAuth" [ref=e355]
- - paragraph [ref=e356]: Not configured
- - generic [ref=e357]:
- - paragraph [ref=e358]: "0"
- - paragraph [ref=e359]: models
- - button "kilocode Kilo Code OAuth Not configured 0 models" [ref=e360] [cursor=pointer]:
- - generic [ref=e361]:
- - img "kilocode" [ref=e364]
- - generic [ref=e365]:
- - generic [ref=e366]:
- - paragraph [ref=e367]: Kilo Code
- - generic "OAuth" [ref=e368]
- - paragraph [ref=e369]: Not configured
- - generic [ref=e370]:
- - paragraph [ref=e371]: "0"
- - paragraph [ref=e372]: models
- - button "cline Cline OAuth Not configured 0 models" [ref=e373] [cursor=pointer]:
- - generic [ref=e374]:
- - img "cline" [ref=e377]
- - generic [ref=e378]:
- - generic [ref=e379]:
- - paragraph [ref=e380]: Cline
- - generic "OAuth" [ref=e381]
- - paragraph [ref=e382]: Not configured
- - generic [ref=e383]:
- - paragraph [ref=e384]: "0"
- - paragraph [ref=e385]: models
- - button "OpenRouter OpenRouter API Key Not configured 0 models" [ref=e386] [cursor=pointer]:
- - generic [ref=e387]:
- - img "OpenRouter" [ref=e390]
- - generic [ref=e392]:
- - generic [ref=e393]:
- - paragraph [ref=e394]: OpenRouter
- - generic "API Key" [ref=e395]
- - paragraph [ref=e396]: Not configured
- - generic [ref=e397]:
- - paragraph [ref=e398]: "0"
- - paragraph [ref=e399]: models
- - button "glm GLM Coding API Key Not configured 0 models" [ref=e400] [cursor=pointer]:
- - generic [ref=e401]:
- - img "glm" [ref=e404]
- - generic [ref=e405]:
- - generic [ref=e406]:
- - paragraph [ref=e407]: GLM Coding
- - generic "API Key" [ref=e408]
- - paragraph [ref=e409]: Not configured
- - generic [ref=e410]:
- - paragraph [ref=e411]: "0"
- - paragraph [ref=e412]: models
- - button "bailian-coding-plan Alibaba Coding Plan API Key Not configured 0 models" [ref=e413] [cursor=pointer]:
- - generic [ref=e414]:
- - img "bailian-coding-plan" [ref=e417]
- - generic [ref=e418]:
- - generic [ref=e419]:
- - paragraph [ref=e420]: Alibaba Coding Plan
- - generic "API Key" [ref=e421]
- - paragraph [ref=e422]: Not configured
- - generic [ref=e423]:
- - paragraph [ref=e424]: "0"
- - paragraph [ref=e425]: models
- - button "MoonshotAI Kimi API Key Not configured 0 models" [ref=e426] [cursor=pointer]:
- - generic [ref=e427]:
- - img "MoonshotAI" [ref=e430]
- - generic [ref=e432]:
- - generic [ref=e433]:
- - paragraph [ref=e434]: Kimi
- - generic "API Key" [ref=e435]
- - paragraph [ref=e436]: Not configured
- - generic [ref=e437]:
- - paragraph [ref=e438]: "0"
- - paragraph [ref=e439]: models
- - button "kimi-coding-apikey Kimi Coding (API Key) API Key Not configured 0 models" [ref=e440] [cursor=pointer]:
- - generic [ref=e441]:
- - img "kimi-coding-apikey" [ref=e444]
- - generic [ref=e445]:
- - generic [ref=e446]:
- - paragraph [ref=e447]: Kimi Coding (API Key)
- - generic "API Key" [ref=e448]
- - paragraph [ref=e449]: Not configured
- - generic [ref=e450]:
- - paragraph [ref=e451]: "0"
- - paragraph [ref=e452]: models
- - button "Minimax Minimax Coding API Key Not configured 0 models" [ref=e453] [cursor=pointer]:
- - generic [ref=e454]:
- - img "Minimax" [ref=e457]
- - generic [ref=e459]:
- - generic [ref=e460]:
- - paragraph [ref=e461]: Minimax Coding
- - generic "API Key" [ref=e462]
- - paragraph [ref=e463]: Not configured
- - generic [ref=e464]:
- - paragraph [ref=e465]: "0"
- - paragraph [ref=e466]: models
- - button "minimax-cn Minimax (China) API Key Not configured 0 models" [ref=e467] [cursor=pointer]:
- - generic [ref=e468]:
- - img "minimax-cn" [ref=e471]
- - generic [ref=e472]:
- - generic [ref=e473]:
- - paragraph [ref=e474]: Minimax (China)
- - generic "API Key" [ref=e475]
- - paragraph [ref=e476]: Not configured
- - generic [ref=e477]:
- - paragraph [ref=e478]: "0"
- - paragraph [ref=e479]: models
- - button "alicode Alibaba API Key Not configured 0 models" [ref=e480] [cursor=pointer]:
- - generic [ref=e481]:
- - img "alicode" [ref=e484]
- - generic [ref=e485]:
- - generic [ref=e486]:
- - paragraph [ref=e487]: Alibaba
- - generic "API Key" [ref=e488]
- - paragraph [ref=e489]: Not configured
- - generic [ref=e490]:
- - paragraph [ref=e491]: "0"
- - paragraph [ref=e492]: models
- - button "alicode-intl Alibaba Intl API Key Not configured 0 models" [ref=e493] [cursor=pointer]:
- - generic [ref=e494]:
- - img "alicode-intl" [ref=e497]
- - generic [ref=e498]:
- - generic [ref=e499]:
- - paragraph [ref=e500]: Alibaba Intl
- - generic "API Key" [ref=e501]
- - paragraph [ref=e502]: Not configured
- - generic [ref=e503]:
- - paragraph [ref=e504]: "0"
- - paragraph [ref=e505]: models
- - button "OpenAI OpenAI API Key Not configured 0 models" [ref=e506] [cursor=pointer]:
- - generic [ref=e507]:
- - img "OpenAI" [ref=e510]
- - generic [ref=e512]:
- - generic [ref=e513]:
- - paragraph [ref=e514]: OpenAI
- - generic "API Key" [ref=e515]
- - paragraph [ref=e516]: Not configured
- - generic [ref=e517]:
- - paragraph [ref=e518]: "0"
- - paragraph [ref=e519]: models
- - button "Anthropic Anthropic API Key Not configured 0 models" [ref=e520] [cursor=pointer]:
- - generic [ref=e521]:
- - img "Anthropic" [ref=e524]
- - generic [ref=e526]:
- - generic [ref=e527]:
- - paragraph [ref=e528]: Anthropic
- - generic "API Key" [ref=e529]
- - paragraph [ref=e530]: Not configured
- - generic [ref=e531]:
- - paragraph [ref=e532]: "0"
- - paragraph [ref=e533]: models
- - button "Google Gemini (Google AI Studio) API Key Not configured 0 models" [ref=e534] [cursor=pointer]:
- - generic [ref=e535]:
- - img "Google" [ref=e538]
- - generic [ref=e543]:
- - generic [ref=e544]:
- - paragraph [ref=e545]: Gemini (Google AI Studio)
- - generic "API Key" [ref=e546]
- - paragraph [ref=e547]: Not configured
- - generic [ref=e548]:
- - paragraph [ref=e549]: "0"
- - paragraph [ref=e550]: models
- - button "DeepSeek DeepSeek API Key Not configured 0 models" [ref=e551] [cursor=pointer]:
- - generic [ref=e552]:
- - img "DeepSeek" [ref=e555]
- - generic [ref=e557]:
- - generic [ref=e558]:
- - paragraph [ref=e559]: DeepSeek
- - generic "API Key" [ref=e560]
- - paragraph [ref=e561]: Not configured
- - generic [ref=e562]:
- - paragraph [ref=e563]: "0"
- - paragraph [ref=e564]: models
- - button "Groq Groq API Key Not configured 0 models" [ref=e565] [cursor=pointer]:
- - generic [ref=e566]:
- - img "Groq" [ref=e569]
- - generic [ref=e571]:
- - generic [ref=e572]:
- - paragraph [ref=e573]: Groq
- - generic "API Key" [ref=e574]
- - paragraph [ref=e575]: Not configured
- - generic [ref=e576]:
- - paragraph [ref=e577]: "0"
- - paragraph [ref=e578]: models
- - button "Blackbox AI API Key Not configured 0 models" [ref=e579] [cursor=pointer]:
- - generic [ref=e580]:
- - img [ref=e583]:
- - img [ref=e584]
- - generic [ref=e589]:
- - generic [ref=e590]:
- - paragraph [ref=e591]: Blackbox AI
- - generic "API Key" [ref=e592]
- - paragraph [ref=e593]: Not configured
- - generic [ref=e594]:
- - paragraph [ref=e595]: "0"
- - paragraph [ref=e596]: models
- - button "Grok xAI (Grok) API Key Not configured 0 models" [ref=e597] [cursor=pointer]:
- - generic [ref=e598]:
- - img "Grok" [ref=e601]
- - generic [ref=e603]:
- - generic [ref=e604]:
- - paragraph [ref=e605]: xAI (Grok)
- - generic "API Key" [ref=e606]
- - paragraph [ref=e607]: Not configured
- - generic [ref=e608]:
- - paragraph [ref=e609]: "0"
- - paragraph [ref=e610]: models
- - button "Mistral Mistral API Key Not configured 0 models" [ref=e611] [cursor=pointer]:
- - generic [ref=e612]:
- - img "Mistral" [ref=e615]
- - generic [ref=e621]:
- - generic [ref=e622]:
- - paragraph [ref=e623]: Mistral
- - generic "API Key" [ref=e624]
- - paragraph [ref=e625]: Not configured
- - generic [ref=e626]:
- - paragraph [ref=e627]: "0"
- - paragraph [ref=e628]: models
- - button "Perplexity Perplexity API Key Not configured 0 models" [ref=e629] [cursor=pointer]:
- - generic [ref=e630]:
- - img "Perplexity" [ref=e633]
- - generic [ref=e635]:
- - generic [ref=e636]:
- - paragraph [ref=e637]: Perplexity
- - generic "API Key" [ref=e638]
- - paragraph [ref=e639]: Not configured
- - generic [ref=e640]:
- - paragraph [ref=e641]: "0"
- - paragraph [ref=e642]: models
- - button "together.ai Together AI API Key Not configured 0 models" [ref=e643] [cursor=pointer]:
- - generic [ref=e644]:
- - img "together.ai" [ref=e647]
- - generic [ref=e651]:
- - generic [ref=e652]:
- - paragraph [ref=e653]: Together AI
- - generic "API Key" [ref=e654]
- - paragraph [ref=e655]: Not configured
- - generic [ref=e656]:
- - paragraph [ref=e657]: "0"
- - paragraph [ref=e658]: models
- - button "Fireworks AI API Key Not configured 0 models" [ref=e659] [cursor=pointer]:
- - generic [ref=e660]:
- - img [ref=e663]:
- - img [ref=e664]
- - generic [ref=e669]:
- - generic [ref=e670]:
- - paragraph [ref=e671]: Fireworks AI
- - generic "API Key" [ref=e672]
- - paragraph [ref=e673]: Not configured
- - generic [ref=e674]:
- - paragraph [ref=e675]: "0"
- - paragraph [ref=e676]: models
- - button "Cerebras Cerebras API Key Not configured 0 models" [ref=e677] [cursor=pointer]:
- - generic [ref=e678]:
- - img "Cerebras" [ref=e681]
- - generic [ref=e684]:
- - generic [ref=e685]:
- - paragraph [ref=e686]: Cerebras
- - generic "API Key" [ref=e687]
- - paragraph [ref=e688]: Not configured
- - generic [ref=e689]:
- - paragraph [ref=e690]: "0"
- - paragraph [ref=e691]: models
- - button "Cohere Cohere API Key Not configured 0 models" [ref=e692] [cursor=pointer]:
- - generic [ref=e693]:
- - img "Cohere" [ref=e696]
- - generic [ref=e700]:
- - generic [ref=e701]:
- - paragraph [ref=e702]: Cohere
- - generic "API Key" [ref=e703]
- - paragraph [ref=e704]: Not configured
- - generic [ref=e705]:
- - paragraph [ref=e706]: "0"
- - paragraph [ref=e707]: models
- - button "Nvidia NVIDIA NIM API Key Not configured 0 models" [ref=e708] [cursor=pointer]:
- - generic [ref=e709]:
- - img "Nvidia" [ref=e712]
- - generic [ref=e714]:
- - generic [ref=e715]:
- - paragraph [ref=e716]: NVIDIA NIM
- - generic "API Key" [ref=e717]
- - paragraph [ref=e718]: Not configured
- - generic [ref=e719]:
- - paragraph [ref=e720]: "0"
- - paragraph [ref=e721]: models
- - button "nebius Nebius AI API Key Not configured 0 models" [ref=e722] [cursor=pointer]:
- - generic [ref=e723]:
- - img "nebius" [ref=e726]
- - generic [ref=e727]:
- - generic [ref=e728]:
- - paragraph [ref=e729]: Nebius AI
- - generic "API Key" [ref=e730]
- - paragraph [ref=e731]: Not configured
- - generic [ref=e732]:
- - paragraph [ref=e733]: "0"
- - paragraph [ref=e734]: models
- - button "siliconflow SiliconFlow API Key Not configured 0 models" [ref=e735] [cursor=pointer]:
- - generic [ref=e736]:
- - img "siliconflow" [ref=e739]
- - generic [ref=e740]:
- - generic [ref=e741]:
- - paragraph [ref=e742]: SiliconFlow
- - generic "API Key" [ref=e743]
- - paragraph [ref=e744]: Not configured
- - generic [ref=e745]:
- - paragraph [ref=e746]: "0"
- - paragraph [ref=e747]: models
- - button "hyperbolic Hyperbolic API Key Not configured 0 models" [ref=e748] [cursor=pointer]:
- - generic [ref=e749]:
- - img "hyperbolic" [ref=e752]
- - generic [ref=e753]:
- - generic [ref=e754]:
- - paragraph [ref=e755]: Hyperbolic
- - generic "API Key" [ref=e756]
- - paragraph [ref=e757]: Not configured
- - generic [ref=e758]:
- - paragraph [ref=e759]: "0"
- - paragraph [ref=e760]: models
- - button "deepgram Deepgram API Key Not configured 0 models" [ref=e761] [cursor=pointer]:
- - generic [ref=e762]:
- - img "deepgram" [ref=e765]
- - generic [ref=e766]:
- - generic [ref=e767]:
- - paragraph [ref=e768]: Deepgram
- - generic "API Key" [ref=e769]
- - paragraph [ref=e770]: Not configured
- - generic [ref=e771]:
- - paragraph [ref=e772]: "0"
- - paragraph [ref=e773]: models
- - button "assemblyai AssemblyAI API Key Not configured 0 models" [ref=e774] [cursor=pointer]:
- - generic [ref=e775]:
- - img "assemblyai" [ref=e778]
- - generic [ref=e779]:
- - generic [ref=e780]:
- - paragraph [ref=e781]: AssemblyAI
- - generic "API Key" [ref=e782]
- - paragraph [ref=e783]: Not configured
- - generic [ref=e784]:
- - paragraph [ref=e785]: "0"
- - paragraph [ref=e786]: models
- - button "nanobanana NanoBanana API Key Not configured 0 models" [ref=e787] [cursor=pointer]:
- - generic [ref=e788]:
- - img "nanobanana" [ref=e791]
- - generic [ref=e792]:
- - generic [ref=e793]:
- - paragraph [ref=e794]: NanoBanana
- - generic "API Key" [ref=e795]
- - paragraph [ref=e796]: Not configured
- - generic [ref=e797]:
- - paragraph [ref=e798]: "0"
- - paragraph [ref=e799]: models
- - button "ollama-cloud Ollama Cloud API Key Not configured 0 models" [ref=e800] [cursor=pointer]:
- - generic [ref=e801]:
- - img "ollama-cloud" [ref=e804]
- - generic [ref=e805]:
- - generic [ref=e806]:
- - paragraph [ref=e807]: Ollama Cloud
- - generic "API Key" [ref=e808]
- - paragraph [ref=e809]: Not configured
- - generic [ref=e810]:
- - paragraph [ref=e811]: "0"
- - paragraph [ref=e812]: models
- - button "elevenlabs ElevenLabs API Key Not configured 0 models" [ref=e813] [cursor=pointer]:
- - generic [ref=e814]:
- - img "elevenlabs" [ref=e817]
- - generic [ref=e818]:
- - generic [ref=e819]:
- - paragraph [ref=e820]: ElevenLabs
- - generic "API Key" [ref=e821]
- - paragraph [ref=e822]: Not configured
- - generic [ref=e823]:
- - paragraph [ref=e824]: "0"
- - paragraph [ref=e825]: models
- - button "cartesia Cartesia API Key Not configured 0 models" [ref=e826] [cursor=pointer]:
- - generic [ref=e827]:
- - img "cartesia" [ref=e830]
- - generic [ref=e831]:
- - generic [ref=e832]:
- - paragraph [ref=e833]: Cartesia
- - generic "API Key" [ref=e834]
- - paragraph [ref=e835]: Not configured
- - generic [ref=e836]:
- - paragraph [ref=e837]: "0"
- - paragraph [ref=e838]: models
- - button "playht PlayHT API Key Not configured 0 models" [ref=e839] [cursor=pointer]:
- - generic [ref=e840]:
- - img "playht" [ref=e843]
- - generic [ref=e844]:
- - generic [ref=e845]:
- - paragraph [ref=e846]: PlayHT
- - generic "API Key" [ref=e847]
- - paragraph [ref=e848]: Not configured
- - generic [ref=e849]:
- - paragraph [ref=e850]: "0"
- - paragraph [ref=e851]: models
- - button "inworld Inworld API Key Not configured 0 models" [ref=e852] [cursor=pointer]:
- - generic [ref=e853]:
- - img "inworld" [ref=e856]
- - generic [ref=e857]:
- - generic [ref=e858]:
- - paragraph [ref=e859]: Inworld
- - generic "API Key" [ref=e860]
- - paragraph [ref=e861]: Not configured
- - generic [ref=e862]:
- - paragraph [ref=e863]: "0"
- - paragraph [ref=e864]: models
- - button "sdwebui SD WebUI API Key Not configured 0 models" [ref=e865] [cursor=pointer]:
- - generic [ref=e866]:
- - img "sdwebui" [ref=e869]
- - generic [ref=e870]:
- - generic [ref=e871]:
- - paragraph [ref=e872]: SD WebUI
- - generic "API Key" [ref=e873]
- - paragraph [ref=e874]: Not configured
- - generic [ref=e875]:
- - paragraph [ref=e876]: "0"
- - paragraph [ref=e877]: models
- - button "comfyui ComfyUI API Key Not configured 0 models" [ref=e878] [cursor=pointer]:
- - generic [ref=e879]:
- - img "comfyui" [ref=e882]
- - generic [ref=e883]:
- - generic [ref=e884]:
- - paragraph [ref=e885]: ComfyUI
- - generic "API Key" [ref=e886]
- - paragraph [ref=e887]: Not configured
- - generic [ref=e888]:
- - paragraph [ref=e889]: "0"
- - paragraph [ref=e890]: models
- - button "HuggingFace HuggingFace API Key Not configured 0 models" [ref=e891] [cursor=pointer]:
- - generic [ref=e892]:
- - img "HuggingFace" [ref=e895]
- - generic [ref=e902]:
- - generic [ref=e903]:
- - paragraph [ref=e904]: HuggingFace
- - generic "API Key" [ref=e905]
- - paragraph [ref=e906]: Not configured
- - generic [ref=e907]:
- - paragraph [ref=e908]: "0"
- - paragraph [ref=e909]: models
- - button "synthetic Synthetic API Key Not configured 0 models" [ref=e910] [cursor=pointer]:
- - generic [ref=e911]:
- - img "synthetic" [ref=e914]
- - generic [ref=e915]:
- - generic [ref=e916]:
- - paragraph [ref=e917]: Synthetic
- - generic "API Key" [ref=e918]
- - paragraph [ref=e919]: Not configured
- - generic [ref=e920]:
- - paragraph [ref=e921]: "0"
- - paragraph [ref=e922]: models
- - button "kilo-gateway Kilo Gateway API Key Not configured 0 models" [ref=e923] [cursor=pointer]:
- - generic [ref=e924]:
- - img "kilo-gateway" [ref=e927]
- - generic [ref=e928]:
- - generic [ref=e929]:
- - paragraph [ref=e930]: Kilo Gateway
- - generic "API Key" [ref=e931]
- - paragraph [ref=e932]: Not configured
- - generic [ref=e933]:
- - paragraph [ref=e934]: "0"
- - paragraph [ref=e935]: models
- - button "vertex Vertex AI API Key Not configured 0 models" [ref=e936] [cursor=pointer]:
- - generic [ref=e937]:
- - img "vertex" [ref=e940]
- - generic [ref=e941]:
- - generic [ref=e942]:
- - paragraph [ref=e943]: Vertex AI
- - generic "API Key" [ref=e944]
- - paragraph [ref=e945]: Not configured
- - generic [ref=e946]:
- - paragraph [ref=e947]: "0"
- - paragraph [ref=e948]: models
- - button "zai Z.AI API Key Not configured 0 models" [ref=e949] [cursor=pointer]:
- - generic [ref=e950]:
- - img "zai" [ref=e953]
- - generic [ref=e954]:
- - generic [ref=e955]:
- - paragraph [ref=e956]: Z.AI
- - generic "API Key" [ref=e957]
- - paragraph [ref=e958]: Not configured
- - generic [ref=e959]:
- - paragraph [ref=e960]: "0"
- - paragraph [ref=e961]: models
- - button "perplexity-search Perplexity Search API Key Not configured 0 models" [ref=e962] [cursor=pointer]:
- - generic [ref=e963]:
- - img "perplexity-search" [ref=e966]
- - generic [ref=e967]:
- - generic [ref=e968]:
- - paragraph [ref=e969]: Perplexity Search
- - generic "API Key" [ref=e970]
- - paragraph [ref=e971]: Not configured
- - generic [ref=e972]:
- - paragraph [ref=e973]: "0"
- - paragraph [ref=e974]: models
- - button "serper-search Serper Search API Key Not configured 0 models" [ref=e975] [cursor=pointer]:
- - generic [ref=e976]:
- - img "serper-search" [ref=e979]
- - generic [ref=e980]:
- - generic [ref=e981]:
- - paragraph [ref=e982]: Serper Search
- - generic "API Key" [ref=e983]
- - paragraph [ref=e984]: Not configured
- - generic [ref=e985]:
- - paragraph [ref=e986]: "0"
- - paragraph [ref=e987]: models
- - button "brave-search Brave Search API Key Not configured 0 models" [ref=e988] [cursor=pointer]:
- - generic [ref=e989]:
- - img "brave-search" [ref=e992]
- - generic [ref=e993]:
- - generic [ref=e994]:
- - paragraph [ref=e995]: Brave Search
- - generic "API Key" [ref=e996]
- - paragraph [ref=e997]: Not configured
- - generic [ref=e998]:
- - paragraph [ref=e999]: "0"
- - paragraph [ref=e1000]: models
- - button "exa-search Exa Search API Key Not configured 0 models" [ref=e1001] [cursor=pointer]:
- - generic [ref=e1002]:
- - img "exa-search" [ref=e1005]
- - generic [ref=e1006]:
- - generic [ref=e1007]:
- - paragraph [ref=e1008]: Exa Search
- - generic "API Key" [ref=e1009]
- - paragraph [ref=e1010]: Not configured
- - generic [ref=e1011]:
- - paragraph [ref=e1012]: "0"
- - paragraph [ref=e1013]: models
- - button "tavily-search Tavily Search API Key Not configured 0 models" [ref=e1014] [cursor=pointer]:
- - generic [ref=e1015]:
- - img "tavily-search" [ref=e1018]
- - generic [ref=e1019]:
- - generic [ref=e1020]:
- - paragraph [ref=e1021]: Tavily Search
- - generic "API Key" [ref=e1022]
- - paragraph [ref=e1023]: Not configured
- - generic [ref=e1024]:
- - paragraph [ref=e1025]: "0"
- - paragraph [ref=e1026]: models
- - button "opencode-zen OpenCode Zen API Key Not configured 0 models" [ref=e1027] [cursor=pointer]:
- - generic [ref=e1028]:
- - img "opencode-zen" [ref=e1031]
- - generic [ref=e1032]:
- - generic [ref=e1033]:
- - paragraph [ref=e1034]: OpenCode Zen
- - generic "API Key" [ref=e1035]
- - paragraph [ref=e1036]: Not configured
- - generic [ref=e1037]:
- - paragraph [ref=e1038]: "0"
- - paragraph [ref=e1039]: models
- - button "opencode-go OpenCode Go API Key Not configured 0 models" [ref=e1040] [cursor=pointer]:
- - generic [ref=e1041]:
- - img "opencode-go" [ref=e1044]
- - generic [ref=e1045]:
- - generic [ref=e1046]:
- - paragraph [ref=e1047]: OpenCode Go
- - generic "API Key" [ref=e1048]
- - paragraph [ref=e1049]: Not configured
- - generic [ref=e1050]:
- - paragraph [ref=e1051]: "0"
- - paragraph [ref=e1052]: models
- - button "AlibabaCloud Alibaba Cloud (DashScope) API Key Not configured 0 models" [ref=e1053] [cursor=pointer]:
- - generic [ref=e1054]:
- - img "AlibabaCloud" [ref=e1057]
- - generic [ref=e1060]:
- - generic [ref=e1061]:
- - paragraph [ref=e1062]: Alibaba Cloud (DashScope)
- - generic "API Key" [ref=e1063]
- - paragraph [ref=e1064]: Not configured
- - generic [ref=e1065]:
- - paragraph [ref=e1066]: "0"
- - paragraph [ref=e1067]: models
- - button "longcat LongCat AI API Key Not configured 0 models" [ref=e1068] [cursor=pointer]:
- - generic [ref=e1069]:
- - img "longcat" [ref=e1072]
- - generic [ref=e1073]:
- - generic [ref=e1074]:
- - paragraph [ref=e1075]: LongCat AI
- - generic "API Key" [ref=e1076]
- - paragraph [ref=e1077]: Not configured
- - generic [ref=e1078]:
- - paragraph [ref=e1079]: "0"
- - paragraph [ref=e1080]: models
- - button "Pollinations AI API Key Not configured 0 models" [ref=e1081] [cursor=pointer]:
- - generic [ref=e1082]:
- - img [ref=e1085]:
- - img [ref=e1086]
- - generic [ref=e1091]:
- - generic [ref=e1092]:
- - paragraph [ref=e1093]: Pollinations AI
- - generic "API Key" [ref=e1094]
- - paragraph [ref=e1095]: Not configured
- - generic [ref=e1096]:
- - paragraph [ref=e1097]: "0"
- - paragraph [ref=e1098]: models
- - button "puter Puter AI API Key Not configured 0 models" [ref=e1099] [cursor=pointer]:
- - generic [ref=e1100]:
- - img "puter" [ref=e1103]
- - generic [ref=e1104]:
- - generic [ref=e1105]:
- - paragraph [ref=e1106]: Puter AI
- - generic "API Key" [ref=e1107]
- - paragraph [ref=e1108]: Not configured
- - generic [ref=e1109]:
- - paragraph [ref=e1110]: "0"
- - paragraph [ref=e1111]: models
- - button "Cloudflare Cloudflare Workers AI API Key Not configured 0 models" [ref=e1112] [cursor=pointer]:
- - generic [ref=e1113]:
- - img "Cloudflare" [ref=e1116]
- - generic [ref=e1119]:
- - generic [ref=e1120]:
- - paragraph [ref=e1121]: Cloudflare Workers AI
- - generic "API Key" [ref=e1122]
- - paragraph [ref=e1123]: Not configured
- - generic [ref=e1124]:
- - paragraph [ref=e1125]: "0"
- - paragraph [ref=e1126]: models
- - button "scaleway Scaleway AI API Key Not configured 0 models" [ref=e1127] [cursor=pointer]:
- - generic [ref=e1128]:
- - img "scaleway" [ref=e1131]
- - generic [ref=e1132]:
- - generic [ref=e1133]:
- - paragraph [ref=e1134]: Scaleway AI
- - generic "API Key" [ref=e1135]
- - paragraph [ref=e1136]: Not configured
- - generic [ref=e1137]:
- - paragraph [ref=e1138]: "0"
- - paragraph [ref=e1139]: models
- - button "aimlapi AI/ML API API Key Not configured 0 models" [ref=e1140] [cursor=pointer]:
- - generic [ref=e1141]:
- - img "aimlapi" [ref=e1144]
- - generic [ref=e1145]:
- - generic [ref=e1146]:
- - paragraph [ref=e1147]: AI/ML API
- - generic "API Key" [ref=e1148]
- - paragraph [ref=e1149]: Not configured
- - generic [ref=e1150]:
- - paragraph [ref=e1151]: "0"
- - paragraph [ref=e1152]: models
- - button "Novita AI API Key Not configured 0 models" [ref=e1153] [cursor=pointer]:
- - generic [ref=e1154]:
- - img [ref=e1157]
- - generic [ref=e1160]:
- - generic [ref=e1161]:
- - paragraph [ref=e1162]: Novita AI
- - generic "API Key" [ref=e1163]
- - paragraph [ref=e1164]: Not configured
- - generic [ref=e1165]:
- - paragraph [ref=e1166]: "0"
- - paragraph [ref=e1167]: models
- - button "PiAPI API Key Not configured 0 models" [ref=e1168] [cursor=pointer]:
- - generic [ref=e1169]:
- - img [ref=e1172]
- - generic [ref=e1175]:
- - generic [ref=e1176]:
- - paragraph [ref=e1177]: PiAPI
- - generic "API Key" [ref=e1178]
- - paragraph [ref=e1179]: Not configured
- - generic [ref=e1180]:
- - paragraph [ref=e1181]: "0"
- - paragraph [ref=e1182]: models
- - button "GoAPI API Key Not configured 0 models" [ref=e1183] [cursor=pointer]:
- - generic [ref=e1184]:
- - img [ref=e1187]
- - generic [ref=e1190]:
- - generic [ref=e1191]:
- - paragraph [ref=e1192]: GoAPI
- - generic "API Key" [ref=e1193]
- - paragraph [ref=e1194]: Not configured
- - generic [ref=e1195]:
- - paragraph [ref=e1196]: "0"
- - paragraph [ref=e1197]: models
- - button "LaoZhang AI API Key Not configured 0 models" [ref=e1198] [cursor=pointer]:
- - generic [ref=e1199]:
- - img [ref=e1202]
- - generic [ref=e1205]:
- - generic [ref=e1206]:
- - paragraph [ref=e1207]: LaoZhang AI
- - generic "API Key" [ref=e1208]
- - paragraph [ref=e1209]: Not configured
- - generic [ref=e1210]:
- - paragraph [ref=e1211]: "0"
- - paragraph [ref=e1212]: models
- - button "CLIProxyAPI API Key Not configured 0 models" [ref=e1213] [cursor=pointer]:
- - generic [ref=e1214]:
- - img [ref=e1217]
- - generic [ref=e1220]:
- - generic [ref=e1221]:
- - paragraph [ref=e1222]: CLIProxyAPI
- - generic "API Key" [ref=e1223]
- - paragraph [ref=e1224]: Not configured
- - generic [ref=e1225]:
- - paragraph [ref=e1226]: "0"
- - paragraph [ref=e1227]: models
- - alert [ref=e158]
diff --git a/.playwright-mcp/page-2026-04-04T08-04-46-313Z.yml b/.playwright-mcp/page-2026-04-04T08-04-46-313Z.yml
deleted file mode 100644
index c83adc1401..0000000000
--- a/.playwright-mcp/page-2026-04-04T08-04-46-313Z.yml
+++ /dev/null
@@ -1,132 +0,0 @@
-- generic [active] [ref=e1]:
- - link "Skip to content" [ref=e2] [cursor=pointer]:
- - /url: "#main-content"
- - generic [ref=e3]:
- - complementary [ref=e5]:
- - link "Skip to content" [ref=e6] [cursor=pointer]:
- - /url: "#main-content"
- - link "OmniRoute v3.5.0" [ref=e12] [cursor=pointer]:
- - /url: /dashboard
- - img [ref=e14]
- - generic [ref=e26]:
- - heading "OmniRoute" [level=1] [ref=e27]
- - generic [ref=e28]: v3.5.0
- - navigation "Main navigation" [ref=e29]:
- - generic [ref=e30]:
- - link "home Home" [ref=e31] [cursor=pointer]:
- - /url: /dashboard
- - generic [ref=e32]: home
- - generic [ref=e33]: Home
- - link "api Endpoints" [ref=e34] [cursor=pointer]:
- - /url: /dashboard/endpoint
- - generic [ref=e35]: api
- - generic [ref=e36]: Endpoints
- - link "vpn_key API Manager" [ref=e37] [cursor=pointer]:
- - /url: /dashboard/api-manager
- - generic [ref=e38]: vpn_key
- - generic [ref=e39]: API Manager
- - link "dns Providers" [ref=e40] [cursor=pointer]:
- - /url: /dashboard/providers
- - generic [ref=e41]: dns
- - generic [ref=e42]: Providers
- - link "layers Combos" [ref=e43] [cursor=pointer]:
- - /url: /dashboard/combos
- - generic [ref=e44]: layers
- - generic [ref=e45]: Combos
- - link "auto_awesome Auto Combo" [ref=e46] [cursor=pointer]:
- - /url: /dashboard/auto-combo
- - generic [ref=e47]: auto_awesome
- - generic [ref=e48]: Auto Combo
- - link "account_balance_wallet Costs" [ref=e49] [cursor=pointer]:
- - /url: /dashboard/costs
- - generic [ref=e50]: account_balance_wallet
- - generic [ref=e51]: Costs
- - link "analytics Analytics" [ref=e52] [cursor=pointer]:
- - /url: /dashboard/analytics
- - generic [ref=e53]: analytics
- - generic [ref=e54]: Analytics
- - link "tune Limits & Quotas" [ref=e55] [cursor=pointer]:
- - /url: /dashboard/limits
- - generic [ref=e56]: tune
- - generic [ref=e57]: Limits & Quotas
- - link "cached Cache" [ref=e58] [cursor=pointer]:
- - /url: /dashboard/cache
- - generic [ref=e59]: cached
- - generic [ref=e60]: Cache
- - link "perm_media Media" [ref=e61] [cursor=pointer]:
- - /url: /dashboard/cache/media
- - generic [ref=e62]: perm_media
- - generic [ref=e63]: Media
- - generic [ref=e64]:
- - paragraph [ref=e65]: CLI
- - link "terminal Tools" [ref=e66] [cursor=pointer]:
- - /url: /dashboard/cli-tools
- - generic [ref=e67]: terminal
- - generic [ref=e68]: Tools
- - link "smart_toy Agents" [ref=e69] [cursor=pointer]:
- - /url: /dashboard/agents
- - generic [ref=e70]: smart_toy
- - generic [ref=e71]: Agents
- - link "psychology Memory" [ref=e72] [cursor=pointer]:
- - /url: /dashboard/memory
- - generic [ref=e73]: psychology
- - generic [ref=e74]: Memory
- - link "auto_fix_high Skills" [ref=e75] [cursor=pointer]:
- - /url: /dashboard/skills
- - generic [ref=e76]: auto_fix_high
- - generic [ref=e77]: Skills
- - generic [ref=e78]:
- - paragraph [ref=e79]: System
- - link "health_and_safety Health" [ref=e80] [cursor=pointer]:
- - /url: /dashboard/health
- - generic [ref=e81]: health_and_safety
- - generic [ref=e82]: Health
- - link "description Logs" [ref=e83] [cursor=pointer]:
- - /url: /dashboard/logs
- - generic [ref=e84]: description
- - generic [ref=e85]: Logs
- - link "history Audit Log" [ref=e86] [cursor=pointer]:
- - /url: /dashboard/audit
- - generic [ref=e87]: history
- - generic [ref=e88]: Audit Log
- - link "settings Settings" [ref=e89] [cursor=pointer]:
- - /url: /dashboard/settings
- - generic [ref=e90]: settings
- - generic [ref=e91]: Settings
- - generic [ref=e92]:
- - paragraph [ref=e93]: Help
- - link "menu_book Docs" [ref=e94] [cursor=pointer]:
- - /url: /docs
- - generic [ref=e95]: menu_book
- - generic [ref=e96]: Docs
- - link "bug_report Issues" [ref=e97] [cursor=pointer]:
- - /url: https://github.com/diegosouzapw/OmniRoute/issues
- - generic [ref=e98]: bug_report
- - generic [ref=e99]: Issues
- - generic [ref=e100]:
- - button "restart_alt Restart" [ref=e101]:
- - generic: restart_alt
- - text: Restart
- - button "power_settings_new Shutdown" [ref=e102]:
- - generic: power_settings_new
- - text: Shutdown
- - main [ref=e103]:
- - generic [ref=e104]:
- - button "menu" [ref=e106]:
- - generic: menu
- - generic [ref=e107]:
- - button "🇺🇸 EN expand_more" [ref=e109]:
- - generic [ref=e110]: 🇺🇸
- - generic [ref=e111]: EN
- - generic: expand_more
- - button "Switch to dark mode" [ref=e112]:
- - generic: dark_mode
- - button "logout" [ref=e113]:
- - generic: logout
- - navigation "Breadcrumb" [ref=e116]:
- - link "Dashboard" [ref=e118] [cursor=pointer]:
- - /url: /dashboard
- - generic [ref=e119]:
- - generic [ref=e120]: ›
- - generic [ref=e121]: Providers
- - alert [ref=e135]
diff --git a/AGENTS.md b/AGENTS.md
index 1e20037ee5..e2d957d5be 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -135,7 +135,8 @@ All persistence uses SQLite through domain-specific modules:
`core.ts`, `providers.ts`, `models.ts`, `combos.ts`, `apiKeys.ts`, `settings.ts`,
`backup.ts`, `proxies.ts`, `prompts.ts`, `webhooks.ts`, `detailedLogs.ts`,
`domainState.ts`, `registeredKeys.ts`, `quotaSnapshots.ts`, `modelComboMappings.ts`,
-`cliToolState.ts`, `encryption.ts`, `readCache.ts`, `secrets.ts`, `stateReset.ts`.
+`cliToolState.ts`, `encryption.ts`, `readCache.ts`, `secrets.ts`, `stateReset.ts`,
+`contextHandoffs.ts`.
Schema migrations live in `db/migrations/` and run via `migrationRunner.ts`.
`src/lib/localDb.ts` is a **re-export layer only** — never add logic there.
@@ -189,7 +190,7 @@ Includes request/response translators with helpers for image handling.
`autoCombo/`, `intentClassifier.ts`, `taskAwareRouter.ts`, `thinkingBudget.ts`,
`contextManager.ts`, `modelDeprecation.ts`, `modelFamilyFallback.ts`,
`emergencyFallback.ts`, `workflowFSM.ts`, `backgroundTaskDetector.ts`, `ipFilter.ts`,
-`signatureCache.ts`, `volumeDetector.ts`, and more.
+`signatureCache.ts`, `volumeDetector.ts`, `contextHandoff.ts`, and more.
### Domain Layer (`src/domain/`)
diff --git a/CHANGELOG.md b/CHANGELOG.md
index c0f5e46243..0ade82dd8f 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -4,6 +4,37 @@
---
+## [3.5.5] — 2026-04-08
+
+### ✨ New Features
+
+- **Node.js 24 Compatibility Warning:** Added a proactive version incompatibility warning on the login page to guide users to the stable Node.js 22 LTS, preventing native sqlite binding crashes.
+- **Context Relay Combo Strategy:** Added the new `context-relay` combo strategy with priority-style routing, structured handoff summary generation once quota usage reaches the warning threshold, and handoff injection after the next real account switch.
+- **Global Context Relay Defaults:** Added global Settings defaults plus combo-level configuration for `handoffThreshold`, `handoffModel`, and `handoffProviders`, so new or unconfigured combos can inherit the feature consistently.
+
+### 🐛 Bug Fixes
+
+- **Proxy Connection Healthchecks:** Applied proxy resolution per connection in the sweeping loop (`tokenHealthCheck.ts`) and global provider validation sweeps, resolving Node 22 bypass and improving proxy stability (#1051, #1056, #1061).
+- **Security Vulnerability Remediation:** Resolved multiple CodeQL scanning alerts including SSRF in model sync, insecure randomness in web crypto (`generateSessionId`), and incomplete URL sanitization.
+- **Context Relay Typing & Synchronization:** Reverted out-of-scope test breakages and resolved `handoffProvider` and response `input` extraction payload typing.
+- **Legacy OpenAI-Compatible Responses Routing:** Fixed legacy/imported OpenAI-compatible providers (for example `openai-compatible-sp-openai`) incorrectly routing Chat Completions traffic to `/chat/completions` when the real provider node was configured as `apiType: "responses"`. OmniRoute now treats `providerSpecificData.apiType` as authoritative across routing, executors, and translator tools, avoiding false empty-content failures during combo/provider smoke tests (#1069).
+- **Gemini PDF Attachment Integration:** Fixed payload generation and format for parsing `inline_data` and generic base64 sources for deep Gemini PDF routing (#993, #1021).
+- **Vercel AI SDK Fallbacks:** Mapped `max_output_tokens` to `max_tokens` for strict OpenAI-compatible providers, resolving errors from standard AI agents and frameworks (#994).
+- **External Auth & UI Reliability:** Handled null `state` failures in Cline OAuth exchange (#1016), added 3rd-party 400 error patterns to combo fallback (#1024), and resolved desktop sidebar layout and popover overflows (#1039, #1001).
+- **Context Relay In-Flight Deduplication:** Prevented duplicate handoff generation for the same session/combo while an earlier summary request is still in flight.
+- **Context Relay Provider Gating:** Aligned runtime behavior with configuration so explicit `handoffProviders` exclusions, including an empty array, now disable handoff generation as expected.
+
+### 🛠️ Maintenance & Dependabot
+
+- **Updated Sub-dependencies:** Bumped `hono` to `4.12.12` and `@hono/node-server` to `1.19.13` to patch critical security gaps (#1063, #1064, #1067, #1068).
+
+### 📚 Documentation
+
+- **Documentation Synchronization:** Updated system documentation (README, Architecture, Features, Tools, Troubleshooting) and synced `i18n` configurations to match the v3.5.5 context relay patterns and proxy troubleshooting steps.
+- **Context Relay Delivery Notes:** Documented the current architecture, runtime flow, and Codex-focused scope in the feature docs, changelog, and agent guidance.
+
+---
+
## [3.5.4] — 2026-04-07
### ✨ New Features
diff --git a/README.md b/README.md
index b392f0cf73..1e09af1a31 100644
--- a/README.md
+++ b/README.md
@@ -243,7 +243,7 @@ Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Eve
- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention
- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI
- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next
-- **Custom Combos** — Customizable fallback chains with 9 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random)
+- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**)
- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard
@@ -1304,7 +1304,17 @@ Then in `/dashboard/media` → **Transcription** tab: upload any audio or video
## 💡 Key Features
-OmniRoute v2.0 is built as an operational platform, not just a relay proxy.
+OmniRoute v3.5 is built as an operational platform, not just a relay proxy.
+
+### 🆕 New — v3.5.5 Highlights (Apr 2026)
+
+| Feature | What It Does |
+| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation |
+| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs |
+| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions |
+| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection |
+| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts |
### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
@@ -1356,7 +1366,8 @@ OmniRoute v2.0 is built as an operational platform, not just a relay proxy.
| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions |
| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection |
| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry |
-| 🎨 **Custom Combos** | 9 balancing strategies + fallback chain control |
+| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control |
+| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session |
| 🌐 **Wildcard Router** | `provider/*` dynamic routing |
| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits |
| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety |
@@ -2183,9 +2194,10 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux
| ---------------------------------------------- | --------------------------------------------------- |
| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment |
| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples |
-| [MCP Server](open-sse/mcp-server/README.md) | 16 MCP tools, IDE configs, Python/TS/Go clients |
+| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients |
| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt |
| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing |
+| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation |
| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions |
| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals |
| [Contributing](CONTRIBUTING.md) | Development setup and guidelines |
diff --git a/audit_results.json b/audit_results.json
deleted file mode 100644
index 7efd50c459..0000000000
--- a/audit_results.json
+++ /dev/null
@@ -1,203 +0,0 @@
-{
- "auditReportVersion": 2,
- "vulnerabilities": {
- "next": {
- "name": "next",
- "severity": "high",
- "isDirect": true,
- "via": [
- {
- "source": 1112592,
- "name": "next",
- "dependency": "next",
- "title": "Next.js self-hosted applications vulnerable to DoS via Image Optimizer remotePatterns configuration",
- "url": "https://github.com/advisories/GHSA-9g9p-9gw9-jx7f",
- "severity": "moderate",
- "cwe": ["CWE-400", "CWE-770"],
- "cvss": {
- "score": 5.9,
- "vectorString": "CVSS:3.1/AV:N/AC:H/PR:N/UI:N/S:U/C:N/I:N/A:H"
- },
- "range": ">=15.6.0-canary.0 <16.1.5"
- },
- {
- "source": 1112646,
- "name": "next",
- "dependency": "next",
- "title": "Next.js HTTP request deserialization can lead to DoS when using insecure React Server Components",
- "url": "https://github.com/advisories/GHSA-h25m-26qc-wcjf",
- "severity": "high",
- "cwe": ["CWE-400", "CWE-502"],
- "cvss": {
- "score": 7.5,
- "vectorString": "CVSS:3.1/AV:N/AC:L/PR:N/UI:N/S:U/C:N/I:N/A:H"
- },
- "range": ">=16.0.0-beta.0 <16.0.11"
- },
- {
- "source": 1112990,
- "name": "next",
- "dependency": "next",
- "title": "Next.js has Unbounded Memory Consumption via PPR Resume Endpoint ",
- "url": "https://github.com/advisories/GHSA-5f7q-jpqc-wp7h",
- "severity": "moderate",
- "cwe": ["CWE-400", "CWE-409", "CWE-770"],
- "cvss": {
- "score": 5.9,
- "vectorString": "CVSS:3.1/AV:N/AC:H/PR:N/UI:N/S:U/C:N/I:N/A:H"
- },
- "range": ">=16.0.0-beta.0 <16.1.5"
- },
- {
- "source": 1114898,
- "name": "next",
- "dependency": "next",
- "title": "Next.js: HTTP request smuggling in rewrites",
- "url": "https://github.com/advisories/GHSA-ggv3-7p47-pfv8",
- "severity": "moderate",
- "cwe": ["CWE-444"],
- "cvss": {
- "score": 0,
- "vectorString": null
- },
- "range": ">=16.0.0-beta.0 <16.1.7"
- },
- {
- "source": 1114941,
- "name": "next",
- "dependency": "next",
- "title": "Next.js: Unbounded next/image disk cache growth can exhaust storage",
- "url": "https://github.com/advisories/GHSA-3x4c-7xq6-9pq8",
- "severity": "moderate",
- "cwe": ["CWE-400"],
- "cvss": {
- "score": 0,
- "vectorString": null
- },
- "range": ">=16.0.0-beta.0 <16.1.7"
- },
- {
- "source": 1114942,
- "name": "next",
- "dependency": "next",
- "title": "Next.js: Unbounded postponed resume buffering can lead to DoS",
- "url": "https://github.com/advisories/GHSA-h27x-g6w4-24gq",
- "severity": "moderate",
- "cwe": ["CWE-770"],
- "cvss": {
- "score": 0,
- "vectorString": null
- },
- "range": ">=16.0.1 <16.1.7"
- },
- {
- "source": 1114943,
- "name": "next",
- "dependency": "next",
- "title": "Next.js: null origin can bypass Server Actions CSRF checks",
- "url": "https://github.com/advisories/GHSA-mq59-m269-xvcx",
- "severity": "moderate",
- "cwe": ["CWE-352"],
- "cvss": {
- "score": 0,
- "vectorString": null
- },
- "range": ">=16.0.1 <16.1.7"
- },
- {
- "source": 1115360,
- "name": "next",
- "dependency": "next",
- "title": "Next.js: null origin can bypass dev HMR websocket CSRF checks",
- "url": "https://github.com/advisories/GHSA-jcc7-9wpm-mj36",
- "severity": "low",
- "cwe": ["CWE-1385"],
- "cvss": {
- "score": 0,
- "vectorString": null
- },
- "range": ">=16.0.1 <16.1.7"
- }
- ],
- "effects": [],
- "range": "15.6.0-canary.0 - 16.1.6",
- "nodes": ["node_modules/next"],
- "fixAvailable": {
- "name": "next",
- "version": "16.2.2",
- "isSemVerMajor": false
- }
- },
- "vite": {
- "name": "vite",
- "severity": "high",
- "isDirect": false,
- "via": [
- {
- "source": 1116007,
- "name": "vite",
- "dependency": "vite",
- "title": "Vite Vulnerable to Path Traversal in Optimized Deps `.map` Handling",
- "url": "https://github.com/advisories/GHSA-4w7w-66w2-5vf9",
- "severity": "moderate",
- "cwe": ["CWE-22", "CWE-200"],
- "cvss": {
- "score": 0,
- "vectorString": null
- },
- "range": ">=8.0.0 <=8.0.4"
- },
- {
- "source": 1116009,
- "name": "vite",
- "dependency": "vite",
- "title": "Vite: `server.fs.deny` bypassed with queries",
- "url": "https://github.com/advisories/GHSA-v2wj-q39q-566r",
- "severity": "high",
- "cwe": ["CWE-180", "CWE-284"],
- "cvss": {
- "score": 0,
- "vectorString": null
- },
- "range": ">=8.0.0 <=8.0.4"
- },
- {
- "source": 1116012,
- "name": "vite",
- "dependency": "vite",
- "title": "Vite Vulnerable to Arbitrary File Read via Vite Dev Server WebSocket",
- "url": "https://github.com/advisories/GHSA-p9ff-h696-f583",
- "severity": "high",
- "cwe": ["CWE-200", "CWE-306"],
- "cvss": {
- "score": 0,
- "vectorString": null
- },
- "range": ">=8.0.0 <=8.0.4"
- }
- ],
- "effects": [],
- "range": "8.0.0 - 8.0.4",
- "nodes": ["node_modules/vite"],
- "fixAvailable": true
- }
- },
- "metadata": {
- "vulnerabilities": {
- "info": 0,
- "low": 0,
- "moderate": 0,
- "high": 2,
- "critical": 0,
- "total": 2
- },
- "dependencies": {
- "prod": 407,
- "dev": 485,
- "optional": 154,
- "peer": 480,
- "peerOptional": 0,
- "total": 1455
- }
- }
-}
diff --git a/dbsetup.js b/dbsetup.js
deleted file mode 100755
index deb6abef78..0000000000
--- a/dbsetup.js
+++ /dev/null
@@ -1,23 +0,0 @@
-#!/usr/bin/env node
-
-import { spawn } from 'node:child_process'
-
-const env = { ...process.env }
-
-await exec('npx next build --experimental-build-mode generate')
-
-// launch application
-await exec(process.argv.slice(2).join(' '))
-
-function exec(command) {
- const child = spawn(command, { shell: true, stdio: 'inherit', env })
- return new Promise((resolve, reject) => {
- child.on('exit', code => {
- if (code === 0) {
- resolve()
- } else {
- reject(new Error(`${command} failed rc=${code}`))
- }
- })
- })
-}
diff --git a/debug.log b/debug.log
deleted file mode 100644
index fbf94da8ed..0000000000
--- a/debug.log
+++ /dev/null
@@ -1,1775 +0,0 @@
-TAP version 13
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [CREDENTIALS] No external credentials file found, using defaults.
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [SECURITY] API_KEY_SECRET is not set. API key CRC validation is disabled.
-# [DB] Added api_keys.allowed_connections column
-# [DB] Added api_keys.auto_resolve column
-# [DB] Added api_keys.is_active column
-# [DB] Added api_keys.access_schedule column
-# [DB] Added api_keys.max_requests_per_day column
-# [DB] Added api_keys.max_requests_per_minute column
-# [DB] Added api_keys.max_sessions column
-# [23:37:04.405] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:04.406] [34mDEBUG[39m: [36m[sse] API Key: sk-m...d970[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:04.415] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# 🛡️ [RATE-LIMIT] Loaded 0 explicit + 1 auto-enabled + 0 custom rpm/tpm protection(s)
-# [23:37:04.419] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:04.419] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:04.420] [32mINFO[39m: [36m[sse] Using openai account: d92ece8b...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:04.436] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:04.443] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [32m[20:37] 📊 [USAGE] OPENAI | in=4 | out=2 | account=d92ece8b...[0m
-# [23:37:04.511] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline handles OpenAI passthrough with valid API key auth
-ok 1 - chat pipeline handles OpenAI passthrough with valid API key auth
- ---
- duration_ms: 240.858479
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [23:37:04.649] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | claude/claude-3-5-sonnet-20241022 | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | claude/claude-3-5-sonnet-20241022 | 1 msgs"
-# [23:37:04.650] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [32m[20:37] 📊 [USAGE] CLAUDE | in=10 | out=4 | account=c5a7d3b2...[0m
-# [23:37:04.660] [32mINFO[39m: [36m[sse] Provider: claude, Model: claude-3-5-sonnet-20241022[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:04.661] [34mDEBUG[39m: [36m[sse] claude | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:04.661] [34mDEBUG[39m: [36m[sse] claude | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:04.661] [32mINFO[39m: [36m[sse] Using claude account: c5a7d3b2...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:04.663] [34mDEBUG[39m: [36m[sse] openai → claude | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:04.666] [34mDEBUG[39m: [36m[sse] CLAUDE | claude-3-5-sonnet-20241022 | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:04.672] [34mDEBUG[39m: [36m[sse] Stored response for claude-3-5-sonnet-20241022 (14 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline translates OpenAI requests to Claude and returns OpenAI-shaped responses
-ok 2 - chat pipeline translates OpenAI requests to Claude and returns OpenAI-shaped responses
- ---
- duration_ms: 153.337563
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [23:37:04.800] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | gemini/gemini-2.5-flash | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | gemini/gemini-2.5-flash | 1 msgs"
-# [23:37:04.800] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [32m[20:37] 📊 [USAGE] GEMINI | in=5 | out=7 | account=51322177...[0m
-# [23:37:04.803] [32mINFO[39m: [36m[sse] Provider: gemini, Model: gemini-2.5-flash[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:04.804] [34mDEBUG[39m: [36m[sse] gemini | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:04.804] [34mDEBUG[39m: [36m[sse] gemini | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:04.804] [32mINFO[39m: [36m[sse] Using gemini account: 51322177...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:04.806] [34mDEBUG[39m: [36m[sse] openai → gemini | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:04.809] [34mDEBUG[39m: [36m[sse] GEMINI | gemini-2.5-flash | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:04.814] [34mDEBUG[39m: [36m[sse] Stored response for gemini-2.5-flash (12 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline translates OpenAI requests to Gemini and returns OpenAI-shaped responses
-ok 3 - chat pipeline translates OpenAI requests to Gemini and returns OpenAI-shaped responses
- ---
- duration_ms: 141.730313
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [23:37:04.937] [32mINFO[39m: [36m[sse] 📥 POST /v1/messages | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/messages | openai/gpt-4o-mini | 1 msgs"
-# [23:37:04.937] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [32m[20:37] 📊 [USAGE] OPENAI | in=4 | out=2 | account=2c7bd399...[0m
-# [23:37:04.939] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:04.940] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:04.940] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:04.941] [32mINFO[39m: [36m[sse] Using openai account: 2c7bd399...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:04.942] [34mDEBUG[39m: [36m[sse] claude → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:04.944] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 2 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:04.947] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline translates Claude-format requests into OpenAI upstream and back to Claude
-ok 4 - chat pipeline translates Claude-format requests into OpenAI upstream and back to Claude
- ---
- duration_ms: 133.00343
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [23:37:05.072] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | claude/claude-sonnet-4-6 | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | claude/claude-sonnet-4-6 | 1 msgs"
-# [23:37:05.072] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [20:37:05] 📊 [32m[USAGE] CLAUDE | in=0 | out=3 | account=5cfec526...[0m
-# [20:37:05] 🌊 [STREAM] CLAUDE | claude-sonnet-4-6 | 25ms | complete
-# [23:37:05.074] [32mINFO[39m: [36m[sse] Provider: claude, Model: claude-sonnet-4-6[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:05.076] [34mDEBUG[39m: [36m[sse] claude | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.076] [34mDEBUG[39m: [36m[sse] claude | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.076] [32mINFO[39m: [36m[sse] Using claude account: 5cfec526...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.077] [34mDEBUG[39m: [36m[sse] openai → claude | stream=true[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:05.078] [34mDEBUG[39m: [36m[sse] CLAUDE | claude-sonnet-4-6 | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:05.080] [34mDEBUG[39m: [36m[sse] Translation mode: claude → openai[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "STREAM"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline converts Claude SSE streams into OpenAI SSE output
-ok 5 - chat pipeline converts Claude SSE streams into OpenAI SSE output
- ---
- duration_ms: 153.585214
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [DB] Added api_keys.allowed_connections column
-# [DB] Added api_keys.auto_resolve column
-# [DB] Added api_keys.is_active column
-# [DB] Added api_keys.access_schedule column
-# [DB] Added api_keys.max_requests_per_day column
-# [DB] Added api_keys.max_requests_per_minute column
-# [DB] Added api_keys.max_sessions column
-# [23:37:05.229] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:05.230] [34mDEBUG[39m: [36m[sse] API Key: does...xist[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.239] [33mWARN[39m: [36m[sse] API key not found or invalid (must be created in API Manager)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.240] [33mWARN[39m: [36m[sse] Invalid JSON body[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline rejects invalid API keys and malformed JSON bodies
-ok 6 - chat pipeline rejects invalid API keys and malformed JSON bodies
- ---
- duration_ms: 137.638739
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [23:37:05.338] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:05.338] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.338] [33mWARN[39m: [36m[sse] Missing API key while REQUIRE_API_KEY=true[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline rejects requests without a bearer key when strict API key mode is enabled
-ok 7 - chat pipeline rejects requests without a bearer key when strict API key mode is enabled
- ---
- duration_ms: 126.083827
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [23:37:05.464] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | undefined | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | undefined | 1 msgs"
-# [23:37:05.464] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.464] [33mWARN[39m: [36m[sse] Missing model[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline returns 400 when the model field is omitted
-ok 8 - chat pipeline returns 400 when the model field is omitted
- ---
- duration_ms: 121.287276
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [23:37:05.622] [34mDEBUG[39m: [36m[sse] Accept: text/event-stream header → overriding stream=true (body had no stream field)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "STREAM"
-# [23:37:05.622] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:05.623] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [20:37:05] 📊 [32m[USAGE] OPENAI | in=2062 | out=6 | account=2cb832e8...[0m [33m(estimated)[0m
-# [20:37:05] 🌊 [STREAM] OPENAI | gpt-4o-mini | 8ms | complete
-# [23:37:05.626] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:05.628] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.628] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.628] [32mINFO[39m: [36m[sse] Using openai account: 2cb832e8...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.630] [34mDEBUG[39m: [36m[sse] openai → openai | stream=true[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:05.631] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:05.632] [34mDEBUG[39m: [36m[sse] Standard passthrough mode[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "STREAM"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline treats Accept text/event-stream as streaming mode and returns a session header
-ok 9 - chat pipeline treats Accept text/event-stream as streaming mode and returns a session header
- ---
- duration_ms: 150.696127
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [32m[20:37] 📊 [USAGE] OPENAI | in=4 | out=2 | account=ae22a379...[0m
-# [23:37:05.767] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | local-router | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | local-router | 1 msgs"
-# [23:37:05.767] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.768] [32mINFO[39m: [36m[sse] Combo "local-router" [priority] with 1 models[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [23:37:05.770] [32mINFO[39m: [36m[sse] priority with nested resolution: 1 total models[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [23:37:05.772] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.772] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.772] [32mINFO[39m: [36m[sse] Trying model 1/1: openai/gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [23:37:05.773] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:05.774] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.774] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.774] [32mINFO[39m: [36m[sse] Using openai account: ae22a379...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.776] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:05.776] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:05.780] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [23:37:05.783] [32mINFO[39m: [36m[sse] Model openai/gpt-4o-mini succeeded (12ms, 0 fallbacks)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline supports local mode without Authorization on explicit combos
-ok 10 - chat pipeline supports local mode without Authorization on explicit combos
- ---
- duration_ms: 144.671066
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [DB] Added api_keys.allowed_connections column
-# [DB] Added api_keys.auto_resolve column
-# [DB] Added api_keys.is_active column
-# [DB] Added api_keys.access_schedule column
-# [DB] Added api_keys.max_requests_per_day column
-# [DB] Added api_keys.max_requests_per_minute column
-# [DB] Added api_keys.max_sessions column
-# [23:37:05.916] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:05.916] [34mDEBUG[39m: [36m[sse] API Key: sk-m...63cc[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [32m[20:37] 📊 [USAGE] OPENAI | in=4 | out=2 | account=015388a8...[0m
-# [23:37:05.920] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:05.921] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.921] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.921] [32mINFO[39m: [36m[sse] Using openai account: 015388a8...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:05.922] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:05.923] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:05.929] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline honors noLog by redacting persisted call log payloads
-ok 11 - chat pipeline honors noLog by redacting persisted call log payloads
- ---
- duration_ms: 175.46898
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [23:37:06.057] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:06.057] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.097] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:06.098] [34mDEBUG[39m: [36m[sse] openai | total connections: 0, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.098] [34mDEBUG[39m: [36m[sse] openai | all connections (incl inactive): 0[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.098] [33mWARN[39m: [36m[sse] No credentials for openai[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.098] [31mERROR[39m: [36m[sse] No credentials for provider: openai[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline returns current no-credentials contract when no provider connection exists
-ok 12 - chat pipeline returns current no-credentials contract when no provider connection exists
- ---
- duration_ms: 139.59565
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [23:37:06.226] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:06.226] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.229] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:06.229] [33mWARN[39m: [36m[sse] openai/gpt-4o-mini is in cooldown, rejecting request[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AVAILABILITY"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline returns 503 when the requested model is temporarily unavailable
-ok 13 - chat pipeline returns 503 when the requested model is temporarily unavailable
- ---
- duration_ms: 130.400878
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [23:37:06.404] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:06.404] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [31m[ERROR] [500]: provider exploded[0m
-# ❌ openai [500]: [500]: provider exploded
-# [23:37:06.407] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:06.409] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.409] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.409] [32mINFO[39m: [36m[sse] Using openai account: 5fafd15d...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.412] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:06.415] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:06.450] [33mWARN[39m: [36m[sse] Account 5fafd15d... unavailable (500), trying fallback[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.451] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: 5fafd15d-d14e-49d1-8f7a-4c98ffc15b88[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.451] [34mDEBUG[39m: [36m[sse] openai | available: 0/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.451] [34mDEBUG[39m: [36m[sse] → 5fafd15d | excluded rateLimited until 2026-04-06T23:37:09.447Z[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.451] [33mWARN[39m: [36m[sse] openai | all 1 active accounts rate limited (reset after 3s) | lastErrorCode=500.0, lastError=[500]: provider exploded[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.451] [33mWARN[39m: [36m[sse] [openai/gpt-4o-mini] [500]: provider exploded (reset after 3s)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline surfaces upstream 500 responses as structured errors
-ok 14 - chat pipeline surfaces upstream 500 responses as structured errors
- ---
- duration_ms: 221.19967
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [23:37:06.624] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:06.624] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.628] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:06.629] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.629] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.630] [32mINFO[39m: [36m[sse] Using openai account: 520c2927...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.632] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:06.633] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:06.635] [34mDEBUG[39m: [36m[sse] 429 intra-retry 1/2 on https://api.openai.com/v1/chat/completions — waiting 0ms[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "RETRY"
-# [23:37:06.638] [34mDEBUG[39m: [36m[sse] 429 intra-retry 2/2 on https://api.openai.com/v1/chat/completions — waiting 0ms[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "RETRY"
-# [provider] Node 520c2927-8996-4455-aab8-9bd9b2849476 rate limited (429) - Next available at 2026-04-06T23:37:36.643Z
-# [31m[ERROR] [429]: Rate limit exceeded. Your quota will reset after 30s.[0m
-# [23:37:06.649] [32mINFO[39m: [36m[sse] 520c2927 terminal status=credits_exhausted, skipping cooldown overwrite[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.650] [33mWARN[39m: [36m[sse] Account 520c2927... unavailable (429), trying fallback[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.650] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: 520c2927-8996-4455-aab8-9bd9b2849476[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.651] [34mDEBUG[39m: [36m[sse] openai | available: 0/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.651] [34mDEBUG[39m: [36m[sse] → 520c2927 | excluded rateLimited until 2026-04-06T23:37:36.643Z[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.651] [33mWARN[39m: [36m[sse] openai | all 1 active accounts rate limited (reset after 30s) | lastErrorCode=429.0, lastError=Rate limit exceeded. Your quota will reset after 3[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.651] [32mINFO[39m: [36m[sse] openai/gpt-4o-mini marked unavailable — all accounts exhausted (HTTP 429)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AVAILABILITY"
-# [23:37:06.651] [33mWARN[39m: [36m[sse] [openai/gpt-4o-mini] [429]: Rate limit exceeded. Your quota will reset after 30s. (reset after 30s)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline returns 429 with Retry-After when the upstream rate-limits the only account
-ok 15 - chat pipeline returns 429 with Retry-After when the upstream rate-limits the only account
- ---
- duration_ms: 222.318358
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [23:37:06.835] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:06.836] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [31m[ERROR] [504]: upstream timed out[0m
-# ❌ openai [504]: [504]: upstream timed out
-# [23:37:06.838] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:06.839] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.839] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.839] [32mINFO[39m: [36m[sse] Using openai account: f2cfd6ee...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.840] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:06.841] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:06.842] [33mWARN[39m: [36m[sse] Fetch timeout after 600000ms on https://api.openai.com/v1/chat/completions[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "TIMEOUT"
-# [23:37:06.845] [33mWARN[39m: [36m[sse] Account f2cfd6ee... unavailable (504), trying fallback[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.845] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: f2cfd6ee-5788-445d-95ab-a42c4755c98c[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.845] [34mDEBUG[39m: [36m[sse] openai | available: 0/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.845] [34mDEBUG[39m: [36m[sse] → f2cfd6ee | excluded rateLimited until 2026-04-06T23:37:09.844Z[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.845] [33mWARN[39m: [36m[sse] openai | all 1 active accounts rate limited (reset after 3s) | lastErrorCode=504.0, lastError=[504]: upstream timed out[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.845] [33mWARN[39m: [36m[sse] [openai/gpt-4o-mini] [504]: upstream timed out (reset after 3s)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline maps upstream timeouts to 504 responses
-ok 16 - chat pipeline maps upstream timeouts to 504 responses
- ---
- duration_ms: 169.592617
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [DB] Added api_keys.allowed_connections column
-# [DB] Added api_keys.auto_resolve column
-# [DB] Added api_keys.is_active column
-# [DB] Added api_keys.access_schedule column
-# [DB] Added api_keys.max_requests_per_day column
-# [DB] Added api_keys.max_requests_per_minute column
-# [DB] Added api_keys.max_sessions column
-# [23:37:06.978] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:06.978] [34mDEBUG[39m: [36m[sse] API Key: sk-m...b45a[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [32m[20:37] 📊 [USAGE] OPENAI | in=4 | out=2 | account=10055017...[0m
-# [23:37:06.980] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:06.981] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.981] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.982] [32mINFO[39m: [36m[sse] Using openai account: 10055017...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:06.983] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:06.984] [34mDEBUG[39m: [36m[sse] Injected 1 memories for key=b87f32c4-2259-496c-bea6-e3a84ee2ceeb[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "MEMORY"
-# [23:37:06.985] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 2 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:06.991] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline injects memory context before sending the upstream request
-ok 17 - chat pipeline injects memory context before sending the upstream request
- ---
- duration_ms: 146.457203
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [DB] Added api_keys.allowed_connections column
-# [DB] Added api_keys.auto_resolve column
-# [DB] Added api_keys.is_active column
-# [DB] Added api_keys.access_schedule column
-# [DB] Added api_keys.max_requests_per_day column
-# [DB] Added api_keys.max_requests_per_minute column
-# [DB] Added api_keys.max_sessions column
-# [23:37:07.121] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:07.121] [34mDEBUG[39m: [36m[sse] API Key: sk-m...e321[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [32m[20:37] 📊 [USAGE] OPENAI | in=6 | out=4 | account=d2c071f5...[0m
-# [23:37:07.123] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:07.124] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.124] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.124] [32mINFO[39m: [36m[sse] Using openai account: d2c071f5...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.125] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:07.126] [34mDEBUG[39m: [36m[sse] Injected 1 skills[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "SKILLS"
-# [23:37:07.128] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:07.136] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (10 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline injects skills into tools and intercepts tool calls with skill output
-ok 18 - chat pipeline injects skills into tools and intercepts tool calls with skill output
- ---
- duration_ms: 143.886786
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [23:37:07.253] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:07.253] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [31m[ERROR] [500]: first account failed[0m
-# ❌ openai [500]: [500]: first account failed
-# [32m[20:37] 📊 [USAGE] OPENAI | in=4 | out=2 | account=f16327df...[0m
-# [23:37:07.255] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:07.255] [34mDEBUG[39m: [36m[sse] openai | total connections: 2, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.255] [34mDEBUG[39m: [36m[sse] openai | available: 2/2[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.256] [32mINFO[39m: [36m[sse] Using openai account: dce45848...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.257] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:07.257] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:07.262] [33mWARN[39m: [36m[sse] Account dce45848... unavailable (500), trying fallback[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.263] [34mDEBUG[39m: [36m[sse] openai | total connections: 2, excludeId: dce45848-4780-4192-afd1-e53ca9b89c71[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.263] [34mDEBUG[39m: [36m[sse] openai | available: 1/2[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.263] [34mDEBUG[39m: [36m[sse] → dce45848 | excluded rateLimited until 2026-04-06T23:37:10.261Z[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.263] [32mINFO[39m: [36m[sse] Using openai account: f16327df...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.264] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:07.264] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:07.267] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline falls back to the next account after a provider failure
-ok 19 - chat pipeline falls back to the next account after a provider failure
- ---
- duration_ms: 131.562397
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [31m[ERROR] [503]: openai combo miss[0m
-# ❌ openai [503]: [503]: openai combo miss
-# [23:37:07.383] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | combo-fallback | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | combo-fallback | 1 msgs"
-# [23:37:07.383] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.383] [32mINFO[39m: [36m[sse] Combo "combo-fallback" [priority] with 2 models[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [23:37:07.384] [32mINFO[39m: [36m[sse] priority with nested resolution: 2 total models[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [23:37:07.385] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.385] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.385] [32mINFO[39m: [36m[sse] Trying model 1/2: openai/gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [23:37:07.386] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:07.386] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.386] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.386] [32mINFO[39m: [36m[sse] Using openai account: 4c323792...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.387] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:07.388] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:07.395] [33mWARN[39m: [36m[sse] Account 4c323792... unavailable (503), trying fallback[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.395] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: 4c323792-707c-4d82-b957-1d5e8da61b5c[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.395] [34mDEBUG[39m: [36m[sse] openai | available: 0/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.395] [34mDEBUG[39m: [36m[sse] → 4c323792 | excluded rateLimited until 2026-04-06T23:37:10.394Z[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.395] [33mWARN[39m: [36m[sse] openai | all 1 active accounts rate limited (reset after 3s) | lastErrorCode=503.0, lastError=[503]: openai combo miss[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:07.395] [32mINFO[39m: [36m[sse] openai/gpt-4o-mini marked unavailable — all accounts exhausted (HTTP 503)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AVAILABILITY"
-# [23:37:07.395] [33mWARN[39m: [36m[sse] [openai/gpt-4o-mini] [503]: openai combo miss (reset after 3s)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [23:37:07.396] [33mWARN[39m: [36m[sse] Model openai/gpt-4o-mini failed, trying next[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [35mstatus[39m: 503
-# [23:37:07.396] [32mINFO[39m: [36m[sse] Waiting 3000ms before fallback to next model[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [32m[20:37] 📊 [USAGE] CLAUDE | in=10 | out=4 | account=0c1621a2...[0m
-# [23:37:10.400] [34mDEBUG[39m: [36m[sse] claude | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:10.400] [34mDEBUG[39m: [36m[sse] claude | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:10.400] [32mINFO[39m: [36m[sse] Trying model 2/2: claude/claude-3-5-sonnet-20241022[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [23:37:10.401] [32mINFO[39m: [36m[sse] Provider: claude, Model: claude-3-5-sonnet-20241022[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:10.402] [34mDEBUG[39m: [36m[sse] claude | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:10.402] [34mDEBUG[39m: [36m[sse] claude | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:10.402] [32mINFO[39m: [36m[sse] Using claude account: 0c1621a2...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:10.404] [34mDEBUG[39m: [36m[sse] openai → claude | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:10.405] [34mDEBUG[39m: [36m[sse] CLAUDE | claude-3-5-sonnet-20241022 | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:10.409] [34mDEBUG[39m: [36m[sse] Stored response for claude-3-5-sonnet-20241022 (14 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [23:37:10.411] [32mINFO[39m: [36m[sse] Model claude/claude-3-5-sonnet-20241022 succeeded (3027ms, 0 fallbacks)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# Subtest: chat pipeline falls back across combo models when the first provider fails
-ok 20 - chat pipeline falls back across combo models when the first provider fails
- ---
- duration_ms: 3215.387755
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [23:37:10.671] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:10.671] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:10.672] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:37:10.672] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:10.674] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:10.675] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:10.675] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:10.675] [32mINFO[39m: [36m[sse] Using openai account: 32de68a2...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:10.677] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:10.678] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:37:10.679] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:37:10.680] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:10.680] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:10.680] [32mINFO[39m: [36m[sse] Using openai account: 32de68a2...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:37:10.681] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:37:10.682] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [32m[20:37] 📊 [USAGE] OPENAI | in=4 | out=2 | account=32de68a2...[0m
-# [32m[20:37] 📊 [USAGE] OPENAI | in=4 | out=2 | account=32de68a2...[0m
-# [23:37:10.704] [34mDEBUG[39m: [36m[sse] Joined in-flight request hash=f467c6e15f3cb6e2[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "DEDUP"
-# [23:37:10.707] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [23:37:10.709] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-jTKNWy/storage.sqlite
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# Subtest: chat pipeline deduplicates concurrent identical non-stream requests
-ok 21 - chat pipeline deduplicates concurrent identical non-stream requests
- ---
- duration_ms: 227.367687
- type: 'test'
- ...
-# ACTIVE HANDLES: 2
-# Handle 0: Socket
-# Socket _peername: undefined
-# Socket remoteAddress: undefined
-# Handle 1: Socket
-# Socket _peername: undefined
-# Socket remoteAddress: undefined
-# WAITING UNTIL EVERYTHING CLEARS...
-1..21
-# tests 21
-# suites 0
-# pass 21
-# fail 0
-# cancelled 0
-# skipped 0
-# todo 0
-# duration_ms 36094.10465
diff --git a/debug2.log b/debug2.log
deleted file mode 100644
index bac9284c45..0000000000
--- a/debug2.log
+++ /dev/null
@@ -1,339 +0,0 @@
-TAP version 13
-# Subtest: Chat Pipeline — handleSingleModelChat decomposition
- # Subtest: should define resolveModelOrError helper
- ok 1 - should define resolveModelOrError helper
- ---
- duration_ms: 2.870676
- type: 'test'
- ...
- # Subtest: should define checkPipelineGates helper
- ok 2 - should define checkPipelineGates helper
- ---
- duration_ms: 0.539959
- type: 'test'
- ...
- # Subtest: should define executeChatWithBreaker helper
- ok 3 - should define executeChatWithBreaker helper
- ---
- duration_ms: 0.540611
- type: 'test'
- ...
- # Subtest: should keep cost accounting in the core chat pipeline
- ok 4 - should keep cost accounting in the core chat pipeline
- ---
- duration_ms: 0.583888
- type: 'test'
- ...
- # Subtest: handleSingleModelChat should use resolveModelOrError
- ok 5 - handleSingleModelChat should use resolveModelOrError
- ---
- duration_ms: 0.395358
- type: 'test'
- ...
- # Subtest: handleSingleModelChat should use checkPipelineGates
- ok 6 - handleSingleModelChat should use checkPipelineGates
- ---
- duration_ms: 0.725771
- type: 'test'
- ...
- # Subtest: handleSingleModelChat should use executeChatWithBreaker
- ok 7 - handleSingleModelChat should use executeChatWithBreaker
- ---
- duration_ms: 0.384009
- type: 'test'
- ...
- # Subtest: chatCore should record cost for both non-streaming and streaming responses
- ok 8 - chatCore should record cost for both non-streaming and streaming responses
- ---
- duration_ms: 0.717301
- type: 'test'
- ...
- 1..8
-ok 1 - Chat Pipeline — handleSingleModelChat decomposition
- ---
- duration_ms: 12.510313
- type: 'suite'
- ...
-# Subtest: Chat Pipeline — combo fallback support
- # Subtest: should import handleComboChat
- ok 1 - should import handleComboChat
- ---
- duration_ms: 0.683569
- type: 'test'
- ...
- # Subtest: should delegate to handleSingleModelChat for each combo model
- ok 2 - should delegate to handleSingleModelChat for each combo model
- ---
- duration_ms: 0.903143
- type: 'test'
- ...
- # Subtest: should check model availability before attempting combo models
- ok 3 - should check model availability before attempting combo models
- ---
- duration_ms: 0.37494
- type: 'test'
- ...
- 1..3
-ok 2 - Chat Pipeline — combo fallback support
- ---
- duration_ms: 2.526966
- type: 'suite'
- ...
-# Subtest: Chat Pipeline — circuit breaker integration
- # Subtest: should import CircuitBreakerOpenError
- ok 1 - should import CircuitBreakerOpenError
- ---
- duration_ms: 0.500942
- type: 'test'
- ...
- # Subtest: should handle CircuitBreakerOpenError with retry-after
- ok 2 - should handle CircuitBreakerOpenError with retry-after
- ---
- duration_ms: 0.262275
- type: 'test'
- ...
- # Subtest: should reject requests when circuit is open
- ok 3 - should reject requests when circuit is open
- ---
- duration_ms: 0.399071
- type: 'test'
- ...
- 1..3
-ok 3 - Chat Pipeline — circuit breaker integration
- ---
- duration_ms: 1.532005
- type: 'suite'
- ...
-# Subtest: DI Container — container.ts
- # Subtest: should export a container singleton
- ok 1 - should export a container singleton
- ---
- duration_ms: 293.16418
- type: 'test'
- ...
- # Subtest: should register and resolve a custom service
- ok 2 - should register and resolve a custom service
- ---
- duration_ms: 9.047061
- type: 'test'
- ...
- # Subtest: should return cached singleton on repeated resolve
- ok 3 - should return cached singleton on repeated resolve
- ---
- duration_ms: 2.700355
- type: 'test'
- ...
- # Subtest: should throw on resolving unregistered service
- ok 4 - should throw on resolving unregistered service
- ---
- duration_ms: 3.280776
- type: 'test'
- ...
- # Subtest: should have default registrations
- ok 5 - should have default registrations
- ---
- duration_ms: 2.960963
- type: 'test'
- ...
- # Subtest: should support re-registration (overwrite)
- ok 6 - should support re-registration (overwrite)
- ---
- duration_ms: 2.683065
- type: 'test'
- ...
- 1..6
-ok 4 - DI Container — container.ts
- ---
- duration_ms: 314.82733
- type: 'suite'
- ...
-# [Plugins] Registered "test-logger" (priority: 10, enabled: true)
-# Subtest: Plugin Architecture — plugins/index.ts
- # Subtest: should register and list plugins
- ok 1 - should register and list plugins
- ---
- duration_ms: 7.470741
- type: 'test'
- ...
-# [Plugins] Registered "low" (priority: 200, enabled: true)
-# [Plugins] Registered "high" (priority: 1, enabled: true)
-# [Plugins] Registered "mid" (priority: 50, enabled: true)
- # Subtest: should sort plugins by priority
- ok 2 - should sort plugins by priority
- ---
- duration_ms: 3.51882
- type: 'test'
- ...
-# [Plugins] Registered "first" (priority: 1, enabled: true)
-# [Plugins] Registered "second" (priority: 2, enabled: true)
- # Subtest: should run onRequest hooks in order
- ok 3 - should run onRequest hooks in order
- ---
- duration_ms: 2.940782
- type: 'test'
- ...
-# [Plugins] Registered "blocker" (priority: 1, enabled: true)
-# [Plugins] Registered "never-runs" (priority: 2, enabled: true)
-# [Plugins] Request blocked by "blocker"
- # Subtest: should support request blocking
- ok 4 - should support request blocking
- ---
- duration_ms: 3.293988
- type: 'test'
- ...
-# [Plugins] Registered "toggle-me" (priority: 100, enabled: true)
- # Subtest: should enable/disable plugins at runtime
- ok 5 - should enable/disable plugins at runtime
- ---
- duration_ms: 2.784824
- type: 'test'
- ...
-# [Plugins] Registered "removable" (priority: 100, enabled: true)
- # Subtest: should unregister plugins
- ok 6 - should unregister plugins
- ---
- duration_ms: 2.608295
- type: 'test'
- ...
-# [Plugins] Registered "response-modifier" (priority: 100, enabled: true)
- # Subtest: should run onResponse hooks
- ok 7 - should run onResponse hooks
- ---
- duration_ms: 2.746361
- type: 'test'
- ...
-# [Plugins] Registered "error-handler" (priority: 100, enabled: true)
-# [Plugins] Error recovered by "error-handler"
- # Subtest: should run onError hooks and allow recovery
- ok 8 - should run onError hooks and allow recovery
- ---
- duration_ms: 2.899465
- type: 'test'
- ...
- # Subtest: should return null from onError if no recovery
- ok 9 - should return null from onError if no recovery
- ---
- duration_ms: 2.313171
- type: 'test'
- ...
- 1..9
-ok 5 - Plugin Architecture — plugins/index.ts
- ---
- duration_ms: 31.770504
- type: 'suite'
- ...
-# Subtest: Prompt Template Versioning — prompts.ts module existence
- # Subtest: prompts.ts should exist
- ok 1 - prompts.ts should exist
- ---
- duration_ms: 0.369635
- type: 'test'
- ...
- # Subtest: should export CRUD functions
- ok 2 - should export CRUD functions
- ---
- duration_ms: 0.391134
- type: 'test'
- ...
- # Subtest: should define PromptTemplate interface
- ok 3 - should define PromptTemplate interface
- ---
- duration_ms: 0.333598
- type: 'test'
- ...
- # Subtest: should use content hashing for deduplication
- ok 4 - should use content hashing for deduplication
- ---
- duration_ms: 0.31716
- type: 'test'
- ...
- 1..4
-ok 6 - Prompt Template Versioning — prompts.ts module existence
- ---
- duration_ms: 1.752975
- type: 'suite'
- ...
-# Subtest: Eval Scheduler — scheduler.ts module existence
- # Subtest: scheduler.ts should exist
- ok 1 - scheduler.ts should exist
- ---
- duration_ms: 0.347059
- type: 'test'
- ...
- # Subtest: should export scheduling functions
- ok 2 - should export scheduling functions
- ---
- duration_ms: 0.582836
- type: 'test'
- ...
- # Subtest: should define ScheduledEval and EvalRunResult types
- ok 3 - should define ScheduledEval and EvalRunResult types
- ---
- duration_ms: 0.459196
- type: 'test'
- ...
- # Subtest: should have pluggable output provider
- ok 4 - should have pluggable output provider
- ---
- duration_ms: 0.310267
- type: 'test'
- ...
- 1..4
-ok 7 - Eval Scheduler — scheduler.ts module existence
- ---
- duration_ms: 1.961338
- type: 'suite'
- ...
-# Subtest: Migration System — files exist
- # Subtest: migrationRunner.ts should exist
- ok 1 - migrationRunner.ts should exist
- ---
- duration_ms: 0.381312
- type: 'test'
- ...
- # Subtest: 001_initial_schema.sql should exist
- ok 2 - 001_initial_schema.sql should exist
- ---
- duration_ms: 0.248336
- type: 'test'
- ...
- # Subtest: core.ts should reference migration runner
- ok 3 - core.ts should reference migration runner
- ---
- duration_ms: 0.542947
- type: 'test'
- ...
- 1..3
-ok 8 - Migration System — files exist
- ---
- duration_ms: 1.448447
- type: 'suite'
- ...
-# Subtest: CORS — centralized configuration
- # Subtest: shared/utils/cors.ts should exist
- ok 1 - shared/utils/cors.ts should exist
- ---
- duration_ms: 0.374376
- type: 'test'
- ...
- # Subtest: should export CORS_HEADERS and CORS_ORIGIN
- ok 2 - should export CORS_HEADERS and CORS_ORIGIN
- ---
- duration_ms: 0.390035
- type: 'test'
- ...
- 1..2
-ok 9 - CORS — centralized configuration
- ---
- duration_ms: 0.962465
- type: 'suite'
- ...
-1..9
-# tests 42
-# suites 9
-# pass 42
-# fail 0
-# cancelled 0
-# skipped 0
-# todo 0
-# duration_ms 869.387894
diff --git a/debug3.log b/debug3.log
deleted file mode 100644
index 1f0e7424f2..0000000000
--- a/debug3.log
+++ /dev/null
@@ -1,1772 +0,0 @@
-TAP version 13
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [CREDENTIALS] No external credentials file found, using defaults.
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [SECURITY] API_KEY_SECRET is not set. API key CRC validation is disabled.
-# [DB] Added api_keys.allowed_connections column
-# [DB] Added api_keys.auto_resolve column
-# [DB] Added api_keys.is_active column
-# [DB] Added api_keys.access_schedule column
-# [DB] Added api_keys.max_requests_per_day column
-# [DB] Added api_keys.max_requests_per_minute column
-# [DB] Added api_keys.max_sessions column
-# [23:40:30.381] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:30.382] [34mDEBUG[39m: [36m[sse] API Key: sk-m...01c2[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:30.392] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# 🛡️ [RATE-LIMIT] Loaded 0 explicit + 1 auto-enabled + 0 custom rpm/tpm protection(s)
-# [23:40:30.396] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:30.396] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:30.397] [32mINFO[39m: [36m[sse] Using openai account: 4782b395...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:30.415] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:30.425] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [32m[20:40] 📊 [USAGE] OPENAI | in=4 | out=2 | account=4782b395...[0m
-# [23:40:30.493] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline handles OpenAI passthrough with valid API key auth
-ok 1 - chat pipeline handles OpenAI passthrough with valid API key auth
- ---
- duration_ms: 279.06402
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [23:40:30.652] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | claude/claude-3-5-sonnet-20241022 | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | claude/claude-3-5-sonnet-20241022 | 1 msgs"
-# [23:40:30.652] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [32m[20:40] 📊 [USAGE] CLAUDE | in=10 | out=4 | account=9ced487d...[0m
-# [23:40:30.656] [32mINFO[39m: [36m[sse] Provider: claude, Model: claude-3-5-sonnet-20241022[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:30.658] [34mDEBUG[39m: [36m[sse] claude | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:30.658] [34mDEBUG[39m: [36m[sse] claude | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:30.658] [32mINFO[39m: [36m[sse] Using claude account: 9ced487d...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:30.660] [34mDEBUG[39m: [36m[sse] openai → claude | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:30.664] [34mDEBUG[39m: [36m[sse] CLAUDE | claude-3-5-sonnet-20241022 | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:30.671] [34mDEBUG[39m: [36m[sse] Stored response for claude-3-5-sonnet-20241022 (14 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline translates OpenAI requests to Claude and returns OpenAI-shaped responses
-ok 2 - chat pipeline translates OpenAI requests to Claude and returns OpenAI-shaped responses
- ---
- duration_ms: 203.116864
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [23:40:30.852] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | gemini/gemini-2.5-flash | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | gemini/gemini-2.5-flash | 1 msgs"
-# [23:40:30.852] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [32m[20:40] 📊 [USAGE] GEMINI | in=5 | out=7 | account=3c7613b4...[0m
-# [23:40:30.855] [32mINFO[39m: [36m[sse] Provider: gemini, Model: gemini-2.5-flash[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:30.856] [34mDEBUG[39m: [36m[sse] gemini | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:30.856] [34mDEBUG[39m: [36m[sse] gemini | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:30.856] [32mINFO[39m: [36m[sse] Using gemini account: 3c7613b4...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:30.858] [34mDEBUG[39m: [36m[sse] openai → gemini | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:30.860] [34mDEBUG[39m: [36m[sse] GEMINI | gemini-2.5-flash | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:30.864] [34mDEBUG[39m: [36m[sse] Stored response for gemini-2.5-flash (12 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline translates OpenAI requests to Gemini and returns OpenAI-shaped responses
-ok 3 - chat pipeline translates OpenAI requests to Gemini and returns OpenAI-shaped responses
- ---
- duration_ms: 157.635756
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [23:40:31.026] [32mINFO[39m: [36m[sse] 📥 POST /v1/messages | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/messages | openai/gpt-4o-mini | 1 msgs"
-# [23:40:31.026] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [32m[20:40] 📊 [USAGE] OPENAI | in=4 | out=2 | account=9e6e0e97...[0m
-# [23:40:31.029] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:31.030] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:31.030] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:31.031] [32mINFO[39m: [36m[sse] Using openai account: 9e6e0e97...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:31.032] [34mDEBUG[39m: [36m[sse] claude → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:31.035] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 2 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:31.041] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline translates Claude-format requests into OpenAI upstream and back to Claude
-ok 4 - chat pipeline translates Claude-format requests into OpenAI upstream and back to Claude
- ---
- duration_ms: 179.12553
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [23:40:31.207] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | claude/claude-sonnet-4-6 | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | claude/claude-sonnet-4-6 | 1 msgs"
-# [23:40:31.207] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [20:40:31] 📊 [32m[USAGE] CLAUDE | in=0 | out=3 | account=5d0e9af4...[0m
-# [20:40:31] 🌊 [STREAM] CLAUDE | claude-sonnet-4-6 | 26ms | complete
-# [23:40:31.209] [32mINFO[39m: [36m[sse] Provider: claude, Model: claude-sonnet-4-6[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:31.211] [34mDEBUG[39m: [36m[sse] claude | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:31.211] [34mDEBUG[39m: [36m[sse] claude | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:31.211] [32mINFO[39m: [36m[sse] Using claude account: 5d0e9af4...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:31.212] [34mDEBUG[39m: [36m[sse] openai → claude | stream=true[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:31.213] [34mDEBUG[39m: [36m[sse] CLAUDE | claude-sonnet-4-6 | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:31.214] [34mDEBUG[39m: [36m[sse] Translation mode: claude → openai[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "STREAM"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline converts Claude SSE streams into OpenAI SSE output
-ok 5 - chat pipeline converts Claude SSE streams into OpenAI SSE output
- ---
- duration_ms: 263.312873
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [DB] Added api_keys.allowed_connections column
-# [DB] Added api_keys.auto_resolve column
-# [DB] Added api_keys.is_active column
-# [DB] Added api_keys.access_schedule column
-# [DB] Added api_keys.max_requests_per_day column
-# [DB] Added api_keys.max_requests_per_minute column
-# [DB] Added api_keys.max_sessions column
-# [23:40:31.448] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:31.448] [34mDEBUG[39m: [36m[sse] API Key: does...xist[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:31.456] [33mWARN[39m: [36m[sse] API key not found or invalid (must be created in API Manager)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:31.458] [33mWARN[39m: [36m[sse] Invalid JSON body[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline rejects invalid API keys and malformed JSON bodies
-ok 6 - chat pipeline rejects invalid API keys and malformed JSON bodies
- ---
- duration_ms: 144.197861
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [23:40:31.570] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:31.570] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:31.570] [33mWARN[39m: [36m[sse] Missing API key while REQUIRE_API_KEY=true[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline rejects requests without a bearer key when strict API key mode is enabled
-ok 7 - chat pipeline rejects requests without a bearer key when strict API key mode is enabled
- ---
- duration_ms: 159.810061
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [23:40:31.727] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | undefined | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | undefined | 1 msgs"
-# [23:40:31.727] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:31.727] [33mWARN[39m: [36m[sse] Missing model[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline returns 400 when the model field is omitted
-ok 8 - chat pipeline returns 400 when the model field is omitted
- ---
- duration_ms: 152.12799
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [23:40:31.907] [34mDEBUG[39m: [36m[sse] Accept: text/event-stream header → overriding stream=true (body had no stream field)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "STREAM"
-# [23:40:31.907] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:31.907] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [20:40:31] 📊 [32m[USAGE] OPENAI | in=2062 | out=6 | account=93009bc3...[0m [33m(estimated)[0m
-# [20:40:31] 🌊 [STREAM] OPENAI | gpt-4o-mini | 10ms | complete
-# [23:40:31.910] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:31.911] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:31.911] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:31.911] [32mINFO[39m: [36m[sse] Using openai account: 93009bc3...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:31.913] [34mDEBUG[39m: [36m[sse] openai → openai | stream=true[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:31.913] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:31.915] [34mDEBUG[39m: [36m[sse] Standard passthrough mode[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "STREAM"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline treats Accept text/event-stream as streaming mode and returns a session header
-ok 9 - chat pipeline treats Accept text/event-stream as streaming mode and returns a session header
- ---
- duration_ms: 153.539772
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [32m[20:40] 📊 [USAGE] OPENAI | in=4 | out=2 | account=9fef864f...[0m
-# [23:40:32.063] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | local-router | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | local-router | 1 msgs"
-# [23:40:32.063] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.064] [32mINFO[39m: [36m[sse] Combo "local-router" [priority] with 1 models[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [23:40:32.068] [32mINFO[39m: [36m[sse] priority with nested resolution: 1 total models[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [23:40:32.070] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.070] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.071] [32mINFO[39m: [36m[sse] Trying model 1/1: openai/gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [23:40:32.071] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:32.073] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.073] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.074] [32mINFO[39m: [36m[sse] Using openai account: 9fef864f...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.076] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:32.077] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:32.083] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [23:40:32.088] [32mINFO[39m: [36m[sse] Model openai/gpt-4o-mini succeeded (20ms, 0 fallbacks)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline supports local mode without Authorization on explicit combos
-ok 10 - chat pipeline supports local mode without Authorization on explicit combos
- ---
- duration_ms: 170.688713
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [DB] Added api_keys.allowed_connections column
-# [DB] Added api_keys.auto_resolve column
-# [DB] Added api_keys.is_active column
-# [DB] Added api_keys.access_schedule column
-# [DB] Added api_keys.max_requests_per_day column
-# [DB] Added api_keys.max_requests_per_minute column
-# [DB] Added api_keys.max_sessions column
-# [23:40:32.256] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:32.256] [34mDEBUG[39m: [36m[sse] API Key: sk-m...2216[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [32m[20:40] 📊 [USAGE] OPENAI | in=4 | out=2 | account=50eccfa9...[0m
-# [23:40:32.259] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:32.261] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.261] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.261] [32mINFO[39m: [36m[sse] Using openai account: 50eccfa9...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.263] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:32.264] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:32.273] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline honors noLog by redacting persisted call log payloads
-ok 11 - chat pipeline honors noLog by redacting persisted call log payloads
- ---
- duration_ms: 211.592743
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [23:40:32.408] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:32.408] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.443] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:32.443] [34mDEBUG[39m: [36m[sse] openai | total connections: 0, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.444] [34mDEBUG[39m: [36m[sse] openai | all connections (incl inactive): 0[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.444] [33mWARN[39m: [36m[sse] No credentials for openai[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.444] [31mERROR[39m: [36m[sse] No credentials for provider: openai[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline returns current no-credentials contract when no provider connection exists
-ok 12 - chat pipeline returns current no-credentials contract when no provider connection exists
- ---
- duration_ms: 140.10352
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [23:40:32.576] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:32.576] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.579] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:32.579] [33mWARN[39m: [36m[sse] openai/gpt-4o-mini is in cooldown, rejecting request[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AVAILABILITY"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline returns 503 when the requested model is temporarily unavailable
-ok 13 - chat pipeline returns 503 when the requested model is temporarily unavailable
- ---
- duration_ms: 132.250444
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [23:40:32.711] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:32.711] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [31m[ERROR] [500]: provider exploded[0m
-# ❌ openai [500]: [500]: provider exploded
-# [23:40:32.726] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:32.727] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.727] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.728] [32mINFO[39m: [36m[sse] Using openai account: 4ab558c6...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.732] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:32.735] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:32.772] [33mWARN[39m: [36m[sse] Account 4ab558c6... unavailable (500), trying fallback[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.773] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: 4ab558c6-3588-4e36-b434-a324e8cf1bc3[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.773] [34mDEBUG[39m: [36m[sse] openai | available: 0/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.773] [34mDEBUG[39m: [36m[sse] → 4ab558c6 | excluded rateLimited until 2026-04-06T23:40:35.770Z[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.774] [33mWARN[39m: [36m[sse] openai | all 1 active accounts rate limited (reset after 3s) | lastErrorCode=500.0, lastError=[500]: provider exploded[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.774] [33mWARN[39m: [36m[sse] [openai/gpt-4o-mini] [500]: provider exploded (reset after 3s)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline surfaces upstream 500 responses as structured errors
-ok 14 - chat pipeline surfaces upstream 500 responses as structured errors
- ---
- duration_ms: 219.873381
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [23:40:32.973] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:32.973] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.977] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:32.979] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.979] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.980] [32mINFO[39m: [36m[sse] Using openai account: 71ff871a...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:32.983] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:32.984] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:32.986] [34mDEBUG[39m: [36m[sse] 429 intra-retry 1/2 on https://api.openai.com/v1/chat/completions — waiting 0ms[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "RETRY"
-# [23:40:32.989] [34mDEBUG[39m: [36m[sse] 429 intra-retry 2/2 on https://api.openai.com/v1/chat/completions — waiting 0ms[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "RETRY"
-# [provider] Node 71ff871a-0cc7-4cf5-aecd-ee0f6bdd87f2 rate limited (429) - Next available at 2026-04-06T23:41:02.996Z
-# [31m[ERROR] [429]: Rate limit exceeded. Your quota will reset after 30s.[0m
-# [23:40:33.006] [32mINFO[39m: [36m[sse] 71ff871a terminal status=credits_exhausted, skipping cooldown overwrite[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.006] [33mWARN[39m: [36m[sse] Account 71ff871a... unavailable (429), trying fallback[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.007] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: 71ff871a-0cc7-4cf5-aecd-ee0f6bdd87f2[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.007] [34mDEBUG[39m: [36m[sse] openai | available: 0/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.007] [34mDEBUG[39m: [36m[sse] → 71ff871a | excluded rateLimited until 2026-04-06T23:41:02.996Z[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.007] [33mWARN[39m: [36m[sse] openai | all 1 active accounts rate limited (reset after 30s) | lastErrorCode=429.0, lastError=Rate limit exceeded. Your quota will reset after 3[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.007] [32mINFO[39m: [36m[sse] openai/gpt-4o-mini marked unavailable — all accounts exhausted (HTTP 429)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AVAILABILITY"
-# [23:40:33.007] [33mWARN[39m: [36m[sse] [openai/gpt-4o-mini] [429]: Rate limit exceeded. Your quota will reset after 30s. (reset after 30s)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline returns 429 with Retry-After when the upstream rate-limits the only account
-ok 15 - chat pipeline returns 429 with Retry-After when the upstream rate-limits the only account
- ---
- duration_ms: 219.862586
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [23:40:33.195] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:33.195] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [31m[ERROR] [504]: upstream timed out[0m
-# ❌ openai [504]: [504]: upstream timed out
-# [23:40:33.197] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:33.199] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.199] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.199] [32mINFO[39m: [36m[sse] Using openai account: b0f13742...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.201] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:33.202] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:33.203] [33mWARN[39m: [36m[sse] Fetch timeout after 600000ms on https://api.openai.com/v1/chat/completions[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "TIMEOUT"
-# [23:40:33.207] [33mWARN[39m: [36m[sse] Account b0f13742... unavailable (504), trying fallback[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.208] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: b0f13742-172d-4e2f-b1bd-51ff12ca0898[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.208] [34mDEBUG[39m: [36m[sse] openai | available: 0/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.209] [34mDEBUG[39m: [36m[sse] → b0f13742 | excluded rateLimited until 2026-04-06T23:40:36.206Z[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.209] [33mWARN[39m: [36m[sse] openai | all 1 active accounts rate limited (reset after 3s) | lastErrorCode=504.0, lastError=[504]: upstream timed out[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.209] [33mWARN[39m: [36m[sse] [openai/gpt-4o-mini] [504]: upstream timed out (reset after 3s)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline maps upstream timeouts to 504 responses
-ok 16 - chat pipeline maps upstream timeouts to 504 responses
- ---
- duration_ms: 202.717074
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [DB] Added api_keys.allowed_connections column
-# [DB] Added api_keys.auto_resolve column
-# [DB] Added api_keys.is_active column
-# [DB] Added api_keys.access_schedule column
-# [DB] Added api_keys.max_requests_per_day column
-# [DB] Added api_keys.max_requests_per_minute column
-# [DB] Added api_keys.max_sessions column
-# [23:40:33.412] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:33.412] [34mDEBUG[39m: [36m[sse] API Key: sk-m...779b[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [32m[20:40] 📊 [USAGE] OPENAI | in=4 | out=2 | account=c2029285...[0m
-# [23:40:33.417] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:33.418] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.418] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.421] [32mINFO[39m: [36m[sse] Using openai account: c2029285...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.423] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:33.425] [34mDEBUG[39m: [36m[sse] Injected 1 memories for key=75502502-4c99-4f6e-9719-5581b00e0185[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "MEMORY"
-# [23:40:33.426] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 2 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:33.439] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline injects memory context before sending the upstream request
-ok 17 - chat pipeline injects memory context before sending the upstream request
- ---
- duration_ms: 229.129863
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [DB] Added api_keys.allowed_connections column
-# [DB] Added api_keys.auto_resolve column
-# [DB] Added api_keys.is_active column
-# [DB] Added api_keys.access_schedule column
-# [DB] Added api_keys.max_requests_per_day column
-# [DB] Added api_keys.max_requests_per_minute column
-# [DB] Added api_keys.max_sessions column
-# [23:40:33.643] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:33.646] [34mDEBUG[39m: [36m[sse] API Key: sk-m...e420[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [32m[20:40] 📊 [USAGE] OPENAI | in=6 | out=4 | account=eb34b60b...[0m
-# [23:40:33.650] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:33.651] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.651] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.652] [32mINFO[39m: [36m[sse] Using openai account: eb34b60b...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.654] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:33.656] [34mDEBUG[39m: [36m[sse] Injected 1 skills[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "SKILLS"
-# [23:40:33.658] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:33.672] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (10 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline injects skills into tools and intercepts tool calls with skill output
-ok 18 - chat pipeline injects skills into tools and intercepts tool calls with skill output
- ---
- duration_ms: 230.685673
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [23:40:33.815] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:33.815] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [31m[ERROR] [500]: first account failed[0m
-# ❌ openai [500]: [500]: first account failed
-# [32m[20:40] 📊 [USAGE] OPENAI | in=4 | out=2 | account=8ae4bbf9...[0m
-# [23:40:33.817] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:33.818] [34mDEBUG[39m: [36m[sse] openai | total connections: 2, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.818] [34mDEBUG[39m: [36m[sse] openai | available: 2/2[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.818] [32mINFO[39m: [36m[sse] Using openai account: 5cfce4ff...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.819] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:33.820] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:33.824] [33mWARN[39m: [36m[sse] Account 5cfce4ff... unavailable (500), trying fallback[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.825] [34mDEBUG[39m: [36m[sse] openai | total connections: 2, excludeId: 5cfce4ff-00c1-4e25-8c9b-acc7be84082f[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.825] [34mDEBUG[39m: [36m[sse] openai | available: 1/2[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.825] [34mDEBUG[39m: [36m[sse] → 5cfce4ff | excluded rateLimited until 2026-04-06T23:40:36.823Z[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.825] [32mINFO[39m: [36m[sse] Using openai account: 8ae4bbf9...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.826] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:33.826] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:33.829] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline falls back to the next account after a provider failure
-ok 19 - chat pipeline falls back to the next account after a provider failure
- ---
- duration_ms: 147.807701
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [31m[ERROR] [503]: openai combo miss[0m
-# ❌ openai [503]: [503]: openai combo miss
-# [23:40:33.958] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | combo-fallback | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | combo-fallback | 1 msgs"
-# [23:40:33.959] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.959] [32mINFO[39m: [36m[sse] Combo "combo-fallback" [priority] with 2 models[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [23:40:33.960] [32mINFO[39m: [36m[sse] priority with nested resolution: 2 total models[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [23:40:33.961] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.961] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.961] [32mINFO[39m: [36m[sse] Trying model 1/2: openai/gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [23:40:33.961] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:33.962] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.962] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.962] [32mINFO[39m: [36m[sse] Using openai account: 116917d1...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.963] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:33.963] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:33.971] [33mWARN[39m: [36m[sse] Account 116917d1... unavailable (503), trying fallback[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.971] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: 116917d1-5c07-40ff-ba36-b5bea52054ff[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.971] [34mDEBUG[39m: [36m[sse] openai | available: 0/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.971] [34mDEBUG[39m: [36m[sse] → 116917d1 | excluded rateLimited until 2026-04-06T23:40:36.970Z[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.971] [33mWARN[39m: [36m[sse] openai | all 1 active accounts rate limited (reset after 3s) | lastErrorCode=503.0, lastError=[503]: openai combo miss[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:33.971] [32mINFO[39m: [36m[sse] openai/gpt-4o-mini marked unavailable — all accounts exhausted (HTTP 503)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AVAILABILITY"
-# [23:40:33.971] [33mWARN[39m: [36m[sse] [openai/gpt-4o-mini] [503]: openai combo miss (reset after 3s)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CHAT"
-# [23:40:33.972] [33mWARN[39m: [36m[sse] Model openai/gpt-4o-mini failed, trying next[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [35mstatus[39m: 503
-# [23:40:33.972] [32mINFO[39m: [36m[sse] Waiting 3000ms before fallback to next model[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [32m[20:40] 📊 [USAGE] CLAUDE | in=10 | out=4 | account=23e06021...[0m
-# [23:40:36.974] [34mDEBUG[39m: [36m[sse] claude | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:36.975] [34mDEBUG[39m: [36m[sse] claude | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:36.975] [32mINFO[39m: [36m[sse] Trying model 2/2: claude/claude-3-5-sonnet-20241022[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [23:40:36.975] [32mINFO[39m: [36m[sse] Provider: claude, Model: claude-3-5-sonnet-20241022[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:36.976] [34mDEBUG[39m: [36m[sse] claude | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:36.976] [34mDEBUG[39m: [36m[sse] claude | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:36.976] [32mINFO[39m: [36m[sse] Using claude account: 23e06021...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:36.977] [34mDEBUG[39m: [36m[sse] openai → claude | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:36.978] [34mDEBUG[39m: [36m[sse] CLAUDE | claude-3-5-sonnet-20241022 | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:36.981] [34mDEBUG[39m: [36m[sse] Stored response for claude-3-5-sonnet-20241022 (14 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [23:40:36.983] [32mINFO[39m: [36m[sse] Model claude/claude-3-5-sonnet-20241022 succeeded (3023ms, 0 fallbacks)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "COMBO"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# Subtest: chat pipeline falls back across combo models when the first provider fails
-ok 20 - chat pipeline falls back across combo models when the first provider fails
- ---
- duration_ms: 3156.605154
- type: 'test'
- ...
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [23:40:37.124] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:37.124] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:37.125] [32mINFO[39m: [36m[sse] 📥 POST /v1/chat/completions | openai/gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "HTTP"
-# [35mmethod[39m: "POST"
-# [35mpath[39m: "/v1/chat/completions | openai/gpt-4o-mini | 1 msgs"
-# [23:40:37.125] [34mDEBUG[39m: [36m[sse] No API key provided (local mode)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:37.127] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:37.128] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:37.128] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:37.128] [32mINFO[39m: [36m[sse] Using openai account: 10d570ae...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:37.129] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:37.130] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [23:40:37.132] [32mINFO[39m: [36m[sse] Provider: openai, Model: gpt-4o-mini[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "ROUTING"
-# [23:40:37.132] [34mDEBUG[39m: [36m[sse] openai | total connections: 1, excludeId: none[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:37.132] [34mDEBUG[39m: [36m[sse] openai | available: 1/1[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:37.133] [32mINFO[39m: [36m[sse] Using openai account: 10d570ae...[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "AUTH"
-# [23:40:37.134] [34mDEBUG[39m: [36m[sse] openai → openai | stream=false[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "FORMAT"
-# [23:40:37.134] [34mDEBUG[39m: [36m[sse] OPENAI | gpt-4o-mini | 1 msgs[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "REQUEST"
-# [32m[20:40] 📊 [USAGE] OPENAI | in=4 | out=2 | account=10d570ae...[0m
-# [32m[20:40] 📊 [USAGE] OPENAI | in=4 | out=2 | account=10d570ae...[0m
-# [23:40:37.157] [34mDEBUG[39m: [36m[sse] Joined in-flight request hash=f467c6e15f3cb6e2[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "DEDUP"
-# [23:40:37.160] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [23:40:37.161] [34mDEBUG[39m: [36m[sse] Stored response for gpt-4o-mini (6 tokens)[39m
-# [35mmodule[39m: "sse"
-# [35mtag[39m: "CACHE"
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# [Migration] Applied: 002_mcp_a2a_tables
-# [Migration] Applied: 003_provider_node_custom_paths
-# [Migration] Applied: 004_proxy_registry
-# [Migration] Applied: 005_combo_agent_fields
-# [Migration] Applied: 006_detailed_request_logs
-# [Migration] Applied: 007_search_request_type
-# [Migration] Applied: 008_registered_keys
-# [Migration] Applied: 009_requested_model
-# [Migration] Applied: 010_model_combo_mappings
-# [Migration] Applied: 011_webhooks
-# [Migration] Applied: 012_fix_token_input_cache_tokens
-# [Migration] Applied: 013_quota_snapshots
-# [Migration] Applied: 014_unified_log_artifacts
-# [Migration] Applied: 015_create_memories
-# [Migration] Applied: 016_create_skills
-# [Migration] Applied: 017_version_manager_upstream_proxy
-# [Migration] 16 migration(s) applied successfully.
-# [DB] SQLite database ready: /tmp/omniroute-chat-pipeline-5DubiT/storage.sqlite
-# [DB] SQLite WAL checkpoint completed (TRUNCATE).
-# Subtest: chat pipeline deduplicates concurrent identical non-stream requests
-ok 21 - chat pipeline deduplicates concurrent identical non-stream requests
- ---
- duration_ms: 176.545473
- type: 'test'
- ...
-# === WTF NODE DUMP ===
-# [WTF Node?] open handles:
-# - File descriptors: (note: stdio always exists)
-# - fd 2 (stdio)
-# - fd 1 (stdio)
-1..21
-# tests 21
-# suites 0
-# pass 21
-# fail 0
-# cancelled 0
-# skipped 0
-# todo 0
-# duration_ms 36056.905402
diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md
index 5db3edc0cc..c76f482a06 100644
--- a/docs/ARCHITECTURE.md
+++ b/docs/ARCHITECTURE.md
@@ -34,6 +34,7 @@ Core capabilities:
- Anti-thundering herd protection with mutex locking
- Signature-based request deduplication cache
- Domain layer: model availability, cost rules, fallback policy, lockout policy
+- Context Relay: session handoff summaries for account rotation continuity
- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers)
- Policy engine for centralized request evaluation (lockout → budget → fallback)
- Request telemetry with p50/p95/p99 latency aggregation
@@ -220,6 +221,8 @@ Services (business logic):
- Wildcard model routing: `open-sse/services/wildcardRouter.ts`
- Rate limit management: `open-sse/services/rateLimitManager.ts`
- Circuit breaker: `open-sse/services/circuitBreaker.ts`
+- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy
+- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions
Domain layer modules:
@@ -800,7 +803,10 @@ Environment variables actively used by code:
5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency.
6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates).
7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`).
-8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies), AI (thinking budget, system prompt, prompt cache), Advanced (proxy).
+8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy).
+9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed.
+10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22.
+11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible.
## Operational Verification Checklist
diff --git a/docs/FEATURES.md b/docs/FEATURES.md
index dcea4eb0c5..a22f111002 100644
--- a/docs/FEATURES.md
+++ b/docs/FEATURES.md
@@ -16,7 +16,7 @@ Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI)
## 🎨 Combos
-Create model routing combos with 6 strategies: priority, weighted, round-robin, random, least-used, and cost-optimized. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks.
+Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks.

@@ -66,7 +66,7 @@ Comprehensive settings panel with tabs:
- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls
- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info
- **Routing** — Model aliases, background task degradation
-- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring
+- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration
- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode

@@ -92,6 +92,30 @@ Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in a
---
+## 🔗 Context Relay _(v3.5.5+)_
+
+A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context.
+
+Configurable via combo-level or global settings:
+- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%)
+- **Max Messages For Summary** — How much recent history to condense
+- **Summary Model** — Optional override model for generating the handoff summary
+
+Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md).
+
+---
+
+## 🛡️ Proxy Hardening _(v3.5.5+)_
+
+Comprehensive proxy configuration enforcement across the entire request pipeline:
+
+- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments
+- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings
+- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22
+- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS
+
+---
+
## 🖼️ Media _(v2.0.3+)_
Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen.
diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md
index d02d3e34dc..45f17164cd 100644
--- a/docs/TROUBLESHOOTING.md
+++ b/docs/TROUBLESHOOTING.md
@@ -15,6 +15,60 @@ Common problems and solutions for OmniRoute.
| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` |
| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` |
| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) |
+| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below |
+| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below |
+
+---
+
+## Node.js Compatibility
+
+
+
+### Login page crashes or shows "Module self-registration" error
+
+**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database.
+
+**Symptoms:**
+- Login page shows a blank screen or a server error
+- Console shows `Error: Module did not self-register` or similar native binding errors
+- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected
+
+**Fix:**
+
+1. Install Node.js 22 LTS (recommended):
+ ```bash
+ nvm install 22
+ nvm use 22
+ ```
+2. Verify your version: `node --version` should show `v22.x.x`
+3. Reinstall OmniRoute: `npm install -g omniroute`
+4. Restart: `omniroute`
+
+> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**.
+
+---
+
+## Proxy Issues
+
+
+
+### Provider validation shows "fetch failed"
+
+**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing.
+
+**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically.
+
+### Token health check fails with "fetch failed"
+
+**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection.
+
+**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+.
+
+### SOCKS5 proxy returns "invalid onRequestStart method"
+
+**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation.
+
+**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+.
---
diff --git a/docs/i18n/ar/README.md b/docs/i18n/ar/README.md
index eca822e9ff..20338c5756 100644
--- a/docs/i18n/ar/README.md
+++ b/docs/i18n/ar/README.md
@@ -4,11 +4,14 @@
---
+
### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback.
-_وكيل واجهة برمجة التطبيقات العالمي الخاص بك — نقطة نهاية واحدة، وأكثر من 60 موفرًا، بدون أي توقف عن العمل. الآن مع**خادم MCP (25 أداة)**و**بروتوكول A2A**و**أنظمة الذاكرة/المهارات**و**تطبيق Electron Desktop**._
+_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._
-**إكمالات الدردشة • التضمينات • إنشاء الصور • الفيديو • الموسيقى • الصوت • إعادة الترتيب •**بحث الويب**• خادم MCP • بروتوكول A2A • 100% TypeScript**---
+**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript**
+
+---
@@ -39,9 +42,13 @@ _وكيل واجهة برمجة التطبيقات العالمي الخاص ب
[](https://omniroute.online)
[](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-[🌐 موقع الويب](https://omniroute.online) • [🚀 البداية السريعة](#-بدء سريع) • [💡 الميزات](#-key-features) • [📖 المستندات](#-وثائق) • [💰 التسعير](#-تسعير في لمحة) • [💬 واتساب](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-🌐**متوفر باللغة:**🇺🇸 [الإنجليزية](README.md) | 🇧🇷 [البرتغالية (البرازيل)](docs/i18n/pt-BR/README.md) | 🇪🇸 [الإسبانية](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [الإيطالية](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [الألمانية](docs/i18n/de/README.md) | 🇮🇳 [خبر](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [بلغارسكي](docs/i18n/bg/README.md) | 🇩🇰 [الدانسك](docs/i18n/da/README.md) | 🇫🇮 [سومي](docs/i18n/fi/README.md) | 🇮🇱 [العربية](docs/i18n/he/README.md) | 🇭🇺 [المجرية](docs/i18n/hu/README.md) | 🇮🇩 [البهاسا الإندونيسية](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [البهاسا ملايو](docs/i18n/ms/README.md) | 🇳🇱 [هولندا](docs/i18n/nl/README.md) | 🇳🇴 [نورسك](docs/i18n/no/README.md) | 🇵🇹 [البرتغالية (البرتغال)](docs/i18n/pt/README.md) | 🇷🇴 [روماني](docs/i18n/ro/README.md) | 🇵🇱 [بولسكي](docs/i18n/pl/README.md) | 🇸🇰 [سلوفينسينا](docs/i18n/sk/README.md) | 🇸🇪 [السفينسكا](docs/i18n/sv/README.md) | 🇵🇭 [الفلبينية](docs/i18n/phi/README.md) | 🇨🇿 [تشيستينا](docs/i18n/cs/README.md)---
+
+
+🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)
+
+---
## 🖼️ Main Dashboard
@@ -53,554 +60,629 @@ _وكيل واجهة برمجة التطبيقات العالمي الخاص ب
## 📸 Dashboard Preview
-<التفاصيل>
+
+Click to see dashboard screenshots
-انقر لرؤية لقطات شاشة لوحة التحكم
+| Page | Screenshot |
+| -------------- | ------------------------------------------------- |
+| **Providers** |  |
+| **Combos** |  |
+| **Analytics** |  |
+| **Health** |  |
+| **Translator** |  |
+| **Settings** |  |
+| **CLI Tools** |  |
+| **Usage Logs** |  |
+| **Endpoints** |  |
-| صفحة | لقطة شاشة |
-| --------------------- | ------------------------------------------------- | ---------- |
-| **مقدمو الخدمة** |  |
-| **المجموعات** |  |
-| **تحليلات** |  |
-| **الصحة** |  |
-| **مترجم** |  |
-| **الإعدادات** |  |
-| **أدوات سطر الأوامر** |  |
-| **سجلات الاستخدام** |  |
-| **نقاط النهاية** |  | |
+
---
### 🤖 Free AI Provider for your favorite coding agents
-_قم بتوصيل أي أداة IDE أو CLI مدعومة بالذكاء الاصطناعي من خلال OmniRoute - بوابة واجهة برمجة التطبيقات المجانية للترميز غير المحدود._
+_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._
-<الجدول>
-<تر>
-
-
-
-أوبنكلاو
-
-⭐ 205 ألف
- |
-
-
-
-نانوبوت
-
-⭐ 20.9 ألف
- |
-
-
-
-بيكوكلاو
-
-⭐ 14.6 ألف
- |
-
-
-
-المخلب الصفري
-
-⭐ 9.9 ألف
- |
-
-
-
-المخلب الحديدي
-
-⭐ 2.1 كيلو
- |
-
-<تر>
-
-
-
-الرمز المفتوح
-
-⭐ 106 كيلو
- |
-
-
-
-Codex CLI
-
-⭐ 60.8 ألف
- |
-
-
-
-كلود كود
-
-⭐ 67.3 ألف
- |
-
-
-
-CLI الجوزاء
-
-⭐ 94.7 ألف
- |
-
-
-
-كود الكيلو
-
-⭐ 15.5 ألف
- |
-
-الجدول>
+
-📡 يتصل جميع الوكلاء عبر http://localhost:20128/v1 أو http://cloud.omniroute.online/v1 - تكوين واحد ونماذج وحصة غير محدودة---
+📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota
+
+---
## 🤔 Why OmniRoute?
-**توقف عن إهدار المال وضرب الحدود:**
+**Stop wasting money and hitting limits:**
--
تنتهي صلاحية حصة الاشتراك غير المستخدمة كل شهر
--
حدود المعدل تمنعك من الترميز المتوسط
--
واجهات برمجة التطبيقات باهظة الثمن (20-50 دولارًا شهريًا لكل مزود)
--
التبديل اليدوي بين مقدمي الخدمة
+-
Subscription quota expires unused every month
+-
Rate limits stop you mid-coding
+-
Expensive APIs ($20-50/month per provider)
+-
Manual switching between providers
-**OmniRoute يحل هذا:**
+**OmniRoute solves this:**
-- ✅**تعظيم الاشتراكات**- تتبع الحصة، استخدم كل جزء منها قبل إعادة التعيين
-- ✅**الرجوع التلقائي**- الاشتراك → مفتاح واجهة برمجة التطبيقات → رخيص → مجاني، بدون توقف
-- ✅**حسابات متعددة**- جولة روبن بين الحسابات لكل مزود
-- ✅**عالمي**- يعمل مع Claude Code وCodex وGemini CLI وCursor وCline وOpenClaw وأي أداة CLI---
+- ✅ **Maximize subscriptions** - Track quota, use every bit before reset
+- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime
+- ✅ **Multi-account** - Round-robin between accounts per provider
+- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool
+
+---
## 📧 Support
-> 💬**انضم إلى مجتمعنا!**[مجموعة WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — احصل على المساعدة وشارك النصائح وابق على اطلاع.
+> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated.
--**الموقع الإلكتروني**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**المشاكل**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [مجموعة المجتمع](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**المساهمة**: راجع [CONTRIBUTING.md](CONTRIBUTING.md)، أو افتح علاقة عامة، أو اختر `العدد الأول الجيد` -**المشروع الأصلي**: [9router بواسطة decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug?
+- **Website**: [omniroute.online](https://omniroute.online)
+- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute)
+- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue`
+- **Original Project**: [9router by decolua](https://github.com/decolua/9router)
-عند فتح مشكلة، يرجى تشغيل أمر معلومات النظام وإرفاق الملف الذي تم إنشاؤه:```bash
+### 🐛 Reporting a Bug?
+
+When opening an issue, please run the system-info command and attach the generated file:
+
+```bash
npm run system-info
-
```
-يؤدي هذا إلى إنشاء ملف "system-info.txt" مع إصدار Node.js، وإصدار OmniRoute، وتفاصيل نظام التشغيل، وأدوات CLI المثبتة (qoder، وgemini، و claude، وcodex، وantigravity، وdroid، وما إلى ذلك)، وحالة Docker/PM2، وحزم النظام - كل ما نحتاجه لإعادة إنتاج مشكلتك بسرعة. قم بإرفاق الملف مباشرة بمشكلة GitHub الخاصة بك.---
+This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue.
+
+---
## 🔄 How It Works
```
-
┌─────────────┐
-│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
-│ Tool │
+│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
+│ Tool │
└──────┬──────┘
-│ http://localhost:20128/v1
-↓
+ │ http://localhost:20128/v1
+ ↓
┌─────────────────────────────────────────┐
-│ OmniRoute (Smart Router) │
-│ • Format translation (OpenAI ↔ Claude) │
-│ • Quota tracking + Embeddings + Images │
-│ • Auto token refresh │
+│ OmniRoute (Smart Router) │
+│ • Format translation (OpenAI ↔ Claude) │
+│ • Quota tracking + Embeddings + Images │
+│ • Auto token refresh │
└──────┬──────────────────────────────────┘
-│
-├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
-│ ↓ quota exhausted
-├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
-│ ↓ budget limit
-├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
-│ ↓ budget limit
-└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
+ │
+ ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
+ │ ↓ quota exhausted
+ ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
+ │ ↓ budget limit
+ ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
+ │ ↓ budget limit
+ └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
Result: Never stop coding, minimal cost
-
-````
+```
---
## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases
->**يواجه كل مطور يستخدم أدوات الذكاء الاصطناعي هذه المشكلات يوميًا.**تم تصميم OmniRoute لحلها جميعًا — بدءًا من تجاوز التكاليف وحتى الكتل الإقليمية، ومن تدفقات OAuth المعطلة إلى عمليات البروتوكول وإمكانية مراقبة المؤسسة.
+> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability.
-<التفاصيل>
-💸 1. "أدفع مقابل اشتراك باهظ الثمن ولكن لا يزال يتم مقاطعتي بسبب الحدود"
+
+💸 1. "I pay for an expensive subscription but still get interrupted by limits"
-يدفع المطورون ما بين 20 إلى 200 دولار شهريًا مقابل Claude Pro أو Codex Pro أو GitHub Copilot. حتى عند الدفع، فإن الحصة لها حد أقصى — 5 ساعات من الاستخدام، أو حدود أسبوعية، أو حدود لسعر الدقيقة. في منتصف جلسة الترميز، يتوقف الموفر عن الاستجابة ويفقد المطور التدفق والإنتاجية.
+Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity.
-**كيف يحل OmniRoute المشكلة:**
+**How OmniRoute solves it:**
--**الاحتياطي الذكي ذو 4 طبقات**— في حالة نفاد حصة الاشتراك، تتم إعادة التوجيه تلقائيًا إلى مفتاح واجهة برمجة التطبيقات ← رخيص ← مجاني بدون أي تدخل يدوي
--**تتبع حدود الموفر**— يتم تحديث لقطات الحصص المخزنة مؤقتًا وفقًا لجدول من جانب الخادم (الافتراضي `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) مع توفر التحديث اليدوي في واجهة المستخدم
--**دعم الحسابات المتعددة**— حسابات متعددة لكل مزود مع نظام روبن تلقائي — عند نفاد الحساب، يتم التبديل إلى التالي
--**مجموعات مخصصة**— سلاسل احتياطية قابلة للتخصيص مع 9 إستراتيجيات موازنة (الأولوية، الموزونة، التعبئة أولاً، جولة روبن، P2C، عشوائي، الأقل استخدامًا، محسنة التكلفة، عشوائية صارمة)
--**حصص الدستور الغذائي**— مراقبة حصص مساحة عمل الشركة/الفريق مباشرة في لوحة المعلومات
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention
+- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI
+- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next
+- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**)
+- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard
-<التفاصيل>
-🔌 2. "أحتاج إلى استخدام عدة موفري خدمات ولكن لكل منهم واجهة برمجة تطبيقات مختلفة"
+
-يستخدم OpenAI تنسيقًا واحدًا، ويستخدم Claude (Anthropic) تنسيقًا آخر، ويستخدم Gemini تنسيقًا آخر. إذا أراد أحد المطورين اختبار النماذج من موفري خدمات مختلفين أو إجراء بديل فيما بينهم، فسيحتاج إلى إعادة تكوين مجموعات تطوير البرامج (SDK)، وتغيير نقاط النهاية، والتعامل مع التنسيقات غير المتوافقة. لدى موفري الخدمة المخصصين (FriendLI، NIM) نقاط نهاية نموذجية غير قياسية.
+
+🔌 2. "I need to use multiple providers but each has a different API"
-**كيف يحل OmniRoute المشكلة:**
+OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints.
--**نقطة النهاية الموحدة**— يعمل `http://localhost:20128/v1` كوكيل لجميع مقدمي الخدمة الذين يزيد عددهم عن 60
--**تنسيق الترجمة**— تلقائي وشفاف: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
--**تطهير الاستجابة**— إزالة الحقول غير القياسية (`x_groq`، `usage_breakdown`، `service_tier`) التي تكسر OpenAI SDK v1.83+
--**تطبيع الدور**— تحويل "المطور" → "النظام" لمقدمي الخدمات غير التابعين لـ OpenAI؛ "النظام" → "المستخدم" لـ GLM/ERNIE
--**Think Tag Extraction**— يستخرج كتل `` من نماذج مثل DeepSeek R1 إلى ``reasoning_content'' القياسي
--**الإخراج المنظم لـ Gemini**— التحويل التلقائي `json_schema` ← `responseMimeType`/`responseSchema`
--**`stream` الافتراضي هو `false`**- يتماشى مع مواصفات OpenAI، ويتجنب SSE غير المتوقع في Python/Rust/Go SDKs
+**How OmniRoute solves it:**
-<التفاصيل>
-🌐 3. "يحظر مزود الذكاء الاصطناعي الخاص بي منطقتي/بلدي"
+- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers
+- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
+- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+
+- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE
+- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content`
+- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion
+- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs
-يقوم مقدمو الخدمة مثل OpenAI/Codex بحظر الوصول من مناطق جغرافية معينة. يحصل المستخدمون على أخطاء مثل `unsupported_country_region_territory` أثناء اتصالات OAuth وAPI. وهذا أمر محبط بشكل خاص للمطورين من البلدان النامية.
+
-**كيف يحل OmniRoute المشكلة:**
+
+🌐 3. "My AI provider blocks my region/country"
--**تكوين الوكيل ثلاثي المستوى**— وكيل قابل للتكوين على 3 مستويات: عالمي (كل حركة المرور)، لكل مزود (موفر واحد فقط)، ولكل اتصال/مفتاح
--**شارات الوكيل المرمزة بالألوان**— المؤشرات المرئية: 🟢 الوكيل العالمي، 🟡 وكيل الموفر، 🔵 وكيل الاتصال، يظهر دائمًا عنوان IP
--**تبادل رمز OAuth عبر الوكيل**— يمر تدفق OAuth أيضًا عبر الوكيل، مما يؤدي إلى حل مشكلة `unsupported_country_region_territory`
--**اختبارات الاتصال عبر الوكيل**— تستخدم اختبارات الاتصال الوكيل الذي تم تكوينه (لا مزيد من التجاوز المباشر)
--**دعم SOCKS5**— دعم وكيل SOCKS5 الكامل للتوجيه الخارجي
--**انتحال بصمة إصبع TLS**— بصمة TLS تشبه المتصفح عبر `wreq-js` لتجاوز اكتشاف الروبوتات
--**🔏 مطابقة بصمة CLI**— إعادة ترتيب الرؤوس وحقول النص لمطابقة التوقيعات الثنائية لـ CLI الأصلية، مما يقلل بشكل كبير من مخاطر الإبلاغ عن الحساب. يتم الحفاظ على عنوان IP الخاص بالوكيل — حيث يمكنك الحصول على إخفاء**و**IP في وقت واحد
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries.
-<التفاصيل>
-🆓 4. "أريد استخدام الذكاء الاصطناعي في البرمجة ولكن ليس لدي المال"
+**How OmniRoute solves it:**
-لا يستطيع الجميع دفع ما بين 20 إلى 200 دولار شهريًا مقابل اشتراكات الذكاء الاصطناعي. يحتاج الطلاب والمطورون من البلدان الناشئة والهواة والمستقلون إلى الوصول إلى نماذج عالية الجودة بدون تكلفة.
+- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key
+- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP
+- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory`
+- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass)
+- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection
+- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously
-**كيف يحل OmniRoute المشكلة:**
+
--**موفرو الطبقة المجانية المضمنون**— دعم أصلي لمقدمي الخدمة المجانية بنسبة 100%: Qoder (5 نماذج غير محدودة عبر OAuth: kimi-k2-thinking، qwen3-coder-plus، Deepseek-r1، minimax-m2، kimi-k2)، Qwen (4 نماذج غير محدودة: qwen3-coder-plus، qwen3-coder-flash، qwen3-coder-next، Vision-model)، Kiro (Claude + AWS Builder ID مجانًا)، Gemini CLI (180 ألف رمز مميز شهريًا مجانًا)
--**Ollama Cloud**— نماذج Ollama المستضافة على السحابة على `api.ollama.com` مع فئة "الاستخدام الخفيف" مجانًا؛ استخدم البادئة `olmacloud/`
--**المجموعات المجانية فقط**— السلسلة `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 USD/الشهر بدون أي توقف عن العمل
--**NVIDIA NIM Free Access**— ~40 دورة في الدقيقة وصول مجاني للأبد إلى أكثر من 70 نموذجًا على build.nvidia.com (الانتقال من الاعتمادات إلى حدود المعدل النقي)
--**استراتيجية التكلفة المحسنة**— استراتيجية التوجيه التي تختار تلقائيًا أرخص مزود متاح
+
+🆓 4. "I want to use AI for coding but I have no money"
-<التفاصيل>
-🔒 5. "أحتاج إلى حماية بوابة الذكاء الاصطناعي الخاصة بي من الوصول غير المصرح به"
+Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost.
-عند تعريض بوابة AI للشبكة (LAN، VPS، Docker)، يمكن لأي شخص لديه العنوان استهلاك الرموز المميزة/الحصة النسبية للمطور. بدون الحماية، تكون واجهات برمجة التطبيقات (API) عرضة لإساءة الاستخدام والحقن الفوري وإساءة الاستخدام.
+**How OmniRoute solves it:**
-**كيف يحل OmniRoute المشكلة:**
+- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free)
+- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix
+- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime
+- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits)
+- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider
--**إدارة مفاتيح واجهة برمجة التطبيقات**— الإنشاء والتدوير وتحديد النطاق لكل مزود من خلال صفحة `/dashboard/api-manager` المخصصة
--**أذونات على مستوى النموذج**— تقييد مفاتيح واجهة برمجة التطبيقات (API) على نماذج محددة (`openai/*`، أنماط أحرف البدل)، مع تبديل السماح للكل/تقييد
--**API Endpoint Protection**— اطلب مفتاحًا لـ `/v1/models` واحظر موفري خدمة محددين من القائمة
--**Auth Guard + CSRF Protection**— جميع مسارات لوحة المعلومات محمية بالبرمجيات الوسيطة `withAuth` + رموز CSRF المميزة
--**محدد المعدل**— تحديد معدل لكل IP مع نوافذ قابلة للتكوين
--**تصفية IP**— القائمة المسموح بها/القائمة المحظورة للتحكم في الوصول
--**حماية الحقن الفوري**— التعقيم ضد أنماط المطالبة الضارة
--**تشفير AES-256-GCM**— بيانات الاعتماد مشفرة في حالة عدم النشاط
+
-<التفاصيل>
-🛑 6. "تعطل مزود الخدمة الخاص بي وفقدت تدفق الترميز الخاص بي"
+
+🔒 5. "I need to protect my AI gateway from unauthorized access"
-يمكن أن يصبح موفرو الذكاء الاصطناعي غير مستقرين، أو يعرضون أخطاء 5xx، أو يصلون إلى حدود المعدلات المؤقتة. إذا كان أحد المطورين يعتمد على موفر واحد، فسيتم مقاطعته. بدون قواطع الدائرة، يمكن أن تؤدي عمليات إعادة المحاولة المتكررة إلى تعطل التطبيق.
+When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse.
-**كيف يحل OmniRoute المشكلة:**
+**How OmniRoute solves it:**
--**قاطع الدائرة لكل نموذج**— فتح/إغلاق تلقائي مع حدود قابلة للتكوين وفترة تهدئة (مغلق/مفتوح/نصف مفتوح)، محدد النطاق لكل نموذج لتجنب الكتل المتتالية
--**التراجع الأسي**— تأخير إعادة المحاولة التدريجي
--**مكافحة الرعد القطيع**— Mutex + حماية الإشارة ضد عواصف إعادة المحاولة المتزامنة
--**السلاسل الاحتياطية المجمعة**— إذا فشل الموفر الأساسي، فسيتم دخوله تلقائيًا عبر السلسلة دون أي تدخل
--**Combo Circuit Breaker**— التعطيل التلقائي لمقدمي الخدمات الفاشلين ضمن سلسلة التحرير والسرد
--**لوحة معلومات الصحة**— مراقبة وقت التشغيل، وحالات قاطع الدائرة، وعمليات التأمين، وإحصائيات ذاكرة التخزين المؤقت، ووقت الاستجابة p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page
+- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle
+- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing
+- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens
+- **Rate Limiter** — Per-IP rate limiting with configurable windows
+- **IP Filtering** — Allowlist/blocklist for access control
+- **Prompt Injection Guard** — Sanitization against malicious prompt patterns
+- **AES-256-GCM Encryption** — Credentials encrypted at rest
-<التفاصيل>
-🔧 7. "تكوين كل أداة من أدوات الذكاء الاصطناعي أمر ممل ومتكرر"
+
-يستخدم المطورون Cursor وClaude Code وCodex CLI وOpenClaw وGemini CLI وKilo Code... تحتاج كل أداة إلى تكوين مختلف (نقطة نهاية واجهة برمجة التطبيقات، المفتاح، النموذج). تعد إعادة التكوين عند تبديل مقدمي الخدمات أو النماذج مضيعة للوقت.
+
+🛑 6. "My provider went down and I lost my coding flow"
-**كيف يحل OmniRoute المشكلة:**
+AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application.
--**لوحة تحكم أدوات CLI**— صفحة مخصصة مع إعداد بنقرة واحدة لـ Claude Code، وCodex CLI، وOpenClaw، وKilo Code، وAntigravity، وCline
--**GitHub Copilot Config Generator**— يُنشئ `chatLanguageModels.json` لرمز VS مع تحديد نموذج مجمع
--**معالج الإعداد**— إعداد إرشادي من 4 خطوات للمستخدمين لأول مرة
--**نقطة نهاية واحدة، جميع النماذج**— قم بتكوين `http://localhost:20128/v1` مرة واحدة، والوصول إلى أكثر من 60 موفرًا
+**How OmniRoute solves it:**
-<التفاصيل>
-🔑 8. "إدارة رموز OAuth المميزة من موفري خدمات متعددين أمر جحيم"
+- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks
+- **Exponential Backoff** — Progressive retry delays
+- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms
+- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention
+- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain
+- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency
-Claude Code، وCodex، وGemini CLI، وCopilot — جميعهم يستخدمون OAuth 2.0 مع الرموز المميزة التي تنتهي صلاحيتها. يحتاج المطورون إلى إعادة المصادقة باستمرار، والتعامل مع "سر_العميل مفقود"، و"إعادة توجيه_uri_mismatch"، وحالات الفشل على الخوادم البعيدة. يمثل OAuth على LAN/VPS مشكلة بشكل خاص.
+
-**كيف يحل OmniRoute المشكلة:**
+
+🔧 7. "Configuring each AI tool is tedious and repetitive"
--**التحديث التلقائي للرمز المميز**— يتم تحديث رموز OAuth المميزة في الخلفية قبل انتهاء الصلاحية
--**OAuth 2.0 (PKCE) مدمج**— التدفق التلقائي لـ Claude Code وCodex وGemini CLI وCopilot وKiro وQwen وQoder
--**OAuth متعدد الحسابات**— حسابات متعددة لكل مزود عبر استخراج الرمز المميز JWT/ID
--**OAuth LAN/Remote Fix**— اكتشاف IP الخاص لـ `redirect_uri` + وضع URL اليدوي للخوادم البعيدة
--**OAuth Behind Nginx**— يستخدم window.location.origin للتوافق العكسي مع الوكيل
--**دليل OAuth عن بعد**— دليل خطوة بخطوة لبيانات اعتماد Google Cloud على VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time.
-<التفاصيل>
-📊 9. "لا أعرف كم أنفق أو أين"
+**How OmniRoute solves it:**
-يستخدم المطورون العديد من مقدمي الخدمات المدفوعة ولكن ليس لديهم رؤية موحدة للإنفاق. يمتلك كل مزود خدمة لوحة تحكم الفوترة الخاصة به، ولكن لا يوجد عرض موحد. التكاليف غير المتوقعة يمكن أن تتراكم.
+- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
+- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection
+- **Onboarding Wizard** — Guided 4-step setup for first-time users
+- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers
-**كيف يحل OmniRoute المشكلة:**
+
--**لوحة معلومات تحليلات التكلفة**— تتبع التكلفة لكل رمز مميز وإدارة الميزانية لكل مزود
--**حدود الميزانية لكل طبقة**— سقف الإنفاق لكل طبقة يؤدي إلى حدوث تراجع تلقائي
--**تكوين التسعير لكل نموذج**— أسعار قابلة للتكوين لكل نموذج
--**إحصاءات الاستخدام لكل مفتاح API**— عدد الطلبات والطابع الزمني الأخير المستخدم لكل مفتاح
--**لوحة التحكم التحليلية**— بطاقات الإحصائيات، ومخطط استخدام النموذج، وجدول الموفر مع معدلات النجاح وزمن الوصول
+
+🔑 8. "Managing OAuth tokens from multiple providers is hell"
-<التفاصيل>
-🐛 10. "لا أستطيع تشخيص الأخطاء والمشكلات في مكالمات الذكاء الاصطناعي"
+Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic.
-عندما تفشل المكالمة، لا يعرف المطور ما إذا كان هناك حد للسعر، أو رمز مميز منتهي الصلاحية، أو تنسيق خاطئ، أو خطأ في الموفر. سجلات مجزأة عبر محطات مختلفة. وبدون إمكانية الملاحظة، يكون تصحيح الأخطاء عبارة عن تجربة وخطأ.
+**How OmniRoute solves it:**
-**كيف يحل OmniRoute المشكلة:**
+- **Auto Token Refresh** — OAuth tokens refresh in background before expiration
+- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
+- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction
+- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers
+- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility
+- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker
--**لوحة تحكم السجلات الموحدة**— 4 علامات تبويب: سجلات الطلبات، وسجلات الوكيل، وسجلات التدقيق، ووحدة التحكم
--**عارض سجل وحدة التحكم**— عارض بنمط المحطة الطرفية في الوقت الفعلي مع مستويات مرمزة بالألوان، والتمرير التلقائي، والبحث، والتصفية
--**سجلات وكيل SQLite**— السجلات المستمرة التي تستمر حتى بعد إعادة تشغيل الخادم
--**ساحة المترجم**— 4 أوضاع لتصحيح الأخطاء: ساحة اللعب (ترجمة التنسيق)، اختبار الدردشة (ذهابًا وإيابًا)، منصة الاختبار (دفعة)، المراقبة المباشرة (في الوقت الفعلي)
--**قياس الطلب عن بعد**— زمن الاستجابة p50/p95/p99 + تتبع معرف طلب X
--**التسجيل المستند إلى الملف مع التدوير**— يتم تدوير سجلات التطبيق حسب الحجم وأيام الاحتفاظ وعدد الأرشيف؛ يتم تدوير عناصر سجل المكالمات حسب أيام الاحتفاظ وعدد الملفات
--**تقرير معلومات النظام**— يُنشئ `npm run system-info` ملف `system-info.txt` مع بيئتك الكاملة (إصدار Node، إصدار OmniRoute، نظام التشغيل، أدوات CLI، حالة Docker/PM2). قم بإرفاقه عند الإبلاغ عن مشكلات للفرز الفوري.
+
-<التفاصيل>
-🏗️ 11. "إن نشر البوابة وصيانتها أمر معقد"
+
+📊 9. "I don't know how much I'm spending or where"
-يعد تثبيت وكيل AI وتكوينه وصيانته عبر بيئات مختلفة (محلية، VPS، Docker، سحابية) عملية كثيفة العمالة. مشاكل مثل المسارات المضمنة، EACCES في الدلائل، وتعارضات المنافذ، والبنيات عبر الأنظمة الأساسية تزيد من الاحتكاك.
+Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up.
-**كيف يحل OmniRoute المشكلة:**
+**How OmniRoute solves it:**
--**تثبيت npm الشامل**— `npm install -g omniroute && omniroute` - تم
--**منصة Docker المتعددة**— AMD64 + ARM64 الأصلي (Apple Silicon، AWS Graviton، Raspberry Pi)
--**Docker Compose Profiles**— `base` (بدون أدوات CLI) و`cli` (مع Claude Code، وCodex، وOpenClaw)
--**Electron Desktop App**— تطبيق أصلي لنظام التشغيل Windows/macOS/Linux مع علبة النظام، والتشغيل التلقائي، ووضع عدم الاتصال
--**وضع المنفذ المقسم**— واجهة برمجة التطبيقات ولوحة المعلومات على منافذ منفصلة للسيناريوهات المتقدمة (الوكيل العكسي، وشبكات الحاويات)
--**Cloud Sync**— مزامنة التكوين عبر الأجهزة عبر Cloudflare Workers
--**النسخ الاحتياطية لقاعدة البيانات**— النسخ الاحتياطي التلقائي لجميع الإعدادات واستعادتها وتصديرها واستيرادها، باستخدام `DISABLE_SQLITE_AUTO_BACKUP` للنسخ الاحتياطية المُدارة خارجيًا
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider
+- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback
+- **Per-Model Pricing Configuration** — Configurable prices per model
+- **Usage Statistics Per API Key** — Request count and last-used timestamp per key
+- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency
-<التفاصيل>
-🌍 12. "الواجهة باللغة الإنجليزية فقط وفريقي لا يتحدث الإنجليزية"
+
-تواجه الفرق في البلدان غير الناطقة باللغة الإنجليزية، وخاصة في أمريكا اللاتينية وآسيا وأوروبا، صعوبة في التعامل مع الواجهات التي تستخدم اللغة الإنجليزية فقط. تعمل حواجز اللغة على تقليل الاعتماد وزيادة أخطاء التكوين.
+
+🐛 10. "I can't diagnose errors and problems in AI calls"
-**كيف يحل OmniRoute المشكلة:**
+When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error.
--**لوحة المعلومات i18n — 30 لغة**— أكثر من 500 مفتاح مترجم بما في ذلك العربية والبلغارية والدنماركية والألمانية والإسبانية والفنلندية والفرنسية والعبرية والهندية والمجرية والإندونيسية والإيطالية واليابانية والكورية والماليزية والهولندية والنرويجية والبولندية والبرتغالية (PT/BR) والرومانية والروسية والسلوفاكية والسويدية والتايلاندية والأوكرانية والفيتنامية والصينية والفلبينية والإنجليزية
--**دعم RTL**— دعم من اليمين إلى اليسار للغتين العربية والعبرية
--**الملفات التمهيدية متعددة اللغات**— 30 ترجمة كاملة للوثائق
--**محدد اللغة**— رمز الكرة الأرضية في رأس الصفحة للتبديل في الوقت الفعلي
+**How OmniRoute solves it:**
-<التفاصيل>
-🔄 13. "أحتاج إلى أكثر من مجرد الدردشة - أحتاج إلى التضمين والصور والصوت"
+- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console
+- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter
+- **SQLite Proxy Logs** — Persistent logs that survive server restarts
+- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time)
+- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing
+- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count
+- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage.
-الذكاء الاصطناعي ليس مجرد استكمال للدردشة. يحتاج المطورون إلى إنشاء صور، ونسخ الصوت، وإنشاء تضمينات لـ RAG، وإعادة ترتيب المستندات، والإشراف على المحتوى. تحتوي كل واجهة برمجة تطبيقات على نقطة نهاية وتنسيق مختلفين.
+
-**كيف يحل OmniRoute المشكلة:**
+
+🏗️ 11. "Deploying and maintaining the gateway is complex"
--**Embeddings**— `/v1/embeddings` مع 6 موفري خدمات وأكثر من 9 نماذج
--**إنشاء الصور**— `/v1/images/ Generations` مع 10 موفرين وأكثر من 20 نموذجًا (OpenAI، وxAI، وTogether، وFireworks، وNebius، وHyperbolic، وNanoBanana، وAntigravity، وSD WebUI، وComfyUI)
--**تحويل النص إلى فيديو**— `/v1/videos/أجيال` — ComfyUI (AnimateDiff، SVD) وSD WebUI
--**تحويل النص إلى موسيقى**— `/v1/music/generations` — ComfyUI (صوت ثابت مفتوح، MusicGen)
--**نسخ الصوت**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM، HuggingFace، Qwen3
--**تحويل النص إلى كلام**— `/v1/audio/speech` — ElevenLabs، Nvidia NIM، HuggingFace، Coqui، Tortoise، Qwen3،**Inworld**،**Cartesia**،**PlayHT**، + مقدمي الخدمة الحاليين
--**الإشراف**— `/v1/moderations` — التحقق من سلامة المحتوى
--**إعادة الترتيب**— `/v1/rerank` — إعادة ترتيب مدى ملاءمة الوثيقة
--**Responses API**— الدعم الكامل `/v1/responses` لـ Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction.
-<التفاصيل>
-🧪 14. "ليس لدي طريقة لاختبار ومقارنة الجودة عبر النماذج"
+**How OmniRoute solves it:**
-يرغب المطورون في معرفة النموذج الأفضل لحالة الاستخدام الخاصة بهم - التعليمات البرمجية، والترجمة، والتفكير - ولكن المقارنة يدويًا بطيئة. لا توجد أدوات تقييم متكاملة.
+- **npm global install** — `npm install -g omniroute && omniroute` — done
+- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi)
+- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw)
+- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode
+- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking)
+- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers
+- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups
-**كيف يحل OmniRoute المشكلة:**
+
--**تقييمات LLM**— اختبار المجموعة الذهبية مع 10 حالات محملة مسبقًا تغطي التحيات، والرياضيات، والجغرافيا، وإنشاء التعليمات البرمجية، والامتثال لـ JSON، والترجمة، وتخفيض السعر، والرفض الآمن
--**4 إستراتيجيات المطابقة**— `exact`، `contains`، `regex`، `custom` (وظيفة JS)
--**منصة اختبار ساحة المترجم**— اختبار الدفعات بمدخلات متعددة ومخرجات متوقعة، ومقارنة بين الموفرين
--**أداة اختبار الدردشة**— رحلة ذهابًا وإيابًا كاملة مع عرض الاستجابة المرئية
--**المراقبة المباشرة**— البث المباشر لجميع الطلبات المتدفقة عبر الوكيل
+
+🌍 12. "The interface is English-only and my team doesn't speak English"
-<التفاصيل>
-📈 15. "أحتاج إلى التوسع دون فقدان الأداء"
+Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors.
-مع نمو حجم الطلب، يؤدي عدم التخزين المؤقت لنفس الأسئلة إلى توليد تكاليف مكررة. دون العجز، طلبات مكررة معالجة النفايات. يجب احترام حدود الأسعار لكل مزود.
+**How OmniRoute solves it:**
-**كيف يحل OmniRoute المشكلة:**
+- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English
+- **RTL Support** — Right-to-left support for Arabic and Hebrew
+- **Multi-Language READMEs** — 30 complete documentation translations
+- **Language Selector** — Globe icon in header for real-time switching
--**ذاكرة التخزين المؤقت الدلالية**— تعمل ذاكرة التخزين المؤقت ذات المستويين (التوقيع + الدلالي) على تقليل التكلفة ووقت الاستجابة
--**صلاحية الطلب**— نافذة إلغاء البيانات المكررة لمدة 5 ثوانٍ للطلبات المتماثلة
--**الكشف عن حدود المعدل**— عدد الدورات في الدقيقة لكل مزود، والفجوة الدنيا، والحد الأقصى للتتبع المتزامن
--**حدود المعدل القابلة للتحرير**— الإعدادات الافتراضية القابلة للتكوين في الإعدادات → المرونة مع الثبات
--**ذاكرة التخزين المؤقت للتحقق من صحة مفتاح واجهة برمجة التطبيقات**— ذاكرة تخزين مؤقت ثلاثية الطبقات لأداء الإنتاج
--**لوحة معلومات الصحة مع القياس عن بعد**— زمن الاستجابة p50/p95/p99، وإحصائيات ذاكرة التخزين المؤقت، ووقت التشغيل
+
-<التفاصيل>
-🤖 16. "أريد التحكم في سلوك النموذج عالميًا"
+
+🔄 13. "I need more than chat — I need embeddings, images, audio"
-المطورون الذين يريدون جميع الاستجابات بلغة معينة، بنبرة معينة، أو يريدون الحد من الرموز المميزة للاستدلال. يعد تكوين هذا في كل أداة/طلب أمرًا غير عملي.
+AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format.
-**كيف يحل OmniRoute المشكلة:**
+**How OmniRoute solves it:**
--**الحقن الفوري للنظام**— يتم تطبيق المطالبة العامة على جميع الطلبات
--**التحقق من صحة ميزانية التفكير**— التحكم في تخصيص الرمز المميز لكل طلب (العبور، التلقائي، المخصص، التكيفي)
--**9 استراتيجيات التوجيه**— استراتيجيات عالمية تحدد كيفية توزيع الطلبات
--**Wildcard Router**— يتم توجيه أنماط `المزود/*` ديناميكيًا إلى أي مزود
--**تبديل تمكين/تعطيل التحرير والسرد**— تبديل المجموعات مباشرة من لوحة المعلومات
--**تبديل الموفر**— تمكين/تعطيل جميع اتصالات الموفر بنقرة واحدة
--**موفري الخدمة المحظورون**— استبعاد موفري خدمة محددين من قائمة `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models
+- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
+- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI
+- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
+- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
+- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers
+- **Moderations** — `/v1/moderations` — Content safety checks
+- **Reranking** — `/v1/rerank` — Document relevance reranking
+- **Responses API** — Full `/v1/responses` support for Codex
-<التفاصيل>
-🧰 17. "أحتاج إلى أدوات MCP كقدرات منتج من الدرجة الأولى"
+
-تعرض العديد من بوابات الذكاء الاصطناعي MCP فقط كتفاصيل تنفيذ مخفية. تحتاج الفرق إلى طبقة تشغيل مرئية ويمكن التحكم فيها.
+
+🧪 14. "I have no way to test and compare quality across models"
-**كيف يحل OmniRoute المشكلة:**
+Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist.
-- يظهر MCP في لوحة التحكم وعلامة تبويب بروتوكول نقطة النهاية
-- صفحة إدارة MCP مخصصة تحتوي على العمليات والأدوات والنطاقات والتدقيق
-- بداية سريعة مدمجة لـ `omniroute --mcp` وتأهيل العميل
+**How OmniRoute solves it:**
-<التفاصيل>
-🧠 18. "أحتاج إلى تنسيق A2A مع مسارات المهام المتزامنة والدفقية"
+- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal
+- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function)
+- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison
+- **Chat Tester** — Full round-trip with visual response rendering
+- **Live Monitor** — Real-time stream of all requests flowing through the proxy
-تحتاج مسارات عمل الوكيل إلى ردود مباشرة وتنفيذ متدفق طويل الأمد مع التحكم في دورة الحياة.
+
-**كيف يحل OmniRoute المشكلة:**
+
+📈 15. "I need to scale without losing performance"
-- نقطة نهاية A2A JSON-RPC (`POST /a2a`) مع `message/send` و`message/stream`
-- تدفق SSE مع انتشار الحالة الطرفية
-- واجهات برمجة التطبيقات الخاصة بدورة حياة المهام لـ "المهام/الحصول" و"المهام/الإلغاء".
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected.
-<التفاصيل>
-🛰️ 19. "أحتاج إلى صحة عملية MCP حقيقية، وليس حالة تخمينية"
+**How OmniRoute solves it:**
-تحتاج الفرق التشغيلية إلى معرفة ما إذا كان MCP حيًا بالفعل، وليس فقط ما إذا كان يمكن الوصول إلى واجهة برمجة التطبيقات (API).
+- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency
+- **Request Idempotency** — 5s deduplication window for identical requests
+- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking
+- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence
+- **API Key Validation Cache** — 3-tier cache for production performance
+- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime
-**كيف يحل OmniRoute المشكلة:**
+
-- ملف نبضات وقت التشغيل مع PID والطوابع الزمنية والنقل وعدد الأدوات ووضع النطاق
-- واجهة برمجة تطبيقات حالة MCP التي تجمع بين نبضات القلب + النشاط الأخير
-- بطاقات حالة واجهة المستخدم للعملية/وقت التشغيل/نضارة نبضات القلب
+
+🤖 16. "I want to control model behavior globally"
-<التفاصيل>
-📋 20. "أحتاج إلى تنفيذ أداة MCP قابلة للتدقيق"
+Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical.
-عندما تقوم الأدوات بتغيير التكوين أو تشغيل إجراءات العمليات، تحتاج الفرق إلى إمكانية التتبع الجنائي.
+**How OmniRoute solves it:**
-**كيف يحل OmniRoute المشكلة:**
+- **System Prompt Injection** — Global prompt applied to all requests
+- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive)
+- **9 Routing Strategies** — Global strategies that determine how requests are distributed
+- **Wildcard Router** — `provider/*` patterns route dynamically to any provider
+- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard
+- **Provider Toggle** — Enable/disable all connections for a provider with one click
+- **Blocked Providers** — Exclude specific providers from `/v1/models` listing
-- تسجيل التدقيق المدعوم من SQLite لاستدعاءات أداة MCP
-- عوامل التصفية حسب الأداة، والنجاح/الفشل، ومفتاح API، وترقيم الصفحات
-- جدول تدقيق لوحة المعلومات + إحصائيات نقاط النهاية للأتمتة
+
-<التفاصيل>
-🔐 21. "أحتاج إلى أذونات MCP محددة لكل عملية تكامل"
+
+🧰 17. "I need MCP tools as first-class product capabilities"
-يجب أن يتمتع العملاء المختلفون بإمكانية الوصول الأقل امتيازًا إلى فئات الأدوات.
+Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer.
-**كيف يحل OmniRoute المشكلة:**
+**How OmniRoute solves it:**
-- 10 نطاقات MCP محببة للتحكم في الوصول إلى الأدوات
-- إنفاذ النطاق والرؤية في واجهة مستخدم إدارة MCP
-- الوضع الافتراضي الآمن للأدوات التشغيلية
+- MCP appears in the dashboard navigation and endpoint protocol tab
+- Dedicated MCP management page with process, tools, scopes, and audit
+- Built-in quick-start for `omniroute --mcp` and client onboarding
-<التفاصيل>
-⚙️ 22. "أحتاج إلى ضوابط تشغيلية دون إعادة الانتشار"
+
-تحتاج الفرق إلى تغييرات سريعة في وقت التشغيل أثناء الحوادث أو أحداث التكلفة.
+
+🧠 18. "I need A2A orchestration with sync + stream task paths"
-**كيف يحل OmniRoute المشكلة:**
+Agent workflows need both direct replies and long-running streamed execution with lifecycle control.
-- قم بتبديل تنشيط التحرير والسرد مباشرةً من لوحة معلومات MCP
-- تطبيق ملفات تعريف المرونة من حزم السياسات المحددة مسبقًا
-- إعادة ضبط حالة قاطع الدائرة من نفس لوحة العمليات
+**How OmniRoute solves it:**
-<التفاصيل>
-🔄 23. "أحتاج إلى رؤية وإلغاء مباشر لدورة حياة مهمة A2A"
+- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream`
+- SSE streaming with terminal state propagation
+- Task lifecycle APIs for `tasks/get` and `tasks/cancel`
-وبدون رؤية دورة الحياة، يصبح من الصعب فرز حوادث المهام.
+
-**كيف يحل OmniRoute المشكلة:**
+
+🛰️ 19. "I need real MCP process health, not guessed status"
-- قائمة المهام/التصفية حسب الحالة/المهارة مع ترقيم الصفحات
-- التعمق في البيانات الوصفية للمهمة، والأحداث، والتحف
-- نقطة نهاية إلغاء المهمة وإجراء واجهة المستخدم مع التأكيد
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable.
-<التفاصيل>
-🌊 24. "أحتاج إلى مقاييس تيار نشطة لتحميل A2A"
+**How OmniRoute solves it:**
-يتطلب تدفق سير العمل رؤية تشغيلية للتزامن والاتصالات المباشرة.
+- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode
+- MCP status API combining heartbeat + recent activity
+- UI status cards for process/uptime/heartbeat freshness
-**كيف يحل OmniRoute المشكلة:**
+
-- عدادات التدفق النشطة مدمجة في حالة A2A
-- الطابع الزمني للمهمة الأخيرة وعدد كل ولاية
-- بطاقات لوحة القيادة A2A لمراقبة العمليات في الوقت الفعلي
+
+📋 20. "I need auditable MCP tool execution"
-<التفاصيل>
-🪪 25. "أحتاج إلى اكتشاف وكيل قياسي للعملاء"
+When tools mutate config or trigger ops actions, teams need forensic traceability.
-يحتاج العملاء والمنسقون الخارجيون إلى بيانات تعريف يمكن قراءتها آليًا من أجل الإعداد.
+**How OmniRoute solves it:**
-**كيف يحل OmniRoute المشكلة:**
+- SQLite-backed audit logging for MCP tool calls
+- Filters by tool, success/failure, API key, and pagination
+- Dashboard audit table + stats endpoints for automation
-- بطاقة الوكيل معروضة على `/.well-known/agent.json`
-- القدرات والمهارات الموضحة في واجهة المستخدم الإدارية
-- تتضمن واجهة برمجة التطبيقات لحالة A2A بيانات تعريف الاكتشاف للأتمتة
+
-<التفاصيل>
-🧭 26. "أحتاج إلى إمكانية اكتشاف البروتوكول في تجربة المستخدم للمنتج"
+
+🔐 21. "I need scoped MCP permissions per integration"
-إذا لم يتمكن المستخدمون من اكتشاف أسطح البروتوكول، فسوف ينخفض جودة الاعتماد والدعم.
+Different clients should have least-privilege access to tool categories.
-**كيف يحل OmniRoute المشكلة:**
+**How OmniRoute solves it:**
-- صفحة**نقاط النهاية**الموحدة مع علامات تبويب Proxy وMCP وA2A وAPI Endpoints
-- تبديل حالة الخدمة المضمنة (متصل/غير متصل) لـ MCP وA2A
-- روابط من النظرة العامة إلى علامات تبويب الإدارة المخصصة
+- 10 granular MCP scopes for controlled tool access
+- Scope enforcement and visibility in MCP management UI
+- Safe default posture for operational tooling
-<التفاصيل>
-🧪 27. "أحتاج إلى التحقق من صحة البروتوكول الشامل مع عملاء حقيقيين"
+
-الاختبارات الوهمية ليست كافية للتحقق من توافق البروتوكول قبل الإصدار.
+
+⚙️ 22. "I need operational controls without redeploying"
-**كيف يحل OmniRoute المشكلة:**
+Teams need quick runtime changes during incidents or cost events.
-- مجموعة E2E التي تعمل على تشغيل التطبيق وتستخدم نقل عميل MCP SDK الحقيقي
-- اختبارات عميل A2A لاكتشاف التدفقات وإرسالها ودفقها والحصول عليها وإلغائها
-- التحقق من التأكيدات ضد تدقيق MCP وواجهات برمجة تطبيقات مهام A2A
+**How OmniRoute solves it:**
-<التفاصيل>
-📡 28. "أحتاج إلى إمكانية ملاحظة موحدة عبر جميع الواجهات"
+- Switch combo activation directly from MCP dashboard
+- Apply resilience profiles from pre-defined policy packs
+- Reset circuit breaker state from the same operations panel
-يؤدي تقسيم إمكانية المراقبة حسب البروتوكول إلى إنشاء نقاط عمياء وMTTR أطول.
+
-**كيف يحل OmniRoute المشكلة:**
+
+🔄 23. "I need live A2A task lifecycle visibility and cancellation"
-- لوحات معلومات/سجلات/تحليلات موحدة في منتج واحد
-- الصحة + التدقيق + طلب القياس عن بعد عبر طبقات OpenAI وMCP وA2A
-- واجهات برمجة التطبيقات التشغيلية للحالة والأتمتة
+Without lifecycle visibility, task incidents become hard to triage.
-<التفاصيل>
-💼 29. "أحتاج إلى وقت تشغيل واحد للوكيل + الأدوات + تنسيق الوكيل"
+**How OmniRoute solves it:**
-يؤدي تشغيل العديد من الخدمات المنفصلة إلى زيادة تكلفة التشغيل وأوضاع الفشل.
+- Task listing/filtering by state/skill with pagination
+- Drill-down on task metadata, events, and artifacts
+- Task cancellation endpoint and UI action with confirmation
-**كيف يحل OmniRoute المشكلة:**
+
-- وكيل متوافق مع OpenAI وخادم MCP وخادم A2A في مكدس واحد
-- المصادقة المشتركة والمرونة وتخزين البيانات وإمكانية الملاحظة
-- نموذج سياسة متسق عبر جميع أسطح التفاعل
+
+🌊 24. "I need active stream metrics for A2A load"
-<التفاصيل>
-🚀 30. "أحتاج إلى إرسال مهام سير عمل الوكيل دون امتداد التعليمات البرمجية اللاصقة"
+Streaming workflows require operational insight into concurrency and live connections.
-تفقد الفرق سرعتها عند دمج العديد من الخدمات والبرامج النصية المخصصة.
+**How OmniRoute solves it:**
-**كيف يحل OmniRoute المشكلة:**
+- Active stream counters integrated into A2A status
+- Last task timestamp and per-state counts
+- A2A dashboard cards for real-time ops monitoring
-- استراتيجية نقطة النهاية الموحدة للعملاء والوكلاء
-- واجهات مستخدم لإدارة البروتوكول مدمجة ومسارات التحقق من صحة الدخان
-- أسس جاهزة للإنتاج (الأمان، التسجيل، المرونة، النسخ الاحتياطي)
+
+
+
+🪪 25. "I need standard agent discovery for clients"
+
+External clients and orchestrators need machine-readable metadata for onboarding.
+
+**How OmniRoute solves it:**
+
+- Agent Card exposed at `/.well-known/agent.json`
+- Capabilities and skills shown in management UI
+- A2A status API includes discovery metadata for automation
+
+
+
+
+🧭 26. "I need protocol discoverability in the product UX"
+
+If users cannot discover protocol surfaces, adoption and support quality drop.
+
+**How OmniRoute solves it:**
+
+- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints
+- Inline service status toggles (Online/Offline) for MCP and A2A
+- Links from overview to dedicated management tabs
+
+
+
+
+🧪 27. "I need end-to-end protocol validation with real clients"
+
+Mock tests are not enough to validate protocol compatibility before release.
+
+**How OmniRoute solves it:**
+
+- E2E suite that boots app and uses real MCP SDK client transport
+- A2A client tests for discovery, send, stream, get, and cancel flows
+- Cross-check assertions against MCP audit and A2A tasks APIs
+
+
+
+
+📡 28. "I need unified observability across all interfaces"
+
+Splitting observability by protocol creates blind spots and longer MTTR.
+
+**How OmniRoute solves it:**
+
+- Unified dashboards/logs/analytics in one product
+- Health + audit + request telemetry across OpenAI, MCP, and A2A layers
+- Operational APIs for status and automation
+
+
+
+
+💼 29. "I need one runtime for proxy + tools + agent orchestration"
+
+Running many separate services increases operational cost and failure modes.
+
+**How OmniRoute solves it:**
+
+- OpenAI-compatible proxy, MCP server, and A2A server in one stack
+- Shared auth, resilience, data store, and observability
+- Consistent policy model across all interaction surfaces
+
+
+
+
+🚀 30. "I need to ship agentic workflows without glue-code sprawl"
+
+Teams lose velocity when stitching multiple ad-hoc services and scripts.
+
+**How OmniRoute solves it:**
+
+- Unified endpoint strategy for clients and agents
+- Built-in protocol management UIs and smoke validation paths
+- Production-ready foundations (security, logging, resilience, backup)
+
+
### Example Playbooks (Integrated Use Cases)
-**قواعد اللعبة أ: زيادة الاشتراك المدفوع إلى الحد الأقصى + نسخة احتياطية رخيصة**```txt
+**Playbook A: Maximize paid subscription + cheap backup**
+
+```txt
Combo: "maximize-claude"
1. cc/claude-opus-4-6
2. glm/glm-4.7
@@ -608,21 +690,23 @@ Combo: "maximize-claude"
Monthly cost: $20 + small backup spend
Outcome: higher quality, near-zero interruption
-````
+```
-**دليل التشغيل ب: مكدس البرمجة بدون تكلفة**```txt
+**Playbook B: Zero-cost coding stack**
+
+```txt
Combo: "free-forever"
-
-1. gc/gemini-3-flash
-2. if/kimi-k2-thinking
-3. qw/qwen3-coder-plus
+ 1. gc/gemini-3-flash
+ 2. if/kimi-k2-thinking
+ 3. qw/qwen3-coder-plus
Monthly cost: $0
Outcome: stable free coding workflow
+```
-````
+**Playbook C: 24/7 always-on fallback chain**
-**Playbook C: سلسلة احتياطية متاحة دائمًا على مدار 24 ساعة طوال أيام الأسبوع**```txt
+```txt
Combo: "always-on"
1. cc/claude-opus-4-6
2. cx/gpt-5.2-codex
@@ -631,122 +715,134 @@ Combo: "always-on"
5. if/kimi-k2-thinking
Outcome: deep fallback depth for deadline-critical workloads
-````
+```
-**قواعد اللعبة د: عمليات العميل مع MCP + A2A**```txt
+**Playbook D: Agent ops with MCP + A2A**
-1. Start MCP transport (`omniroute --mcp`) for tool-driven operations
-2. Run A2A tasks via `message/send` and `message/stream`
-3. Observe via /dashboard/endpoint (MCP and A2A tabs)
-4. Toggle services via inline status controls
-
-````
+```txt
+1) Start MCP transport (`omniroute --mcp`) for tool-driven operations
+2) Run A2A tasks via `message/send` and `message/stream`
+3) Observe via /dashboard/endpoint (MCP and A2A tabs)
+4) Toggle services via inline status controls
+```
---
## 🆓 Start Free — Zero Configuration Cost
-> قم بإعداد ترميز الذكاء الاصطناعي في دقائق بسعر**$0/الشهر**. قم بتوصيل هذه الحسابات المجانية واستخدم المجموعة المدمجة**Free Stack**.
+> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo.
-| خطوة | العمل | مقدمي الخدمات مقفلة |
+| Step | Action | Providers Unlocked |
| ---- | -------------------------------------------------- | ------------------------------------------------------------------ |
-| 1 | الاتصال**Kiro**(معرف AWS Builder OAuth) | كلود سونيت 4.5، هايكو 4.5 —**غير محدود**|
-| 2 | ربط**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, Deepseek-r1... —**غير محدود**|
-| 3 | ربط**كوين**(رمز الجهاز) | qwen3-coder-plus، qwen3-coder-flash... —**غير محدود**|
-| 4 | الاتصال**Gemini CLI**(Google OAuth) | gemini-3-flash,gemini-2.5-pro —**180 ألف/الشهر مجانًا**|
-| 5 | `/dashboard/combos` →**قالب مكدس مجاني ($0)**| جولة روبن لجميع مقدمي الخدمات المجانية تلقائيًا |
+| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** |
+| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** |
+| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** |
+| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** |
+| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically |
-**قم بتوجيه أي IDE/CLI إلى:**`http://localhost:20128/v1` · مفتاح API: `any-string` · تم.
+**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done.
->**تغطية إضافية اختيارية (مجانية أيضًا):**مفتاح Groq API (30 دورة في الدقيقة مجانًا)، NVIDIA NIM (40 دورة في الدقيقة مجانًا، أكثر من 70 طرازًا)، Cerebras (1 مليون tok/يوم)، مفتاح LongCat API (50 مليون رمز مميز/يوم!)، Cloudflare Workers AI (10 آلاف خلية عصبية/يوم، أكثر من 50 نموذجًا).## بداية سريعة
+> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models).
+
+## بداية سريعة
### 1) Install and run
```bash
npm install -g omniroute
omniroute
-````
+```
-> **مستخدمي pnpm:**قم بتشغيل `pnpmوافق-builds -g` بعد التثبيت لتمكين البرامج النصية للبناء الأصلي المطلوبة من قبل `better-sqlite3` و`@swc/core`:
+> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`:
>
-> ```باش
-> تثبيت pnpm -g في كل الاتجاهات
-> pnpm Approved-builds -g # حدد جميع الحزم → الموافقة
-> الطريق الشامل
+> ```bash
+> pnpm install -g omniroute
+> pnpm approve-builds -g # Select all packages → approve
+> omniroute
> ```
-تفتح لوحة المعلومات على `http://localhost:20128` ويكون عنوان URL الأساسي لواجهة برمجة التطبيقات هو `http://localhost:20128/v1`.
+Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`.
-| الأمر | الوصف |
-| ----------------------------- | ------------------------------------------------------------------------------------- |
-| "الطريق الشامل" | بدء تشغيل الخادم (`PORT=20128` وواجهة برمجة التطبيقات ولوحة المعلومات على نفس المنفذ) |
-| `الطريق الشامل --المنفذ 3000` | اضبط منفذ Canonical/API على 3000 |
-| `الطريق الشامل --mcp` | بدء تشغيل خادم MCP (نقل stdio) |
-| `الطريق الشامل --no-open` | لا تفتح المتصفح تلقائيًا |
-| `الطريق الشامل --مساعدة` | عرض المساعدة |
+| Command | Description |
+| ----------------------- | ----------------------------------------------------------- |
+| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) |
+| `omniroute --port 3000` | Set canonical/API port to 3000 |
+| `omniroute --mcp` | Start MCP server (stdio transport) |
+| `omniroute --no-open` | Don't auto-open browser |
+| `omniroute --help` | Show help |
-وضع المنفذ المقسم الاختياري:```bash
+Optional split-port mode:
+
+```bash
PORT=20128 DASHBOARD_PORT=20129 omniroute
-
-# API: http://localhost:20128/v1
-
+# API: http://localhost:20128/v1
# Dashboard: http://localhost:20129
-
-````
+```
### Long-Running Streaming Timeouts
-بالنسبة لمعظم عمليات النشر، تحتاج فقط إلى:
+For most deployments, you only need:
-| متغير | الافتراضي | الغرض |
-| ------------------------ | ----------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
-| `REQUEST_TIMEOUT_MS` | `600000` | خط أساسي مشترك للجلب الأولي، ومهلات Undici المخفية، وطلبات بصمة TLS، ومهلة طلب/وكيل جسر واجهة برمجة التطبيقات |
-| `STREAM_IDLE_TIMEOUT_MS` | يرث `REQUEST_TIMEOUT_MS` | الحد الأقصى للفجوة بين قطع الدفق قبل أن يقوم OmniRoute بإحباط دفق SSE |
+| Variable | Default | Purpose |
+| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- |
+| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts |
+| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream |
-يتم الحفاظ على التوافق مع الإصدارات السابقة: لا تزال متغيرات مهلة `FETCH_TIMEOUT_MS` وAPI_BRIDGE_PROXY_TIMEOUT_MS الموجودة ومتغيرات المهلة الأخرى لكل طبقة تعمل وتتجاوز الخط الأساسي المشترك.
+Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline.
-تتوفر التجاوزات المتقدمة إذا كنت بحاجة إلى تحكم أفضل:| متغير | الافتراضي | الغرض |
+Advanced overrides are available if you need finer control:
+
+| Variable | Default | Purpose |
| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- |
-| `FETCH_TIMEOUT_MS` | يرث `REQUEST_TIMEOUT_MS` | إجمالي مهلة طلب المنبع المستخدمة بواسطة إشارة إحباط الجلب الرئيسية |
-| `FETCH_HEADERS_TIMEOUT_MS` | يرث `FETCH_TIMEOUT_MS` | الحد الزمني لـ Undici لتلقي رؤوس الاستجابة الأولية |
-| `FETCH_BODY_TIMEOUT_MS` | يرث `FETCH_TIMEOUT_MS` | الحد الزمني Undici بين قطع النص الأساسي (`0` يعطله) |
-| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP مهلة الاتصال |
-| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici مهلة مأخذ التوصيل الخامل |
-| `TLS_CLIENT_TIMEOUT_MS` | يرث `FETCH_TIMEOUT_MS` | انتهت المهلة لطلبات بصمة TLS التي تم إجراؤها من خلال `wreq-js` |
-| `API_BRIDGE_PROXY_TIMEOUT_MS` | يرث `REQUEST_TIMEOUT_MS` أو `30000` | انتهت المهلة لإعادة توجيه الوكيل `/v1` من منفذ API إلى منفذ لوحة المعلومات |
-| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `الحد الأقصى (API_BRIDGE_PROXY_TIMEOUT_MS، 300000)` | انتهت مهلة الطلب الوارد على خادم جسر API |
-| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | انتهت مهلة الرأس الوارد على خادم جسر API |
-| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | مهلة البقاء على قيد الحياة على خادم جسر API |
-| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | انتهت مهلة عدم نشاط مأخذ التوصيل على خادم جسر واجهة برمجة التطبيقات (`0` يعطله) |
+| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal |
+| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers |
+| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) |
+| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout |
+| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout |
+| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` |
+| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port |
+| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server |
+| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server |
+| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server |
+| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) |
-إذا قمت بتشغيل OmniRoute خلف Nginx أو Caddy أو Cloudflare أو وكيل عكسي آخر، فتأكد من الوكيل
-تعد المهلات أيضًا أعلى من مهلات البث/الجلب في OmniRoute.### 2) Connect providers and create your API key
+If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy
+timeouts are also higher than your OmniRoute stream/fetch timeouts.
-1. افتح لوحة المعلومات → "الموفرون" وقم بتوصيل موفر واحد على الأقل (مفتاح OAuth أو API).
-2. افتح لوحة المعلومات ← "نقاط النهاية" وأنشئ مفتاح واجهة برمجة التطبيقات.
-3. (اختياري) افتح لوحة المعلومات → `المجموعات` وقم بتعيين السلسلة الاحتياطية.### 3) Point your coding tool to OmniRoute
+### 2) Connect providers and create your API key
+
+1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key).
+2. Open Dashboard → `Endpoints` and create an API key.
+3. (Optional) Open Dashboard → `Combos` and set your fallback chain.
+
+### 3) Point your coding tool to OmniRoute
```txt
Base URL: http://localhost:20128/v1
API Key: [copy from Endpoint page]
Model: if/kimi-k2-thinking (or any provider/model prefix)
-````
+```
-يعمل مع Claude Code، وCodex CLI، وGemini CLI، وCursor، وCline، وOpenClaw، وOpenCode، وحزم SDK المتوافقة مع OpenAI.### 4) Enable and validate protocols (v2.0)
+Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs.
-**MCP (للعمليات التي تعتمد على الأدوات):**```bash
+### 4) Enable and validate protocols (v2.0)
+
+**MCP (for tool-driven operations):**
+
+```bash
omniroute --mcp
+```
-````
-
-ثم قم بتوصيل عميل MCP الخاص بك عبر أدوات "stdio" واختبار مثل:
+Then connect your MCP client over `stdio` and test tools like:
- `omniroute_get_health`
- `omniroute_list_combos`
-**A2A (لسير العمل من وكيل إلى وكيل):**```bash
+**A2A (for agent-to-agent workflows):**
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
```bash
curl -X POST http://localhost:20128/a2a \
@@ -760,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \
npm run test:protocols:e2e
```
-يتحقق هذا الجناح من تدفقات عميل MCP وA2A الحقيقية مقابل تطبيق قيد التشغيل.### Alternative: run from source
+This suite validates real MCP and A2A client flows against a running app.
+
+### Alternative: run from source
```bash
cp .env.example .env
@@ -768,14 +866,13 @@ npm install
PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev
```
-<التفاصيل>
+
+Void Linux (`xbps-src` template)
-إبطال Linux (قالب `xbps-src`)
-
-بالنسبة لمستخدمي Void Linux، يمكنك إنشاء حزمة أصلية باستخدام `xbps-src`. احفظ هذه الكتلة باسم `srcpkgs/omniroute/template`:```bash
+For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`:
+```bash
# Template file for 'omniroute'
-
pkgname=omniroute
version=3.4.1
revision=1
@@ -787,7 +884,7 @@ license="MIT"
homepage="https://github.com/diegosouzapw/OmniRoute"
distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz"
checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b
-system_accounts="\_omniroute"
+system_accounts="_omniroute"
omniroute_homedir="/var/lib/omniroute"
export NODE_ENV=production
export npm_config_engine_strict=false
@@ -795,71 +892,70 @@ export npm_config_loglevel=error
export npm_config_fund=false
export npm_config_audit=false
-do_build() { # Determine target CPU arch for node-gyp
-local \_gyp_arch
-case "$XBPS_TARGET_MACHINE" in
-aarch64*) \_gyp_arch=arm64 ;;
-armv7*|armv6*) \_gyp_arch=arm ;;
-i686*) \_gyp_arch=ia32 ;;
-\*) \_gyp_arch=x64 ;;
-esac
+do_build() {
+ # Determine target CPU arch for node-gyp
+ local _gyp_arch
+ case "$XBPS_TARGET_MACHINE" in
+ aarch64*) _gyp_arch=arm64 ;;
+ armv7*|armv6*) _gyp_arch=arm ;;
+ i686*) _gyp_arch=ia32 ;;
+ *) _gyp_arch=x64 ;;
+ esac
- # 1) Install all deps – skip scripts (no network in do_build, native modules
- # compiled separately below; better-sqlite3 is serverExternalPackage so
- # Next.js does not execute it during next build)
- NODE_ENV=development npm ci --ignore-scripts
+ # 1) Install all deps – skip scripts (no network in do_build, native modules
+ # compiled separately below; better-sqlite3 is serverExternalPackage so
+ # Next.js does not execute it during next build)
+ NODE_ENV=development npm ci --ignore-scripts
- # 2) Build the Next.js standalone bundle
- npm run build
+ # 2) Build the Next.js standalone bundle
+ npm run build
- # 3) Copy static assets into standalone
- cp -r .next/static .next/standalone/.next/static
- [ -d public ] && cp -r public .next/standalone/public || true
+ # 3) Copy static assets into standalone
+ cp -r .next/static .next/standalone/.next/static
+ [ -d public ] && cp -r public .next/standalone/public || true
- # 4) Compile better-sqlite3 native binding for the target architecture.
- # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
- # without npm altering them.
- local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
- (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
+ # 4) Compile better-sqlite3 native binding for the target architecture.
+ # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
+ # without npm altering them.
+ local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
+ (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
- # 5) Place the compiled binding into the standalone bundle
- local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
- mkdir -p "$_bs3_release"
- cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
+ # 5) Place the compiled binding into the standalone bundle
+ local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
+ mkdir -p "$_bs3_release"
+ cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
- # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
- # so sharp is not used at runtime; x64 .so files would break aarch64 strip
- rm -rf .next/standalone/node_modules/@img
-
- # 7) Copy pino runtime deps omitted by Next.js static analysis:
- # pino-abstract-transport – required by pino's worker thread
- # split2 – dep of pino-abstract-transport
- # process-warning – dep of pino itself
- for _mod in pino-abstract-transport split2 process-warning; do
- cp -r "node_modules/$_mod" .next/standalone/node_modules/
- done
+ # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
+ # so sharp is not used at runtime; x64 .so files would break aarch64 strip
+ rm -rf .next/standalone/node_modules/@img
+ # 7) Copy pino runtime deps omitted by Next.js static analysis:
+ # pino-abstract-transport – required by pino's worker thread
+ # split2 – dep of pino-abstract-transport
+ # process-warning – dep of pino itself
+ for _mod in pino-abstract-transport split2 process-warning; do
+ cp -r "node_modules/$_mod" .next/standalone/node_modules/
+ done
}
do_check() {
-npm run test:unit
+ npm run test:unit
}
do_install() {
-vmkdir usr/lib/omniroute/.next
+ vmkdir usr/lib/omniroute/.next
- vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
+ vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
- # Prevent removal of empty Next.js app router dirs by the post-install hook
- for _d in \
- .next/standalone/.next/server/app/dashboard \
- .next/standalone/.next/server/app/dashboard/settings \
- .next/standalone/.next/server/app/dashboard/providers; do
- touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
- done
-
- cat > "${WRKDIR}/omniroute" <<'EOF'
+ # Prevent removal of empty Next.js app router dirs by the post-install hook
+ for _d in \
+ .next/standalone/.next/server/app/dashboard \
+ .next/standalone/.next/server/app/dashboard/settings \
+ .next/standalone/.next/server/app/dashboard/providers; do
+ touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
+ done
+ cat > "${WRKDIR}/omniroute" <<'EOF'
#!/bin/sh
export PORT="${PORT:-20128}"
export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}"
@@ -871,10 +967,9 @@ EOF
}
post_install() {
-vlicense LICENSE
+ vlicense LICENSE
}
-
-````
+```
@@ -882,9 +977,11 @@ vlicense LICENSE
## 🐳 Docker
-OmniRoute متاح كصورة Docker عامة على [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
+OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
-**الجري السريع:**```bash
+**Quick run:**
+
+```bash
docker run -d \
--name omniroute \
--restart unless-stopped \
@@ -892,85 +989,96 @@ docker run -d \
-p 20128:20128 \
-v omniroute-data:/app/data \
diegosouzapw/omniroute:latest
-````
+```
-**مع ملف البيئة:**```bash
+**With environment file:**
+```bash
# Copy and edit .env first
-
cp .env.example .env
docker run -d \
- --name omniroute \
- --restart unless-stopped \
- --stop-timeout 40 \
- --env-file .env \
- -p 20128:20128 \
- -v omniroute-data:/app/data \
- diegosouzapw/omniroute:latest
+ --name omniroute \
+ --restart unless-stopped \
+ --stop-timeout 40 \
+ --env-file .env \
+ -p 20128:20128 \
+ -v omniroute-data:/app/data \
+ diegosouzapw/omniroute:latest
+```
-````
+**Using Docker Compose:**
-**استخدام Docker Compose:**```bash
+```bash
# Base profile (no CLI tools)
docker compose --profile base up -d
# CLI profile (Claude Code, Codex, OpenClaw built-in)
docker compose --profile cli up -d
-````
+```
-يتضمن دعم لوحة المعلومات لعمليات نشر Docker الآن نقرة واحدة**Cloudflare Quick Tunnel**على `Dashboard → Endpoints`. يقوم الأول بتمكين التنزيلات `cloudflared` فقط عند الحاجة، ويبدأ نفقًا مؤقتًا إلى نقطة النهاية `/v1` الحالية، ويعرض عنوان URL الذي تم إنشاؤه `https://*.trycloudflare.com/v1` مباشرةً أسفل عنوان URL العام العادي.
+Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL.
-ملاحظات:
+Notes:
-- عناوين URL للنفق السريع مؤقتة وتتغير بعد كل إعادة تشغيل.
-- لا تتم استعادة الأنفاق السريعة تلقائيًا بعد إعادة تشغيل OmniRoute أو الحاوية. أعد تمكينها من لوحة التحكم عند الحاجة.
-- التثبيت المُدار يدعم حاليًا Linux وmacOS وWindows على `x64` / `arm64`.
-- الأنفاق السريعة المُدارة هي النقل الافتراضي عبر HTTP/2 لتجنب تحذيرات المخزن المؤقت QUIC UDP المزعجة في بيئات الحاويات المقيدة. قم بتعيين `CLOUDFLARED_PROTOCOL=quic` أو `auto` إذا كنت تريد وسيلة نقل مختلفة.
-- تقوم صور Docker بتجميع جذور CA للنظام وتمريرها إلى `cloudflared` المُدارة، مما يتجنب فشل ثقة TLS عندما يبدأ النفق داخل الحاوية.
-- يعمل SQLite في وضع WAL. يجب السماح لـ "docker stop" بالانتهاء حتى يتمكن OmniRoute من التحقق من أحدث التغييرات مرة أخرى في "storage.sqlite".
-- قامت ملفات الإنشاء المجمعة بالفعل بتعيين فترة سماح للتوقف مدتها 40 ثانية. إذا قمت بتشغيل الصورة مباشرة، فاحتفظ بـ `--stop-timeout 40` (أو ما شابه) حتى لا تؤدي عمليات الإيقاف اليدوية إلى قطع عملية تنظيف إيقاف التشغيل.
-- قم بتعيين `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` إذا كنت تريد أن يستخدم OmniRoute ملفًا ثنائيًا موجودًا بدلاً من تنزيله.
+- Quick Tunnel URLs are temporary and change after every restart.
+- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed.
+- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`.
+- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport.
+- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container.
+- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`.
+- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup.
+- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one.
-**استخدام Docker Compose مع Caddy (HTTPS Auto-TLS):**
+**Using Docker Compose with Caddy (HTTPS Auto-TLS):**
-يمكن كشف OmniRoute بشكل آمن باستخدام توفير SSL التلقائي من Caddy. تأكد من أن سجل DNS A الخاص بنطاقك يشير إلى عنوان IP الخاص بخادمك.```yaml
+OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP.
+
+```yaml
services:
-omniroute:
-image: diegosouzapw/omniroute:latest
-container_name: omniroute
-restart: unless-stopped
-volumes: - omniroute-data:/app/data
-environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com
+ omniroute:
+ image: diegosouzapw/omniroute:latest
+ container_name: omniroute
+ restart: unless-stopped
+ volumes:
+ - omniroute-data:/app/data
+ environment:
+ - PORT=20128
+ - NEXT_PUBLIC_BASE_URL=https://your-domain.com
-caddy:
-image: caddy:latest
-container_name: caddy
-restart: unless-stopped
-ports: - "80:80" - "443:443"
-command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
+ caddy:
+ image: caddy:latest
+ container_name: caddy
+ restart: unless-stopped
+ ports:
+ - "80:80"
+ - "443:443"
+ command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
volumes:
-omniroute-data:
+ omniroute-data:
+```
-````
-
-| صورة | العلامة | الحجم | الوصف |
+| Image | Tag | Size | Description |
| ------------------------ | -------- | ------ | --------------------- |
-| `diegosouzapw/omniroute` | `الأحدث` | ~250 ميجابايت | أحدث إصدار مستقر |
-| `diegosouzapw/omniroute` | `1.0.3` | ~250 ميجابايت | النسخة الحالية |---
+| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release |
+| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version |
+
+---
## 🖥️ Desktop App — Offline & Always-On
-> 🆕**جديد!**OmniRoute متوفر الآن كتطبيق سطح مكتب أصلي**لأنظمة التشغيل Windows وmacOS وLinux.
+> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux.
-قم بتشغيل OmniRoute كتطبيق مستقل لسطح المكتب - لا توجد محطة طرفية أو متصفح أو إنترنت مطلوب للطرز المحلية. يتضمن التطبيق المعتمد على Electron ما يلي:
+Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes:
-- 🖥️**النافذة الأصلية**— نافذة تطبيق مخصصة مع تكامل علبة النظام
-- 🔄**البدء التلقائي**— قم بتشغيل OmniRoute عند تسجيل الدخول إلى النظام
-- 🔔**الإشعارات الأصلية**— احصل على تنبيهات بشأن استنفاد الحصص أو مشكلات المزود
-- ⚡**التثبيت بنقرة واحدة**— NSIS (Windows)، DMG (macOS)، AppImage (Linux)
-- 🌐**وضع عدم الاتصال بالإنترنت**— يعمل بشكل كامل دون اتصال بالإنترنت مع الخادم المُجمَّع### بداية سريعة
+- 🖥️ **Native Window** — Dedicated app window with system tray integration
+- 🔄 **Auto-Start** — Launch OmniRoute on system login
+- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues
+- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux)
+- 🌐 **Offline Mode** — Works fully offline with bundled server
+
+### بداية سريعة
```bash
# Development mode
@@ -981,308 +1089,370 @@ npm run electron:build # Current platform
npm run electron:build:win # Windows (.exe)
npm run electron:build:mac # macOS (.dmg) — x64 & arm64
npm run electron:build:linux # Linux (.AppImage)
-````
+```
### System Tray
-عند تصغيره، يظل OmniRoute موجودًا في علبة النظام لديك من خلال الإجراءات السريعة:
+When minimized, OmniRoute lives in your system tray with quick actions:
-- فتح لوحة القيادة
-- تغيير منفذ الخادم
-- قم بإنهاء التطبيق
+- Open dashboard
+- Change server port
+- Quit application
-📖 التوثيق الكامل: [`electron/README.md`](electron/README.md)---
+📖 Full documentation: [`electron/README.md`](electron/README.md)
+
+---
## 💰 Pricing at a Glance
-| الطبقة | مقدم | التكلفة | إعادة ضبط الحصص | الأفضل لـ |
-| ---------------------------------- | ---------------------------- | ------------------------------------- | ------------------------------------------- | ---------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| **💳الإشتراك** | كلود كود (برو) | 20 دولارًا شهريًا | 5 ساعات + أسبوعي | اشتركت بالفعل |
-| | الدستور الغذائي (زائد / برو) | 20-200 دولار شهريًا | 5 ساعات + أسبوعي | مستخدمي OpenAI |
-| | الجوزاء CLI | **مجاني** | 180 ألف/شهر + 1 ألف/يوم | الجميع! |
-| | جيثب مساعد الطيار | 10-19 دولارًا شهريًا | شهري | مستخدمي جيثب |
-| **🔑 مفتاح واجهة برمجة التطبيقات** | نفيديا نيم | **مجانًا**(مطور للأبد) | ~40 دورة في الدقيقة | 70+ نماذج مفتوحة |
-| | المخيخ | **مجانًا**(1 مليون توك/يوم) | 60 ألف دورة في الدقيقة / 30 دورة في الدقيقة | الأسرع في العالم |
-| | جروك | **مجانًا**(30 دورة في الدقيقة) | 14.4K دورة في الدقيقة | لاما/جيما فائقة السرعة |
-| | ديب سيك V3.2 | 0.27 دولار/1.10 دولار لكل مليون | لا شيء | أفضل منطق السعر/الجودة |
-| | xAI Grok-4 سريع | **0.20 دولار/0.50 دولار لكل مليون**🆕 | لا شيء | أسرع + أداة استدعاء، منخفضة للغاية |
-| | xAI Grok-4 (قياسي) | 0.20 دولار/1.50 دولار لكل مليون 🆕 | لا شيء | المنطق الرائد من xAI |
-| | ميسترال | تجربة مجانية + مدفوعة | معدل محدود | الذكاء الاصطناعي الأوروبي |
-| | اوبن راوتر | الدفع لكل استخدام | لا شيء | 100+ نماذج مجمعة. |
-| **💰 رخيص** | GLM-5 (عبر Z.AI) 🆕 | 0.5 دولار/1 مليون | يوميا 10 صباحا | إخراج 128 كيلو، أحدث الرائد |
-| | جي إل إم-4.7 | 0.6 دولار/1 مليون | يوميا 10 صباحا | نسخة احتياطية للميزانية |
-| | ميني ماكس M2.5 🆕 | إدخال 0.3 دولار/1 مليون | المتداول لمدة 5 ساعات | الاستدلال + المهام الوكيلة |
-| | ميني ماكس M2.1 | 0.2 دولار/1 مليون | المتداول لمدة 5 ساعات | الخيار الأرخص |
-| | كيمي K2.5 (Moonshot API) 🆕 | الدفع لكل استخدام | لا شيء | الوصول المباشر إلى Moonshot API |
-| | كيمي ك2 | 9 دولارات شهريًا مسطحة | 10 مليون رمز/شهر | التكلفة المتوقعة |
-| **🆓مجانًا** | قدير | **$0** | غير محدود | 5 نماذج غير محدودة |
-| | كوين | **$0** | غير محدود | 4 نماذج غير محدودة |
-| | كيرو | **$0** | غير محدود | كلود سونيت/هايكو (AWS Builder) |
-| | LongCat Flash-Lite 🆕 | **$0**(50 مليون توك/يوم 🔥) | 1 دورة في الثانية | أكبر حصة مجانية على وجه الأرض |
-| | التلقيحات AI 🆕 | **$0**(لا حاجة لمفتاح) | 1 متطلب/15 ثانية | جي بي تي-5، كلود، ديب سيك، لاما 4 |
-| | Cloudflare Workers AI 🆕 | **$0**(10 آلاف خلية عصبية/اليوم) | ~150 راحة/يوم | أكثر من 50 نموذجًا، حافة عالمية |
-| | سكيليواي AI 🆕 | **$0**(إجمالي 1 مليون رمز) | معدل محدود | الاتحاد الأوروبي/اللائحة العامة لحماية البيانات، Qwen3 235B، Llama 70B | > 🆕**تمت إضافة نماذج جديدة (مارس 2026):**عائلة Grok-4 Fast بسعر 0.20 دولار أمريكي/0.50 دولار أمريكي/م (تم قياسها عند 1143 مللي ثانية - أسرع بنسبة 30% من Gemini 2.5 Flash)، GLM-5 عبر Z.AI بإخراج 128 ألف، واستدلال MiniMax M2.5، وتسعير DeepSeek V3.2 المحدث، وKimi K2.5 عبر Moonshot direct API. |
+| Tier | Provider | Cost | Quota Reset | Best For |
+| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- |
+| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed |
+| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users |
+| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! |
+| | GitHub Copilot | $10-19/mo | Monthly | GitHub users |
+| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models |
+| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest |
+| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma |
+| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning |
+| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow |
+| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI |
+| | Mistral | Free trial + paid | Rate limited | European AI |
+| | OpenRouter | Pay-per-use | None | 100+ models aggr. |
+| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship |
+| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup |
+| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks |
+| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option |
+| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access |
+| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost |
+| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited |
+| | Qwen | **$0** | Unlimited | 4 models unlimited |
+| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) |
+| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth |
+| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 |
+| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge |
+| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B |
-**💡 $0 Combo Stack — الإعداد المجاني الكامل:**```
+> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API.
+**💡 $0 Combo Stack — The Complete Free Setup:**
+
+```
# 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever
+Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
+Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
+Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
+Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
+NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+```
-Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
-Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
-LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
-Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
-Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
-Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
-Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
-Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
-Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
-NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
-Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever.
-````
-
-**تكلفة صفر. لا تتوقف أبدًا عن البرمجة.**قم بتكوين هذا كمجموعة واحدة من OmniRoute وستحدث جميع الإجراءات الاحتياطية تلقائيًا - لا يوجد تبديل يدوي على الإطلاق.---
+---
---
## 🆓 Free Models — What You Actually Get
-> جميع الموديلات أدناه**مجانية بنسبة 100% ولا تتطلب أي بطاقة ائتمان**. يقوم OmniRoute بالمسارات التلقائية بينهما عند نفاد حصة واحدة - اجمعها جميعًا للحصول على مجموعة غير قابلة للكسر بقيمة 0 دولار.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
+> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo.
-| نموذج | البادئة | الحد | حد السعر |
+### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------- | ------ | ------------- | --------------------- |
-| `كلود-السوناتة-4.5` | `كر/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى اليومي |
-| `كلود-هايكو-4.5` | `كر/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى اليومي |
-| `كلود-أوبوس-4.6` | `كر/` |**غير محدود**| أحدث أعمال أوبوس عبر كيرو |### 🟢 QODER MODELS (Free PAT via qodercli)
+| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro |
-| نموذج | البادئة | الحد | حد السعر |
+### 🟢 QODER MODELS (Free PAT via qodercli)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------ | ------ | ------------- | --------------- |
-| `تفكير كيمي-ك2` | `إذا/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى |
-| `qwen3-coder-plus` | `إذا/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى |
-| `ديبسيك-R1` | `إذا/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى |
-| `مينيماكس-m2.1` | `إذا/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى |
-| `كيمي-k2` | `إذا/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى |
+| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap |
+| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap |
+| `deepseek-r1` | `if/` | **Unlimited** | No reported cap |
+| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap |
+| `kimi-k2` | `if/` | **Unlimited** | No reported cap |
-> طريقة الاتصال الموصى بها:**رمز الوصول الشخصي + `qodercli`**. متصفح OAuth هو
-> تجريبي ومعطل افتراضيًا ما لم يتم تكوين متغيرات البيئة `QODER_OAUTH_*`.### 🟡 QWEN MODELS (Device Code Auth)
+> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is
+> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured.
-| نموذج | البادئة | الحد | حد السعر |
+### 🟡 QWEN MODELS (Device Code Auth)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------- | ------ | ------------- | ------------------- |
-| `qwen3-coder-plus` | `س/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى |
-| `qwen3-coder-flash` | `س/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى |
-| `qwen3-coder-next` | `س/` |**غير محدود**| لم يتم الإبلاغ عن الحد الأقصى |
-| `نموذج الرؤية` | `س/` |**غير محدود**| الوسائط المتعددة (صور) |### 🟣 GEMINI CLI (Google OAuth)
+| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap |
+| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) |
-| نموذج | البادئة | الحد | حد السعر |
+### 🟣 GEMINI CLI (Google OAuth)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------------ | ------ | --------------------------- | ------------- |
-| `الجوزاء-3-معاينة فلاش` | `جي سي/` |**180 ألف توك/شهر**+ 1 ألف/يوم | إعادة الضبط الشهرية |
-| `الجوزاء-2.5-برو` | `جي سي/` | 180 ألف/شهر (مسبح مشترك) | جودة عالية |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
+| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset |
+| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality |
-| الطبقة | الحد اليومي | حد السعر | ملاحظات |
+### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
+
+| Tier | Daily Limit | Rate Limit | Notes |
| ---------- | ------------ | ----------- | ------------------------------------------------------ |
-| مجاني (ديف) | لا يوجد غطاء رمزي |**~40 دورة في الدقيقة**| أكثر من 70 نموذجًا؛ الانتقال إلى حدود المعدل النقي منتصف عام 2025 |
+| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 |
-النماذج المجانية المشهورة: `moonshotai/kimi-k2.5` (Kimi K2.5)، `z-ai/glm4.7` (GLM 4.7)، `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2)، `nvidia/llama-3.3-70b-instruct`، `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
+Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`
-| الطبقة | الحد اليومي | حد السعر | ملاحظات |
+### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
+
+| Tier | Daily Limit | Rate Limit | Notes |
| ---- | ----------------- | ---------------- | ------------------------------------------- |
-| مجاني |**1 مليون قطعة/يوم**| 60 ألف دورة في الدقيقة / 30 دورة في الدقيقة | أسرع استنتاج LLM في العالم؛ يعيد يوميا |
+| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily |
-متاح مجانًا: `llama-3.3-70b`، `llama-3.1-8b`، `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com)
+Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`
-| الطبقة | الحد اليومي | حد السعر | ملاحظات |
+### 🔴 GROQ (Free API Key — console.groq.com)
+
+| Tier | Daily Limit | Rate Limit | Notes |
| ---- | ------------- | ---------------- | ----------------------------------------- |
-| مجاني |**14.4 كيلو دورة في الدقيقة**| 30 دورة في الدقيقة لكل موديل | لا توجد بطاقة ائتمان؛ 429 على الحد، غير مشحونة |
+| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged |
-متاحة مجانًا: `llama-3.3-70b-versatile`، `gemma2-9b-it`، `mixtral-8x7b`، `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`
-| نموذج | البادئة | الحصة اليومية المجانية | ملاحظات |
+### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+
+| Model | Prefix | Daily Free Quota | Notes |
| ----------------------------- | ------ | ----------------- | ----------------------- |
-| `لونجكات-فلاش-لايت` | `لك/` |**50 مليون رمز**💥 | أكبر حصة مجانية على الإطلاق |
-| `LongCat-Flash-Chat` | `لك/` | 500 ألف رمز | دردشة متعددة المنعطفات |
-| ``التفكير الخاطف الطويل`` | `لك/` | 500 ألف رمز | الاستدلال / CoT |
-| `لونجكات-فلاش-التفكير-2601` | `لك/` | 500 ألف رمز | نسخة يناير 2026 |
-| `لونج كات-فلاش-أومني-2603` | `لك/` | 500 ألف رمز | الوسائط المتعددة |
+| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever |
+| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat |
+| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT |
+| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version |
+| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal |
-> مجاني 100% أثناء وجودك في النسخة التجريبية العامة. قم بالتسجيل في [longcat.chat](https://longcat.chat) باستخدام البريد الإلكتروني أو الهاتف. تتم إعادة الضبط يوميًا في تمام الساعة 00:00 بالتوقيت العالمي المنسق.### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC.
-| نموذج | البادئة | حد السعر | مقدم خلف |
+### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+
+| Model | Prefix | Rate Limit | Provider Behind |
| ---------- | ------ | ---------- | ------------------ |
-| `أوبيني` | `بول/` | 1 متطلب/15 ثانية | جي بي تي-5 |
-| "كلود" | `بول/` | 1 متطلب/15 ثانية | أنثروبي كلود |
-| `الجوزاء` | `بول/` | 1 متطلب/15 ثانية | جوجل الجوزاء |
-| `البحث العميق` | `بول/` | 1 متطلب/15 ثانية | ديب سيك V3 |
-| اللاما | `بول/` | 1 متطلب/15 ثانية | ميتا لاما 4 كشاف |
-| `ميسترال` | `بول/` | 1 متطلب/15 ثانية | ميسترال لمنظمة العفو الدولية |
+| `openai` | `pol/` | 1 req/15s | GPT-5 |
+| `claude` | `pol/` | 1 req/15s | Anthropic Claude |
+| `gemini` | `pol/` | 1 req/15s | Google Gemini |
+| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 |
+| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout |
+| `mistral` | `pol/` | 1 req/15s | Mistral AI |
-> ✨**بدون احتكاك:**لا يوجد اشتراك، ولا يوجد مفتاح API. أضف موفر التلقيح بحقل مفتاح فارغ وسيعمل على الفور.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately.
-| الطبقة | الخلايا العصبية اليومية | الاستخدام المعادل | ملاحظات |
+### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+
+| Tier | Daily Neurons | Equivalent Usage | Notes |
| ---- | ------------- | --------------------------------------- | ----------------------- |
-| مجاني |**10,000**| ~150 LLM resp / 500 ثانية صوت / 15 ألف تضمين | الحافة العالمية، أكثر من 50 نموذجًا |
+| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models |
-النماذج المجانية الشهيرة: `@cf/meta/llama-3.3-70b-instruct`، `@cf/google/gemma-3-12b-it`، `@cf/openai/whisper-large-v3-turbo` (صوت مجاني!)، `@cf/qwen/qwen2.5-coder-15b-instruct`
+Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct`
-> يتطلب رمز API المميز + معرف الحساب من [dash.cloudflare.com](https://dash.cloudflare.com). قم بتخزين معرف الحساب في إعدادات الموفر.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings.
-| الطبقة | حصة مجانية | الموقع | ملاحظات |
+### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+
+| Tier | Free Quota | Location | Notes |
| ---- | ------------- | ------------ | ----------------------------------- |
-| مجاني |**مليون قطعة**| 🇫🇷 باريس، الاتحاد الأوروبي | لا حاجة لبطاقة الائتمان ضمن الحدود |
+| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits |
-متاح مجانًا: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!)
+Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
-> متوافقة مع الاتحاد الأوروبي/اللائحة العامة لحماية البيانات. احصل على مفتاح واجهة برمجة التطبيقات على [console.scaleway.com](https://console.scaleway.com).
+> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com).
->**💡 المجموعة المجانية المطلقة (11 مقدمًا، 0 دولار للأبد):**
+> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):**
>
> ```
-> كيرو (kr/) → كلود سونيت/هايكو غير محدود
-> Qoder (if/) → kimi-k2-thinking، qwen3-coder-plus، Deepseek-r1 غير محدود
-> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 مليون رمز/يوم 🔥
-> التلقيح (pol/) → GPT-5، Claude، DeepSeek، Llama 4 - لا حاجة إلى مفتاح
-> Qwen (qw/) → نماذج qwen3-coder غير محدودة
-> Gemini (gemini/) → Gemini 2.5 Flash — 1500 طلب/يوم مجانًا
-> Cloudflare AI (cf/) → أكثر من 50 نموذجًا - 10 آلاف خلية عصبية/اليوم
-> Scaleway (scw/) → Qwen3 235B، Llama 70B — مليون رمز مجاني (الاتحاد الأوروبي)
-> Groq (groq/) → Llama/Gemma — 14.4 ألف طلب/يوم بسرعة فائقة
-> NVIDIA NIM (nvidia/) → أكثر من 70 طرازًا مفتوحًا - 40 دورة في الدقيقة إلى الأبد
-> المخيخ (cerebras/) → اللاما/كوين الأسرع في العالم — مليون توك/اليوم
-> ```## 🎙️ Free Transcription Combo
+> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+> Qwen (qw/) → qwen3-coder models UNLIMITED
+> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free
+> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day
+> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast
+> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+> ```
-> قم بنسخ أي صوت/فيديو مقابل**$0**— تقدم Deepgram مبلغًا مجانيًا بقيمة 200 دولار أمريكي، ونسخة احتياطية من AssemblyAI بقيمة 50 دولارًا أمريكيًا، وGroq Whisper كنسخة احتياطية غير محدودة للطوارئ.
+## 🎙️ Free Transcription Combo
-| مقدم | اعتمادات مجانية | أفضل موديل | حد السعر |
+> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup.
+
+| Provider | Free Credits | Best Model | Rate Limit |
| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- |
-| 🟢**ديبجرام**|**200 دولار مجانًا**(اشتراك) | `nova-3` — أفضل دقة، أكثر من 30 لغة | لا يوجد حد لعدد RPM على الاعتمادات المجانية |
-| 🔵**AssemblyAI**|**50 دولارًا مجانًا**(اشتراك) | `universal-3-pro` - الفصول، المشاعر، معلومات تحديد الهوية الشخصية | لا يوجد حد لعدد RPM على الاعتمادات المجانية |
-| 🔴**جروق**|**مجاني للأبد**| `whisper-large-v3` — OpenAI Whisper | 30 دورة في الدقيقة (معدل محدود) |
+| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits |
+| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits |
+| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) |
-**التحرير والسرد المقترح في `/dashboard/combos`:**```
+**Suggested combo in `/dashboard/combos`:**
+
+```
Name: free-transcription
Strategy: Priority
Nodes:
[1] deepgram/nova-3 → uses $200 free first
[2] assemblyai/universal-3-pro → fallback when Deepgram credits run out
[3] groq/whisper-large-v3 → free forever, emergency fallback
-````
+```
-ثم في `/dashboard/media` → علامة التبويب**Transcription**: قم بتحميل أي ملف صوت أو فيديو ← حدد نقطة نهاية التحرير والسرد الخاصة بك ← احصل على النسخ بتنسيقات مدعومة.## 💡 Key Features
+Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats.
-تم تصميم OmniRoute v2.0 كمنصة تشغيلية، وليس مجرد وكيل ترحيل.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
+## 💡 Key Features
-| ميزة | ماذا يفعل |
-| ------------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ |
-| ⚡**Grok-4 Fast Family** | طرازات xAI بسعر 0.20 دولارًا أمريكيًا/0.50 دولارًا أمريكيًا للمتر المربع - تم قياسها بـ 1143 مللي ثانية (أسرع بنسبة 30% من Gemini 2.5 Flash) |
-| 🧠**GLM-5 عبر Z.AI** | سياق إخراج 128 ألفًا، 0.5 دولار أمريكي/1 مليون — أحدث منتج رئيسي من عائلة GLM |
-| 🔮**ميني ماكس M2.5** | الاستدلال + المهام الوكيلة بسعر 0.30 دولارًا أمريكيًا/مليون واحد — ترقية كبيرة من M2.1 |
-| 🎯**أداة استدعاء العلم لكل نموذج** | لكل نموذج `toolCalling: true/false` في التسجيل - يتخطى AutoCombo النماذج التي لا تحتوي على أدوات |
-| 🌍**كشف النوايا المتعددة اللغات** | الكلمات الأساسية PT/ZH/ES/AR في تسجيل AutoCombo — اختيار نموذج أفضل للمحتوى غير الإنجليزي |
-| 📊**الإجراءات الاحتياطية المستندة إلى المعايير** | زمن استجابة حقيقي p95 من الطلبات المباشرة يغذي تسجيل التحرير والسرد - يتعلم AutoCombo من البيانات الفعلية |
-| 🔁**طلب إلغاء البيانات المكررة** | نافذة إلغاء البيانات المستندة إلى تجزئة المحتوى — آمنة متعددة الوكلاء، وتمنع الرسوم المكررة |
-| 🔌**استراتيجية جهاز التوجيه القابل للتوصيل** | واجهة "RouterStrategy" القابلة للتوسيع - أضف منطق توجيه مخصص كمكونات إضافية | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP |
+OmniRoute v3.5 is built as an operational platform, not just a relay proxy.
-| ميزة | ماذا يفعل |
-| -------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |
-| 🎮**ساحة اللعب النموذجية** | صفحة لوحة التحكم لاختبار أي نموذج مباشرة - محددات الموفر/النموذج/نقطة النهاية، محرر موناكو، البث، الإجهاض، التوقيت |
-| 🔏**مطابقة بصمة CLI** | ترتيب الرأس/النص لكل موفر لمطابقة توقيعات CLI الأصلية - قم بالتبديل لكل موفر في الإعدادات > الأمان.**يتم الاحتفاظ بـ IP الوكيل الخاص بك** |
-| 🤝**دعم ACP (بروتوكول العميل الوكيل)** | اكتشاف وكيل CLI (Codex، Claude، Goose، Gemini CLI، OpenClaw + 9 آخرين)، مولد العمليات، `/api/acp/agents` نقطة النهاية |
-| 🤖**لوحة تحكم وكلاء ACP** | التصحيح › صفحة الوكلاء - شبكة مكونة من 14 وكيلًا مع حالة التثبيت والإصدار ونموذج الوكيل المخصص لأي أداة CLI. يحصل مستخدمو**OpenCode**على زر "تنزيل opencode.json" الذي يقوم تلقائيًا بإنشاء تكوين جاهز للاستخدام مع جميع الطرز المتاحة. |
-| 🔧**توجيه نموذج مخصص `apiFormat`** | النماذج المخصصة ذات `apiFormat: "responses"` توجه الآن بشكل صحيح إلى مترجم Responses API |
-| 🏢**عزل مساحة عمل الدستور الغذائي** | مساحات عمل Codex متعددة لكل بريد إلكتروني - يفصل OAuth الاتصالات بشكل صحيح عن طريق معرف مساحة العمل |
-| 🔄**التحديث التلقائي الإلكتروني** | يتحقق تطبيق سطح المكتب من التحديثات + التثبيت التلقائي عند إعادة التشغيل | ### 🤖 Agent & Protocol Operations (v2.0) |
+### 🆕 New — v3.5.5 Highlights (Apr 2026)
-| ميزة | ماذا يفعل |
-| -------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------- |
-| 🔧**خادم MCP (25 أداة)** | أدوات IDE/agent عبر 3 وسائل نقل: stdio، وSSE (`/api/mcp/sse`)، وHTTP القابل للتدفق (`/api/mcp/stream`). 18 نواة + 3 ذاكرة + 4 أدوات مهارات |
-| 🤝**خادم A2A (JSON-RPC + SSE)** | تنفيذ المهام من وكيل إلى وكيل مع تدفقات المزامنة والتدفق |
-| 🧭**صفحة نقاط النهاية الموحدة** | صفحة إدارة مبوبة مع علامات تبويب Endpoint Proxy وMCP وA2A وAPI Endpoints |
-| 🎚️**تبديل تمكين / تعطيل الخدمة** | مفاتيح التشغيل/الإيقاف لـ MCP وA2A مع ثبات الإعدادات (الافتراضي: OFF) |
-| 🛰️**نبضات وقت تشغيل MCP** | حالة العملية الحقيقية (معرف المنتج، وقت التشغيل، عمر نبضات القلب، النقل، وضع النطاق) |
-| 📋**مسار تدقيق MCP** | سجلات التدقيق القابلة للتصفية مع النجاح/الفشل والإسناد الرئيسي |
-| 🔐**تنفيذ نطاق MCP** | 10 أذونات نطاق تفصيلية للوصول إلى الأدوات الخاضعة للرقابة |
-| 📡**إدارة دورة حياة المهام A2A** | قائمة/تصفية المهام، فحص الأحداث/التحف، إلغاء المهام قيد التشغيل |
-| 📋**اكتشاف بطاقة الوكيل** | `/.well-known/agent.json` للاكتشاف التلقائي للعميل |
-| 🧪**أداة اختبار البروتوكول E2E** | يتدفق عميل MCP SDK + A2A الحقيقي في "اختبار: البروتوكولات: e2e" |
-| ⚙️**ضوابط التشغيل** | مجموعة التبديل، وتطبيق ملفات تعريف المرونة، وإعادة ضبط القواطع من سطح تحكم واحد | ### 🧠 Routing & Intelligence |
+| Feature | What It Does |
+| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation |
+| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs |
+| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions |
+| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection |
+| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts |
-| ميزة | ماذا يفعل |
-| ---------------------------------------- | -------------------------------------------------------------------------- | ----------------------- |
-| 🎯**احتياطي ذكي من 4 طبقات** | المسار التلقائي: الاشتراك → مفتاح API → رخيص → مجاني |
-| 📊**تتبع الحصص في الوقت الفعلي** | عدد الرموز الحية + إعادة تعيين العد التنازلي لكل مزود |
-| 🔄**تنسيق الترجمة** | OpenAI ↔ Claude ↔ Gemini ↔ الردود مع التحويلات الآمنة للمخطط |
-| 👥**دعم الحسابات المتعددة** | حسابات متعددة لكل مزود مع اختيار ذكي |
-| 🔄**تحديث تلقائي للرمز** | يتم تحديث رموز OAuth المميزة تلقائيًا من خلال إعادة المحاولة |
-| 🎨**مجموعات مخصصة** | 9 استراتيجيات موازنة + التحكم في السلسلة الاحتياطية |
-| 🌐**جهاز توجيه Wildcard** | `المزود/*` التوجيه الديناميكي |
-| 🧠**التفكير في ضوابط الميزانية** | حدود التفكير المنطقي والتلقائي والمخصص والتكيفي |
-| 🔀**الأسماء المستعارة للنماذج** | مدمج + اسم مستعار للنموذج المخصص وأمان الترحيل |
-| ⚡**تدهور الخلفية** | قم بتوجيه مهام الخلفية ذات الأولوية المنخفضة إلى نماذج أرخص |
-| 🧪**التوجيه الذكي المدرك للمهام** | تحديد النموذج تلقائيًا حسب نوع المحتوى (الترميز/الرؤية/التحليل/التلخيص) |
-| 🔄**سير عمل وكيل A2A** | منسق ولايات ميكرونيزيا الموحدة الحتمية لعمليات إعدام الوكيل متعددة الخطوات |
-| 🔀**التوجيه التكيفي** | تجاوز الإستراتيجية الديناميكية بناءً على حجم الرمز المميز والتعقيد الفوري |
-| 🎲**تنوع مقدمي الخدمة** | شانون الإنتروبيا التهديف موازنة توزيع حركة المرور والسرد التلقائي |
-| 💬**الحقن الفوري للنظام** | يتم تطبيق ضوابط السلوك العالمية بشكل متسق |
-| 📄**توافق واجهة برمجة التطبيقات للردود** | الدعم الكامل `/v1/responses` لـ Codex وسير العمل الوكيل المتقدم | ### 🎵 Multi-Modal APIs |
+### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
-| ميزة | ماذا يفعل |
-| ------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------- |
-| 🖼️**إنشاء الصور** | `/v1/images/ Generations` مع الواجهات الخلفية السحابية والمحلية |
-| 📐**المضامين** | `/v1/embeddings` لخطوط أنابيب البحث وRAG |
-| 🎤**نسخ صوتي** | `/v1/audio/transcriptions` - 7 مقدمي خدمات (Deepgram Nova 3، AssemblyAI، Groq Whisper، HuggingFace، ElevenLabs، OpenAI، Azure)، الكشف التلقائي عن اللغة، دعم MP4/MP3/WAV |
-| 🔊**تحويل النص إلى كلام** | `/v1/audio/speech` - 10 مقدمي خدمات (ElevenLabs، OpenAI، Deepgram، Cartesia، PlayHT، HuggingFace، Nvidia NIM، Inworld، Coqui، Tortoise) مع رسائل الخطأ الصحيحة |
-| 🎬**توليد الفيديو** | `/v1/videos/أجيال` (سير عمل ComfyUI + SD WebUI) |
-| 🎵**جيل الموسيقى** | `/v1/music/generations` (سير عمل ComfyUI) |
-| 🛡️**اعتدالات** | `/v1/moderations` فحوصات السلامة |
-| 🔀**إعادة الترتيب** | `/v1/rerank` لدرجات الملاءمة |
-| 🔍**بحث الويب**🆕 | `/v1/search` - 5 مقدمي خدمات (Serper، Brave، Perplexity، Exa، Tavily)، أكثر من 6500 خدمة مجانية شهريًا، تجاوز الفشل التلقائي، ذاكرة التخزين المؤقت | ### 🛡️ Resilience, Security & Governance |
+| Feature | What It Does |
+| ------------------------------------ | ------------------------------------------------------------------------------------------- |
+| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) |
+| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family |
+| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 |
+| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models |
+| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content |
+| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data |
+| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges |
+| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins |
-| ميزة | ماذا يفعل |
-| --------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------- | -------------------------------- |
-| 🔌**قواطع الدائرة** | رحلة/استرداد لكل نموذج مع عناصر التحكم في العتبة |
-| 🎯**نماذج تدرك نقطة النهاية** | تعلن النماذج المخصصة عن نقاط النهاية المدعومة + تنسيق API |
-| 🛡️**القطيع المضاد للرعد** | حماية Mutex + الإشارة في أحداث إعادة المحاولة/التقييم |
-| 🧠**ذاكرة التخزين المؤقت الدلالية + التوقيع** | تقليل التكلفة/زمن الوصول باستخدام طبقتين من ذاكرة التخزين المؤقت |
-| ⚡**طلب العجز** | نافذة الحماية المكررة |
-| 🔒**انتحال بصمة الإصبع TLS** | بصمة TLS الشبيهة بالمتصفح -**تقلل من اكتشاف الروبوتات ووضع علامة على الحساب** |
-| 🔏**مطابقة بصمة CLI** | يطابق توقيعات طلب واجهة سطر الأوامر (CLI) الأصلية -**يقلل من مخاطر الحظر مع الحفاظ على عنوان IP الخاص بالوكيل** |
-| 🌐**تصفية IP** | التحكم في القائمة المسموح بها/القائمة المحظورة لعمليات النشر المكشوفة |
-| 📊**حدود المعدل القابلة للتحرير** | حدود عالمية/مستوى مزود قابلة للتكوين مع الثبات |
-| 📉**التحلل الرشيق** | قدرات احتياطية متعددة الطبقات تحمي عمليات البوابة الأساسية |
-| 📜**مسار تدقيق التكوين** | تتبع التغيير القائم على الاختلاف يمنع الانحراف التشغيلي من خلال عمليات التراجع البسيطة |
-| ⏳**مزامنة صحة الموفر** | مراقبة استباقية لانتهاء صلاحية الرمز المميز، مما يؤدي إلى تنبيهات قبل فشل التفويض |
-| 🚪**تعطيل الحسابات المحظورة تلقائيًا** | يقوم قاطع الدائرة التشغيلية بإغلاق حسابات الرموز المميزة المحظورة بشكل دائم تلقائيًا |
-| 🔑**إدارة مفاتيح واجهة برمجة التطبيقات + تحديد النطاق** | تأمين إصدار/تدوير المفتاح وضوابط النموذج/المزود |
-| 👁️**الكشف عن مفتاح واجهة برمجة التطبيقات (Scoped API)**🆕 | الاشتراك في استرداد مفاتيح واجهة برمجة التطبيقات عبر `ALLOW_API_KEY_REVEAL` |
-| 🛡️**محميه `/موديلات`** | بوابة مصادقة اختيارية وإخفاء الموفر لكتالوج النماذج | ### 📊 Observability & Analytics |
+### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP
-| ميزة | ماذا يفعل |
-| ---------------------------------- | ------------------------------------------------------------------------- | ---------------------------- |
-| 📝**الطلب + تسجيل الوكيل** | الطلب/الاستجابة الكاملة وتسجيل الوكيل |
-| 📉**السجلات التفصيلية المتدفقة**🆕 | يعيد بناء تدفقات حمولة SSE بشكل واضح في واجهة المستخدم |
-| 📋**لوحة تحكم السجلات الموحدة** | طلب العروض والوكيل والتدقيق ووحدة التحكم في صفحة واحدة |
-| 🔍**طلب القياس عن بعد** | زمن الاستجابة p50/p95/p99 وطلب التتبع |
-| 🏥**لوحة المعلومات الصحية** | وقت التشغيل، حالات الكسارة، عمليات الإغلاق، إحصائيات ذاكرة التخزين المؤقت |
-| 💰**تتبع التكلفة** | ضوابط الميزانية ورؤية التسعير لكل نموذج |
-| 📈**تصورات التحليلات** | رؤى استخدام النموذج/الموفر وطرق عرض الاتجاه |
-| 🧪**إطار التقييم** | اختبار المجموعة الذهبية مع استراتيجيات المطابقة القابلة للتكوين |
-| 📡**تشخيص مباشر**🆕 | تجاوز ذاكرة التخزين المؤقت الدلالية لإجراء اختبار مباشر دقيق للسرد | ### ☁️ Deployment & Platform |
+| Feature | What It Does |
+| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing |
+| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** |
+| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint |
+| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. |
+| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator |
+| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID |
+| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart |
-| ميزة | ماذا يفعل |
-| --------------------------------------------- | -------------------------------------------------------------------- | --------------------- |
-| 🌐**النشر في أي مكان** | المضيف المحلي، VPS، Docker، البيئات السحابية |
-| 🚇**نفق كلاود فلير**🆕 | تكامل النفق السريع بنقرة واحدة من لوحة المعلومات |
-| 🔑**تصفية نموذج مفتاح واجهة برمجة التطبيقات** | تمت تصفية الاستجابة الأصلية /v1/models عبر أدوار سياق الحامل المعينة |
-| ⚡**تجاوز ذاكرة التخزين المؤقت الذكية** | استدلالات TTL قابلة للتكوين وضوابط إعادة الجلب القسري |
-| 🔄**النسخ الاحتياطي/الاستعادة** | تدفقات التصدير/الاستيراد والتعافي من الكوارث |
-| 🧙**معالج الإعداد** | الإعداد الموجه لأول مرة |
-| 🔧**لوحة تحكم أدوات CLI** | إعداد بنقرة واحدة لأدوات الترميز الشائعة |
-| 🎮**ساحة اللعب النموذجية** | اختبر أي موفر/نموذج/نقطة نهاية من لوحة المعلومات |
-| 🔏**تبديل بصمة الإصبع CLI** | مطابقة بصمات الأصابع لكل موفر في الإعدادات > الأمان |
-| 🌐**i18n (30 لغة)** | لوحة تحكم كاملة + دعم لغة المستندات مع تغطية RTL |
-| 🧹**مسح كافة النماذج** | مسح قائمة النماذج بنقرة واحدة في تفاصيل المزود |
-| 👁️**عناصر التحكم في الشريط الجانبي**🆕 | إخفاء المكونات وعمليات التكامل من إعدادات المظهر |
-| 📋**نماذج الإصدارات** | قوالب GitHub الموحدة للأخطاء والميزات |
-| 📂**دليل البيانات المخصصة** | تجاوز `DATA_DIR` لموقع التخزين | ### Feature Deep Dive |
+### 🤖 Agent & Protocol Operations (v2.0)
+
+| Feature | What It Does |
+| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
+| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools |
+| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows |
+| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs |
+| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) |
+| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) |
+| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution |
+| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access |
+| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks |
+| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery |
+| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` |
+| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface |
+
+### 🧠 Routing & Intelligence
+
+| Feature | What It Does |
+| ---------------------------------- | ------------------------------------------------------------------------ |
+| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free |
+| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider |
+| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions |
+| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection |
+| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry |
+| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control |
+| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session |
+| 🌐 **Wildcard Router** | `provider/*` dynamic routing |
+| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits |
+| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety |
+| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models |
+| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) |
+| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions |
+| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity |
+| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution |
+| 💬 **System Prompt Injection** | Global behavior controls applied consistently |
+| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows |
+
+### 🎵 Multi-Modal APIs
+
+| Feature | What It Does |
+| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends |
+| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines |
+| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support |
+| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages |
+| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) |
+| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) |
+| 🛡️ **Moderations** | `/v1/moderations` safety checks |
+| 🔀 **Reranking** | `/v1/rerank` for relevance scoring |
+| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache |
+
+### 🛡️ Resilience, Security & Governance
+
+| Feature | What It Does |
+| ----------------------------------- | -------------------------------------------------------------------------------------- |
+| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls |
+| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format |
+| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events |
+| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers |
+| ⚡ **Request Idempotency** | Duplicate protection window |
+| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** |
+| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** |
+| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments |
+| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence |
+| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations |
+| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks |
+| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures |
+| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically |
+| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls |
+| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` |
+| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog |
+
+### 📊 Observability & Analytics
+
+| Feature | What It Does |
+| -------------------------------- | ----------------------------------------------------- |
+| 📝 **Request + Proxy Logging** | Full request/response and proxy logging |
+| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI |
+| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page |
+| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing |
+| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats |
+| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility |
+| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views |
+| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies |
+| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing |
+
+### ☁️ Deployment & Platform
+
+| Feature | What It Does |
+| ------------------------------ | --------------------------------------------------------------------- |
+| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments |
+| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard |
+| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles |
+| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls |
+| 🔄 **Backup/Restore** | Export/import and disaster recovery flows |
+| 🧙 **Onboarding Wizard** | First-run guided setup |
+| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools |
+| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard |
+| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security |
+| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage |
+| 🧹 **Clear All Models** | One-click model list clearing in provider details |
+| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings |
+| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features |
+| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location |
+
+### Feature Deep Dive
#### Smart fallback with practical cost control
@@ -1294,105 +1464,132 @@ Combo: "my-coding-stack"
4. if/kimi-k2-thinking
```
-عند فشل الحصة أو المعدل أو الصحة، ينتقل OmniRoute تلقائيًا إلى المرشح التالي دون التبديل اليدوي.#### Protocol management that is visible and operable
+When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching.
-- يمكن اكتشاف MCP + A2A في واجهة المستخدم والمستندات (غير مخفية)
-- تعرض واجهات برمجة التطبيقات لحالة البروتوكول البيانات التشغيلية المباشرة (`/api/mcp/*`، `/api/a2a/*`)
-- تتضمن لوحات المعلومات إجراءات لعمليات اليوم الثاني (تبديل التحرير والسرد، وإعادة ضبط الكسارة، وإلغاء المهام)#### Translator + validation workflow
+#### Protocol management that is visible and operable
-منطقة المترجم تشمل:
+- MCP + A2A are discoverable in UI and docs (not hidden)
+- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`)
+- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation)
--**الملعب**: طلب عمليات التحقق من التحويل -**أداة اختبار الدردشة**: الطلب/الإجابة الكاملة ذهابًا وإيابًا -**منصة الاختبار**: حالات متعددة في جولة واحدة -**المراقبة المباشرة**: عرض حركة المرور في الوقت الحقيقي
+#### Translator + validation workflow
-بالإضافة إلى التحقق من صحة البروتوكول مع عملاء حقيقيين عبر اختبار تشغيل npm:البروتوكولات:e2e.
+The Translator area includes:
-> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— مرجع الأداة، وتكوينات IDE، وأمثلة العميل
+- **Playground**: request transformation checks
+- **Chat Tester**: full request/response round-trip
+- **Test Bench**: multiple cases in one run
+- **Live Monitor**: real-time traffic view
+
+Plus protocol validation with real clients via `npm run test:protocols:e2e`.
+
+> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples
>
-> 📖**[A2A Server README](src/lib/a2a/README.md)**— المهارات، وأساليب JSON-RPC، والبث، ودورة حياة المهمة## 🧪 Evaluations (Evals)
+> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle
-يشتمل OmniRoute على إطار تقييم مدمج لاختبار جودة استجابة LLM مقابل المجموعة الذهبية. يمكنك الوصول إليه عبر**Analytics → Evals**في لوحة التحكم.### Built-in Golden Set
+## 🧪 Evaluations (Evals)
-تحتوي "OmniRoute Golden Set" المحملة مسبقًا على حالات اختبار لما يلي:
+OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard.
-- تحياتي، الرياضيات، الجغرافيا، توليد التعليمات البرمجية
-- الامتثال لتنسيق JSON والترجمة وإنشاء تخفيض السعر
-- رفض السلامة (المحتوى الضار)، العد، المنطق المنطقي### Evaluation Strategies
+### Built-in Golden Set
-| استراتيجية | الوصف | مثال |
-| ---------------- | ------------------------------------------------------------- | --------------------------------- | --- |
-| `بالضبط` | يجب أن يتطابق الإخراج تمامًا مع | `"4"` |
-| `يحتوي على` | يجب أن يحتوي الإخراج على سلسلة فرعية (غير حساسة لحالة الأحرف) | `"باريس"` |
-| "التعبير العادي" | يجب أن يتطابق الإخراج مع نمط regex | `"1.*2.*3"` |
-| "مخصص" | ترجع دالة JS المخصصة صواب/خطأ | `(الإخراج) => الإخراج.الطول > 10` | --- |
+The pre-loaded "OmniRoute Golden Set" contains test cases for:
+
+- Greetings, math, geography, code generation
+- JSON format compliance, translation, markdown generation
+- Safety refusal (harmful content), counting, boolean logic
+
+### Evaluation Strategies
+
+| Strategy | Description | Example |
+| ---------- | ------------------------------------------------ | -------------------------------- |
+| `exact` | Output must match exactly | `"4"` |
+| `contains` | Output must contain substring (case-insensitive) | `"Paris"` |
+| `regex` | Output must match regex pattern | `"1.*2.*3"` |
+| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` |
+
+---
## 📖 Setup Guide
### Protocol Setup (MCP + A2A)
-<التفاصيل>
+
+🧩 MCP Setup (Model Context Protocol)
-🧩 إعداد MCP (بروتوكول السياق النموذجي)
+Start MCP transport in stdio mode:
-بدء نقل MCP في وضع stdio:```bash
+```bash
omniroute --mcp
+```
-````
+Recommended validation flow:
-تدفق التحقق الموصى به:
+1. Connect your MCP client over stdio.
+2. Run `omniroute_get_health`.
+3. Run `omniroute_list_combos`.
+4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit.
-1. قم بتوصيل عميل MCP الخاص بك عبر stdio.
-2. قم بتشغيل "omniroute_get_health".
-3. قم بتشغيل "omniroute_list_combos".
-4. افتح `/dashboard/mcp` لتأكيد نبضات القلب والنشاط والتدقيق.
+Useful APIs for automation:
-واجهات برمجة التطبيقات المفيدة للأتمتة:
+- `GET /api/mcp/status`
+- `GET /api/mcp/tools`
+- `GET /api/mcp/audit`
+- `GET /api/mcp/audit/stats`
-- `الحصول على /api/mcp/status`
-- `الحصول على /api/mcp/tools`
-- `الحصول على /api/mcp/audit`
-- `الحصول على /api/mcp/audit/stats`
+
-<التفاصيل>
-🤝 إعداد A2A (Agent2Agent)
+
+🤝 A2A Setup (Agent2Agent)
-اكتشف الوكيل:```bash
+Discover the agent:
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
-إرسال مهمة:```bash
+Send a task:
+
+```bash
curl -X POST http://localhost:20128/a2a \
- -H 'content-type: application/json' \
- -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+ -H 'content-type: application/json' \
+ -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+```
-````
+Manage lifecycle:
-إدارة دورة الحياة:
-
-- `الحصول على /api/a2a/status`
-- `الحصول على /api/a2a/tasks`
-- `الحصول على /api/a2a/tasks/:id`
+- `GET /api/a2a/status`
+- `GET /api/a2a/tasks`
+- `GET /api/a2a/tasks/:id`
- `POST /api/a2a/tasks/:id/cancel`
-واجهة المستخدم التشغيلية:
+Operational UI:
-- `/dashboard/a2a` لإمكانية ملاحظة المهمة/الحالة/الدفق وإجراءات الدخان
+- `/dashboard/a2a` for task/state/stream observability and smoke actions
-<التفاصيل>
-🧪 التحقق من صحة البروتوكول الشامل
+
-التحقق من صحة كلا البروتوكولين مع عملاء حقيقيين:```bash
+
+🧪 End-to-end protocol validation
+
+Validate both protocols with real clients:
+
+```bash
npm run test:protocols:e2e
-````
+```
-هذا يتحقق:
+This verifies:
-- اتصال/قائمة/اتصال عميل MCP SDK
-- اكتشاف A2A/إرسال/دفق/حصول على/إلغاء
-- التحقق من البيانات في تدقيق MCP وواجهات برمجة التطبيقات لإدارة المهام A2A
+- MCP SDK client connect/list/call
+- A2A discovery/send/stream/get/cancel
+- Cross-check data in MCP audit and A2A task management APIs
-<التفاصيل>
+
-💳 مقدمو الاشتراكات### Claude Code (Pro/Max)
+
+💳 Subscription Providers
+
+### Claude Code (Pro/Max)
```bash
Dashboard → Providers → Connect Claude Code
@@ -1405,7 +1602,9 @@ Models:
cc/claude-haiku-4-5-20251001
```
-**نصيحة احترافية:**استخدم Opus للمهام المعقدة، وSonnet للسرعة. OmniRoute يتتبع الحصة لكل نموذج!### OpenAI Codex (Plus/Pro)
+**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model!
+
+### OpenAI Codex (Plus/Pro)
```bash
Dashboard → Providers → Connect Codex
@@ -1419,20 +1618,22 @@ Models:
#### Codex Account Limit Management (5h + Weekly)
-يحتوي كل حساب Codex الآن على تبديل السياسة في "لوحة المعلومات -> مقدمي الخدمة":
+Each Codex account now has policy toggles in `Dashboard -> Providers`:
-- `5h` (تشغيل/إيقاف): فرض سياسة عتبة النافذة البالغة 5 ساعات.
-- `أسبوعيًا` (تشغيل/إيقاف): فرض سياسة حد النافذة الأسبوعية.
-- سلوك العتبة: عندما تصل النافذة الممكّنة إلى >=90% من الاستخدام، يتم تخطي هذا الحساب.
-- سلوك التناوب: يقوم OmniRoute بتوجيه حساب Codex المؤهل التالي تلقائيًا.
-- إعادة تعيين السلوك: عندما يمر وقت الموفر `resetAt`، يصبح الحساب مؤهلاً مرة أخرى تلقائيًا.
+- `5h` (ON/OFF): enforce the 5-hour window threshold policy.
+- `Weekly` (ON/OFF): enforce the weekly window threshold policy.
+- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped.
+- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically.
+- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically.
-السيناريوهات:
+Scenarios:
-- `5h ON' + `Weekly ON`: يتم تخطي الحساب عندما تصل أي من النافذتين إلى الحد الأدنى.
-- `إيقاف لمدة 5 ساعات` + `تشغيل أسبوعي`: الاستخدام الأسبوعي فقط يمكنه حظر الحساب.
-- `5 ساعات تشغيل' + `إيقاف أسبوعي`: الاستخدام لمدة 5 ساعات فقط يمكنه حظر الحساب.
-- تم `resetAt`: يعود الحساب إلى التدوير تلقائيًا (لا توجد إعادة تمكين يدوية).### Gemini CLI (FREE 180K/month!)
+- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold.
+- `5h OFF` + `Weekly ON`: only weekly usage can block the account.
+- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account.
+- `resetAt` passed: account re-enters rotation automatically (no manual re-enable).
+
+### Gemini CLI (FREE 180K/month!)
```bash
Dashboard → Providers → Connect Gemini CLI
@@ -1444,7 +1645,9 @@ Models:
gc/gemini-2.5-pro
```
-**أفضل قيمة:**طبقة مجانية ضخمة! استخدم هذا قبل المستويات المدفوعة.### GitHub Copilot
+**Best Value:** Huge free tier! Use this before paid tiers.
+
+### GitHub Copilot
```bash
Dashboard → Providers → Connect GitHub
@@ -1459,74 +1662,91 @@ Models:
-<التفاصيل>
+
+🔑 API Key Providers
-🔑 موفري مفاتيح واجهة برمجة التطبيقات
### NVIDIA NIM (FREE developer access — 70+ models)
+### NVIDIA NIM (FREE developer access — 70+ models)
-1. قم بالتسجيل: [build.nvidia.com](https://build.nvidia.com)
-2. احصل على مفتاح واجهة برمجة التطبيقات (API) مجانًا (يتضمن 1000 نقطة استدلال)
-3. لوحة المعلومات → إضافة موفر → NVIDIA NIM:
- - مفتاح API: `nvapi-your-key`
+1. Sign up: [build.nvidia.com](https://build.nvidia.com)
+2. Get free API key (1000 inference credits included)
+3. Dashboard → Add Provider → NVIDIA NIM:
+ - API Key: `nvapi-your-key`
-**النماذج:**`nvidia/llama-3.3-70b-instruct`، `nvidia/mistral-7b-instruct`، وأكثر من 50 طرازًا آخر
+**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more
-**نصيحة احترافية:**واجهة برمجة التطبيقات المتوافقة مع OpenAI — تعمل بسلاسة مع ترجمة تنسيق OmniRoute!### DeepSeek
+**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation!
-1. قم بالتسجيل: [platform.deepseek.com](https://platform.deepseek.com)
-2. احصل على مفتاح API
-3. لوحة المعلومات → إضافة موفر → DeepSeek
+### DeepSeek
-**النماذج:**`deepseek/deepseek-chat`، `deepseek/deepseek-coder`### Groq (Free Tier Available!)
+1. Sign up: [platform.deepseek.com](https://platform.deepseek.com)
+2. Get API key
+3. Dashboard → Add Provider → DeepSeek
-1. قم بالتسجيل: [console.groq.com](https://console.groq.com)
-2. احصل على مفتاح API (الطبقة المجانية متضمنة)
-3. لوحة المعلومات → إضافة موفر → Groq
+**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder`
-**النماذج:**`groq/llama-3.3-70b`، `groq/mixtral-8x7b`
+### Groq (Free Tier Available!)
-**نصيحة احترافية:**استنتاج فائق السرعة — الأفضل للبرمجة في الوقت الفعلي!### OpenRouter (100+ Models)
+1. Sign up: [console.groq.com](https://console.groq.com)
+2. Get API key (free tier included)
+3. Dashboard → Add Provider → Groq
-1. قم بالتسجيل: [openrouter.ai](https://openrouter.ai)
-2. احصل على مفتاح API
-3. لوحة المعلومات → إضافة موفر → OpenRouter
+**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b`
-**النماذج:**يمكنك الوصول إلى أكثر من 100 نموذج من جميع المزودين الرئيسيين من خلال مفتاح واجهة برمجة التطبيقات (API) واحد.
+**Pro Tip:** Ultra-fast inference — best for real-time coding!
-**سلوك لوحة المعلومات:**تتم إدارة نماذج OpenRouter من**النماذج المتوفرة**. تعمل عمليات الإضافة والاستيراد والمزامنة التلقائية يدويًا على تحديث نفس القائمة.
+### OpenRouter (100+ Models)
-<التفاصيل>
+1. Sign up: [openrouter.ai](https://openrouter.ai)
+2. Get API key
+3. Dashboard → Add Provider → OpenRouter
-💰 مقدمو الخدمة الرخيصة (النسخ الاحتياطي)### GLM-4.7 (Daily reset, $0.6/1M)
+**Models:** Access 100+ models from all major providers through a single API key.
-1. قم بالتسجيل: [Zhipu AI](https://open.bigmodel.cn/)
-2. احصل على مفتاح API من خطة الترميز
-3. لوحة المعلومات → إضافة مفتاح واجهة برمجة التطبيقات:
- - المزود: `glm`
- - مفتاح واجهة برمجة التطبيقات: "مفتاحك".
+**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list.
-**الاستخدام:**`glm/glm-4.7`
+
-**نصيحة احترافية:**توفر خطة البرمجة حصة 3× بتكلفة 1/7! إعادة الضبط يوميًا الساعة 10:00 صباحًا.### MiniMax M2.1 (5h reset, $0.20/1M)
+
+💰 Cheap Providers (Backup)
-1. قم بالتسجيل: [MiniMax](https://www.minimax.io/)
-2. احصل على مفتاح API
-3. لوحة المعلومات → إضافة مفتاح API
+### GLM-4.7 (Daily reset, $0.6/1M)
-**الاستخدام:**`minimax/MiniMax-M2.1`
+1. Sign up: [Zhipu AI](https://open.bigmodel.cn/)
+2. Get API key from Coding Plan
+3. Dashboard → Add API Key:
+ - Provider: `glm`
+ - API Key: `your-key`
-**نصيحة احترافية:**الخيار الأرخص للسياق الطويل (مليون رمز)!### Kimi K2 ($9/month flat)
+**Use:** `glm/glm-4.7`
-1. اشترك: [Moonshot AI](https://platform.moonshot.ai/)
-2. احصل على مفتاح API
-3. لوحة المعلومات → إضافة مفتاح API
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM.
-**الاستخدام:**`كيمي/كيمي-أحدث`
+### MiniMax M2.1 (5h reset, $0.20/1M)
-**نصيحة احترافية:**سعر ثابت قدره 9 دولارات شهريًا مقابل 10 ملايين رمز مميز = 0.90 دولارًا أمريكيًا/مليون تكلفة فعالة!
+1. Sign up: [MiniMax](https://www.minimax.io/)
+2. Get API key
+3. Dashboard → Add API Key
-<التفاصيل>
+**Use:** `minimax/MiniMax-M2.1`
-🆓 مقدمو الخدمة مجانًا (النسخ الاحتياطي في حالات الطوارئ)### Qoder (5 FREE models via OAuth)
+**Pro Tip:** Cheapest option for long context (1M tokens)!
+
+### Kimi K2 ($9/month flat)
+
+1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `kimi/kimi-latest`
+
+**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost!
+
+
+
+
+🆓 FREE Providers (Emergency Backup)
+
+### Qoder (5 FREE models via OAuth)
```bash
Dashboard → Connect Qoder
@@ -1567,9 +1787,10 @@ Models:
-<التفاصيل>
+
+🎨 Create Combos
-🎨 أنشئ مجموعات
### Example 1: Maximize Subscription → Cheap Backup
+### Example 1: Maximize Subscription → Cheap Backup
```
Dashboard → Combos → Create New
@@ -1597,9 +1818,10 @@ Cost: $0 forever!
-<التفاصيل>
+
+🔧 CLI Integration
-🔧 تكامل CLI
### Cursor IDE
+### Cursor IDE
```
Settings → Models → Advanced:
@@ -1610,7 +1832,9 @@ Settings → Models → Advanced:
### Claude Code
-استخدم صفحة**أدوات CLI**في لوحة المعلومات للتكوين بنقرة واحدة، أو قم بتحرير `~/.claude/settings.json` يدويًا.### Codex CLI
+Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually.
+
+### Codex CLI
```bash
export OPENAI_BASE_URL="http://localhost:20128"
@@ -1621,12 +1845,15 @@ codex "your prompt"
### OpenClaw
-**الخيار 1 — لوحة التحكم (مستحسن):**```
+**Option 1 — Dashboard (recommended):**
+
+```
Dashboard → CLI Tools → OpenClaw → Select Model → Apply
+```
-````
+**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`:
-**الخيار 2 - يدويًا:**تحرير `~/.openclaw/openclaw.json`:```json
+```json
{
"models": {
"providers": {
@@ -1638,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply
}
}
}
-````
+```
-> **ملاحظة:**يعمل OpenClaw فقط مع OmniRoute المحلي. استخدم "127.0.0.1" بدلاً من "المضيف المحلي" لتجنب مشكلات دقة IPv6.### Cline / Continue / RooCode
+> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues.
+
+### Cline / Continue / RooCode
```
Settings → API Configuration:
@@ -1652,15 +1881,17 @@ Settings → API Configuration:
### OpenCode
-**الخطوة 1:**أضف OmniRoute كموفر مخصص:```bash
+**Step 1:** Add OmniRoute as a custom provider:
+
+```bash
opencode
/connect
-
# Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key
+```
-````
+**Step 2:** Create/edit `opencode.json` in your project root:
-**الخطوة 2:**إنشاء/تحرير `opencode.json` في جذر مشروعك:```json
+```json
{
"$schema": "https://opencode.ai/config.json",
"provider": {
@@ -1678,117 +1909,130 @@ opencode
}
}
}
-````
+```
-**الخطوة 3:**حدد النموذج في OpenCode:```bash
+**Step 3:** Select the model in OpenCode:
+
+```bash
/models
-
# Select any OmniRoute model from the list
+```
-````
+> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard.
->**نصيحة:**أضف أي نموذج متوفر في نقطة نهاية OmniRoute `/v1/models` إلى قسم `models`. استخدم التنسيق "provider/model-id" من لوحة معلومات OmniRoute.
+
---
## استكشاف الأخطاء
-<التفاصيل>
-انقر لتوسيع دليل استكشاف الأخطاء وإصلاحها
+
+Click to expand troubleshooting guide
-**"نموذج اللغة لم يقدم رسائل"**
+**"Language model did not provide messages"**
-- استنفدت حصة الموفر → تحقق من تعقب حصة الموفر في لوحة المعلومات
-- الحل: استخدم خيار التحرير والسرد الاحتياطي أو قم بالتبديل إلى مستوى أرخص
+- Provider quota exhausted → Check dashboard quota tracker
+- Solution: Use combo fallback or switch to cheaper tier
-**الحد من المعدل**
+**Rate limiting**
-- حصة الاشتراك المحددة → الرجوع إلى GLM/MiniMax
-- إضافة التحرير والسرد: `cc/clude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Subscription quota out → Fallback to GLM/MiniMax
+- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-**انتهت صلاحية رمز OAuth**
+**OAuth token expired**
-- يتم التحديث تلقائيًا بواسطة OmniRoute
-- إذا استمرت المشكلات: لوحة المعلومات → الموفر → إعادة الاتصال
+- Auto-refreshed by OmniRoute
+- If issues persist: Dashboard → Provider → Reconnect
-**تكاليف مرتفعة**
+**High costs**
-- التحقق من إحصائيات الاستخدام في لوحة المعلومات → التكاليف
-- تبديل النموذج الأساسي إلى GLM/MiniMax
-- استخدم الطبقة المجانية (Gemini CLI، Qoder) للمهام غير الحرجة
+- Check usage stats in Dashboard → Costs
+- Switch primary model to GLM/MiniMax
+- Use free tier (Gemini CLI, Qoder) for non-critical tasks
-**منافذ لوحة المعلومات/واجهة برمجة التطبيقات غير صحيحة**
+**Dashboard/API ports are wrong**
-- `PORT` هو المنفذ الأساسي الأساسي (ومنفذ API افتراضيًا)
-- `API_PORT` يتخطى فقط مستمع واجهة برمجة التطبيقات المتوافق مع OpenAI
-- `DASHBOARD_PORT` يتجاوز مستمع لوحة المعلومات/Next.js فقط
-- قم بتعيين `NEXT_PUBLIC_BASE_URL` على لوحة التحكم/عنوان URL العام (لردود اتصال OAuth)
+- `PORT` is the canonical base port (and API port by default)
+- `API_PORT` overrides only OpenAI-compatible API listener
+- `DASHBOARD_PORT` overrides only dashboard/Next.js listener
+- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks)
-**أخطاء المزامنة السحابية**
+**Cloud sync errors**
-- تحقق من نقاط `BASE_URL` لمثيلك قيد التشغيل
-- تحقق من نقاط `CLOUD_URL` إلى نقطة النهاية السحابية المتوقعة
-- حافظ على محاذاة قيم `NEXT_PUBLIC_*` مع القيم الموجودة على جانب الخادم
+- Verify `BASE_URL` points to your running instance
+- Verify `CLOUD_URL` points to your expected cloud endpoint
+- Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**تسجيل الدخول الأول لا يعمل**
+**First login not working**
-- حدد "INITIAL_PASSWORD" في ".env".
-- في حالة عدم تعيينها، تكون كلمة المرور الاحتياطية هي `123456`
+- Check `INITIAL_PASSWORD` in `.env`
+- If unset, fallback password is `123456`
-**لا توجد سجلات الطلب**
+**No request logs**
-- تتم كتابة عناصر الطلب إلى `DATA_DIR/call_logs/` كملف JSON واحد لكل طلب
-- تمكين التقاط خط الأنابيب من لوحة المعلومات → السجلات → طلب السجلات إذا كنت بحاجة إلى حمولات مفصلة لكل مرحلة
-- اضبط `APP_LOG_TO_FILE=true` إذا كنت تريد أيضًا وجود سجلات لوحدة تحكم التطبيق في `logs/application/app.log`
-- اضبط `APP_LOG_MAX_FILE_SIZE`، و`APP_LOG_RETENTION_DAYS`، و`APP_LOG_MAX_FILES`، و`CALL_LOG_MAX_ENTRIES` حسب الحاجة
+- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request
+- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads
+- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log`
+- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed
-**يظهر اختبار الاتصال "غير صالح" لمقدمي الخدمات المتوافقين مع OpenAI**
+**Connection test shows "Invalid" for OpenAI-compatible providers**
-- لا يكشف العديد من مقدمي الخدمة عن نقطة نهاية `/models`
-- يتضمن OmniRoute v1.0.6+ التحقق الاحتياطي من خلال إكمال الدردشة
-- تأكد من أن عنوان URL الأساسي يتضمن لاحقة `/v1`### 🔐 OAuth on a Remote Server
+- Many providers don't expose a `/models` endpoint
+- OmniRoute v1.0.6+ includes fallback validation via chat completions
+- Ensure base URL includes `/v1` suffix
+
+### 🔐 OAuth on a Remote Server
->**⚠️ مهم للمستخدمين الذين يقومون بتشغيل OmniRoute على VPS أو Docker أو أي خادم بعيد**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
+> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server**
-يستخدم موفرو**Antigravity**و**Gemini CLI****Google OAuth 2.0**. تتطلب Google أن يكون `redirect_uri` في تدفق OAuth مطابقًا تمامًا لأحد معرفات URI المسجلة مسبقًا في Google Cloud Console للتطبيق.
+#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
-يتم تسجيل بيانات اعتماد OAuth المجمعة في OmniRoute**لـ `المضيف المحلي` فقط**. عند الوصول إلى OmniRoute على خادم بعيد (على سبيل المثال، `https://omniroute.myserver.com`)، يرفض Google المصادقة باستخدام:```
+The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console.
+
+The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solution: Configure your own OAuth credentials
-يلزمك إنشاء**OAuth 2.0 Client ID**في Google Cloud Console باستخدام معرف URI الخاص بخادمك.#### Step-by-step
+You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI.
-**1. افتح Google Cloud Console**
+#### Step-by-step
-انتقل إلى: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
+**1. Open Google Cloud Console**
-**2. قم بإنشاء معرف عميل OAuth 2.0 جديد**
+Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-- انقر على**"+ إنشاء بيانات اعتماد"**→**"معرف عميل OAuth"**
-- نوع التطبيق:**"تطبيق ويب"**
-- الاسم: أي شيء تريده (على سبيل المثال، "OmniRoute Remote")
+**2. Create a new OAuth 2.0 Client ID**
-**3. أضف عناوين URI لإعادة التوجيه المعتمدة**
+- Click **"+ Create Credentials"** → **"OAuth client ID"**
+- Application type: **"Web application"**
+- Name: anything you like (e.g. `OmniRoute Remote`)
-في الحقل**"عناوين URI لإعادة التوجيه المعتمدة"**، أضف:```
+**3. Add Authorized Redirect URIs**
+
+In the **"Authorized redirect URIs"** field, add:
+
+```
https://your-server.com/callback
+```
-````
+> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`).
-> استبدل "your-server.com" بنطاق الخادم الخاص بك أو عنوان IP (قم بتضمين المنفذ إذا لزم الأمر، على سبيل المثال "http://45.33.32.156:20128/callback").
+**4. Save and copy the credentials**
-**4. حفظ ونسخ بيانات الاعتماد**
+After creating, Google will show the **Client ID** and **Client Secret**.
-بعد الإنشاء، ستعرض Google**معرف العميل**و**سر العميل**.
+**5. Set environment variables**
-**5. تعيين متغيرات البيئة**
+In your `.env` (or Docker environment variables):
-في `.env` (أو متغيرات بيئة Docker):```bash
+```bash
# For Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
@@ -1797,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
-````
+```
-**6. أعد تشغيل OmniRoute**```bash
+**6. Restart OmniRoute**
+```bash
# npm:
-
npm run dev
# Docker:
-
docker restart omniroute
+```
-````
+**7. Try connecting again**
-**7. حاول الاتصال مرة أخرى**
+Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth
-لوحة المعلومات → الموفرون → Antigravity (أو Gemini CLI) → OAuth
+Google will now redirect correctly to `https://your-server.com/callback`.
-سيقوم Google الآن بإعادة التوجيه بشكل صحيح إلى `https://your-server.com/callback`.---
+---
#### Temporary workaround (without custom credentials)
-إذا كنت لا ترغب في إعداد بيانات الاعتماد الخاصة بك الآن، فلا يزال بإمكانك استخدام**تدفق عنوان URL اليدوي**:
+If you don't want to set up your own credentials right now, you can still use the **manual URL flow**:
-1. يفتح OmniRoute عنوان URL لتفويض Google
-2. بعد التفويض، يحاول Google إعادة التوجيه إلى "المضيف المحلي" (والذي يفشل على الخادم البعيد)
-3.**انسخ عنوان URL الكامل**من شريط عنوان المتصفح (حتى لو لم يتم تحميل الصفحة)
-4. الصق عنوان URL هذا في الحقل الموضح في نموذج اتصال OmniRoute
-5. انقر**"اتصال"**
+1. OmniRoute opens the Google authorization URL
+2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server)
+3. **Copy the full URL** from your browser's address bar (even if the page doesn't load)
+4. Paste that URL into the field shown in the OmniRoute connection modal
+5. Click **"Connect"**
-> يعمل هذا لأن رمز التفويض الموجود في عنوان URL صالح بغض النظر عما إذا تم تحميل صفحة إعادة التوجيه أم لا.---
+> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded.
-<التفاصيل>
-🇧🇷 النسخة البرتغالية
#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+---
-تم إثبات**Antigravity**و**Gemini CLI**باستخدام**Google OAuth 2.0**للمصادقة. تطلب Google أن يتم استخدام `redirect_uri` دون تدفق OAuth**بالتأكيد**إلى عناوين URI المسبقة لتطبيق Google Cloud Console.
+
+🇧🇷 Versão em Português
-نظرًا لأن اعتمادات OAuth المُدخلة ليست في OmniRoute، فهي عبارة عن سجلات**apenas لـ `المضيف المحلي`**. عند الوصول إلى OmniRoute من خادم بعيد (على سبيل المثال: `https://omniroute.meuservidor.com`)، أو تحصل Google على مصادقة عبر:```
+#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+
+Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo.
+
+As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solução: Configure suas próprias credenciais OAuth
-يجب عليك إنشاء**OAuth 2.0 Client ID**على Google Cloud Console باستخدام URI لخادمك.#### Passo a passo
+Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor.
-**1. الوصول إلى Google Cloud Console**
+#### Passo a passo
-العبرة: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
+**1. Acesse o Google Cloud Console**
-**2. طلب معرف عميل OAuth 2.0**
+Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-- انقر على**"+ إنشاء بيانات الاعتماد"**→**"معرف عميل OAuth"**
-- نوع التطبيق:**"تطبيق ويب"**
-- الاسم: escolha qualquer nome (على سبيل المثال: `OmniRoute Remote`)
+**2. Crie um novo OAuth 2.0 Client ID**
-**3. Adicione كمحددات URI لإعادة التوجيه المعتمدة**
+- Clique em **"+ Create Credentials"** → **"OAuth client ID"**
+- Tipo de aplicativo: **"Web application"**
+- Nome: escolha qualquer nome (ex: `OmniRoute Remote`)
-ليس هناك مجال**"عناوين URI لإعادة التوجيه المعتمدة"**، أضف:```
+**3. Adicione as Authorized Redirect URIs**
+
+No campo **"Authorized redirect URIs"**, adicione:
+
+```
https://seu-servidor.com/callback
+```
-````
+> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`).
-> استبدل `seu-servidor.com` بمنطقتك أو IP بخادمك (بما في ذلك البوابة إذا لزم الأمر، على سبيل المثال: `http://45.33.32.156:20128/callback`).
+**4. Salve e copie as credenciais**
-**4. حفظ ونسخ كموثقات**
+Após criar, o Google mostrará o **Client ID** e o **Client Secret**.
-وبعد ذلك، قم بإنشاء أو عرض Google o**معرف العميل**أو**سر العميل**.
+**5. Configure as variáveis de ambiente**
-**5. تكوين كمتغيرات البيئة**
+No seu `.env` (ou nas variáveis de ambiente do Docker):
-ليس لديك `.env` (أو في بيئة Docker المتنوعة):```bash
+```bash
# Para Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
@@ -1876,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
-````
+```
-**6. Reinicie أو OmniRoute**```bash
+**6. Reinicie o OmniRoute**
+```bash
# Se usando npm:
-
npm run dev
# Se usando Docker:
-
docker restart omniroute
+```
-````
+**7. Tente conectar novamente**
-**7. خيمة تواصل جديدة**
+Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth
-لوحة المعلومات → الموفرون → Antigravity (ou Gemini CLI) → OAuth
+Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará.
-قم بإعادة توجيه Google بشكل صحيح إلى `https://seu-servidor.com/callback` ووظيفة المصادقة.---
+---
#### Workaround temporário (sem configurar credenciais próprias)
-إذا لم ترغب في إنشاء بيانات اعتماد خاصة بك منذ الآن، فمن الممكن استخدام التدفق**دليل URL**:
+Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**:
-1. يفتح OmniRoute عنوان URL لتفويض Google
-2. نسمح لك بأن تقوم Google بإعادة التوجيه إلى "المضيف المحلي" (الذي لا يوجد خادم عن بعد)
-3.**انسخ عنوان URL كاملاً**من شريط الإدخال في متصفحك (حتى لا يتم نقل الصفحة)
-4. هذا هو عنوان URL الذي يظهر في وضع الاتصال بـ OmniRoute
-5. انقر على**"الاتصال"**
+1. O OmniRoute abrirá a URL de autorização do Google
+2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto)
+3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue)
+4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute
+5. Clique em **"Connect"**
-> يعمل هذا الحل البديل لأن رمز التفويض الموجود على عنوان URL يكون صالحًا بشكل مستقل لإعادة التوجيه حيث يتم تحميله أو لا.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+
+
---
@@ -1914,64 +2171,73 @@ docker restart omniroute
## 🛠️ Tech Stack
-<التفاصيل>
-انقر لتوسيع تفاصيل المجموعة التقنية
+
+Click to expand tech stack details
--**وقت التشغيل**: Node.js 18–22 LTS (⚠️ Node.js 24+**غير مدعومة**— الثنائيات الأصلية `better-sqlite3` غير متوافقة)
--**اللغة**: TypeScript 5.9 —**TypeScript بنسبة 100%**عبر `src/` و`open-sse/` (لا يوجد `any` في الوحدات الأساسية منذ الإصدار 2.0)
--**الإطار**: Next.js 16 + React 19 + Tailwind CSS 4
--**قاعدة البيانات**: LowDB (JSON) + SQLite (حالة المجال + سجلات الوكيل + تدقيق MCP + قرارات التوجيه)
--**المخططات**: Zod (التحقق من صحة الإدخال/الإخراج لأداة MCP، وعقود API)
--**البروتوكولات**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
--**البث**: الأحداث المرسلة من الخادم (SSE)
--**المصادقة**: OAuth 2.0 (PKCE) + JWT + مفاتيح API + ترخيص نطاق MCP
--**الاختبار**: مشغل اختبار Node.js + Vitest (أكثر من 900 اختبار بما في ذلك الوحدة والتكامل وE2E)
--**CI/CD**: إجراءات GitHub (نشر npm التلقائي + Docker Hub عند الإصدار)
--**الموقع الإلكتروني**: [omniroute.online](https://omniroute.online)
--**الحزمة**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
--**دوكر**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
--**المرونة**: قاطع الدائرة، والتراجع الأسي، وقطيع مكافحة الرعد، وانتحال TLS، والإصلاح الذاتي للتحرير والسرد التلقائي
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible)
+- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0)
+- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4
+- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions)
+- **Schemas**: Zod (MCP tool I/O validation, API contracts)
+- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
+- **Streaming**: Server-Sent Events (SSE)
+- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization
+- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E)
+- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release)
+- **Website**: [omniroute.online](https://omniroute.online)
+- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
+- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
+- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing
+
+
---
## التوثيق
-| وثيقة | الوصف |
+| Document | Description |
| ---------------------------------------------- | --------------------------------------------------- |
-| [دليل المستخدم](docs/USER_GUIDE.md) | مقدمو الخدمات، والمجموعات، وتكامل CLI، والنشر |
-| [مرجع واجهة برمجة التطبيقات](docs/API_REFERENCE.md) | جميع نقاط النهاية مع الأمثلة |
-| [خادم MCP](open-sse/mcp-server/README.md) | 16 أدوات MCP وتكوينات IDE وعملاء Python/TS/Go |
-| [خادم A2A](src/lib/a2a/README.md) | بروتوكول JSON-RPC 2.0، المهارات، التدفق، إدارة المهام |
-| [محرك التحرير والسرد التلقائي](docs/auto-combo.md) | تسجيل 6 عوامل، حزم الوضع، الشفاء الذاتي |
-| [استكشاف الأخطاء وإصلاحها](docs/TROUBLESHOOTING.md) | المشاكل والحلول الشائعة |
-| [هندسة معمارية](docs/ARCHITECTURE.md) | بنية النظام والداخلية |
-| [مساهمة](CONTRIBUTING.md) | إعداد التطوير والمبادئ التوجيهية |
-| [مواصفات OpenAPI](docs/openapi.yaml) | مواصفات OpenAPI 3.0 |
-| [سياسة الأمان](SECURITY.md) | الإبلاغ عن الثغرات الأمنية والممارسات الأمنية |
-| [نشر الجهاز الافتراضي](docs/VM_DEPLOYMENT_GUIDE.md) | الدليل الكامل: إعداد VM + nginx + Cloudflare |
-| [معرض الميزات](docs/FEATURES.md) | جولة لوحة القيادة المرئية مع لقطات الشاشة |
-| [قائمة مراجعة الإصدار](docs/RELEASE_CHECKLIST.md) | خطوات التحقق من صحة الإصدار المسبق |---
+| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment |
+| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples |
+| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients |
+| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt |
+| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing |
+| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation |
+| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions |
+| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals |
+| [Contributing](CONTRIBUTING.md) | Development setup and guidelines |
+| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification |
+| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices |
+| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup |
+| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots |
+| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps |
+
+---
## 🗺️ Roadmap
-يحتوي OmniRoute على**210+ ميزات مخطط لها**عبر مراحل تطوير متعددة. فيما يلي المجالات الرئيسية:
+OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas:
-| الفئة | الميزات المخططة | أبرز الأحداث |
+| Category | Planned Features | Highlights |
| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- |
-| 🧠**التوجيه والاستخبارات**| 25+ | التوجيه ذو زمن الاستجابة الأقل، والتوجيه القائم على العلامات، والاختبار المبدئي للحصة، واختيار حساب P2C |
-| 🔒**الأمان والامتثال**| 20+ | تقوية SSRF، وإخفاء بيانات الاعتماد، والحد الأقصى للمعدل لكل نقطة نهاية، وتحديد نطاق مفتاح الإدارة |
-| 📊**قابلية الملاحظة**| 15+ | تكامل OpenTelemetry ومراقبة الحصص في الوقت الفعلي وتتبع التكلفة لكل نموذج |
-| 🔄**تكامل الموفر**| 20+ | تسجيل النموذج الديناميكي، فترات تهدئة الموفر، الدستور الغذائي متعدد الحسابات، تحليل حصة الطيار المساعد |
-| ⚡**الأداء**| 15+ | طبقة ذاكرة التخزين المؤقت المزدوجة، ذاكرة التخزين المؤقت السريعة، ذاكرة التخزين المؤقت للاستجابة، استمرار البث، واجهة برمجة التطبيقات الدفعية |
-| 🌐**النظام البيئي**| 10+ | WebSocket API، إعادة تحميل التكوين السريع، مخزن التكوين الموزع، الوضع التجاري |### 🔜 Coming Soon
+| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection |
+| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping |
+| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model |
+| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing |
+| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API |
+| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode |
-- 🔗**تكامل OpenCode**— دعم الموفر الأصلي لـ OpenCode AI IDE للترميز
-- 🔗**تكامل TRAE**— الدعم الكامل لإطار تطوير TRAE AI
-- 📦**Batch API**— معالجة الدفعات غير المتزامنة للطلبات المجمعة
-- 🎯**التوجيه المعتمد على العلامات**— توجيه الطلبات بناءً على العلامات المخصصة والبيانات الوصفية
-- 💰**إستراتيجية أقل تكلفة**— تحديد أرخص مزود متاح تلقائيًا
+### 🔜 Coming Soon
-> 📝 مواصفات الميزات الكاملة متوفرة في [`docs/new-features/`](docs/new-features/) (217 مواصفات تفصيلية)---
+- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE
+- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework
+- 📦 **Batch API** — Asynchronous batch processing for bulk requests
+- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata
+- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider
+
+> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs)
+
+---
## 👥 Contributors
@@ -1979,18 +2245,20 @@ docker restart omniroute
### How to Contribute
-1. شوكة المستودع
-2. قم بإنشاء فرع الميزات الخاص بك (`git checkout -b feature/amazing-feature`)
-3. تنفيذ التغييرات ("git الالتزام -m "إضافة ميزة مذهلة")
-4. ادفع إلى الفرع ("ميزة git Push Origin/ميزة مذهلة")
-5. افتح طلب السحب
+1. Fork the repository
+2. Create your feature branch (`git checkout -b feature/amazing-feature`)
+3. Commit your changes (`git commit -m 'Add amazing feature'`)
+4. Push to the branch (`git push origin feature/amazing-feature`)
+5. Open a Pull Request
-راجع [CONTRIBUTING.md](CONTRIBUTING.md) للحصول على إرشادات مفصلة.### Releasing a New Version
+See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines.
+
+### Releasing a New Version
```bash
# Create a release — npm publish happens automatically
gh release create v2.0.0 --title "v2.0.0" --generate-notes
-````
+```
---
@@ -2002,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes
## 🙏 Acknowledgments
-شكر خاص لـ**[9router](https://github.com/decolua/9router)**بواسطة**[decolua](https://github.com/decolua)**— المشروع الأصلي الذي ألهم هذه الشوكة. يعتمد OmniRoute على هذا الأساس المذهل مع ميزات إضافية وواجهات برمجة التطبيقات متعددة الوسائط وإعادة كتابة TypeScript كاملة.
+Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite.
-شكر خاص لـ**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— تطبيق Go الأصلي الذي ألهم منفذ JavaScript هذا.---
+Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port.
+
+---
## الرخصة
-ترخيص MIT - راجع [الترخيص](الترخيص) للحصول على التفاصيل.---
+MIT License - see [LICENSE](LICENSE) for details.
+
+---
Built with ❤️ for developers who code 24/7
diff --git a/docs/i18n/ar/docs/ARCHITECTURE.md b/docs/i18n/ar/docs/ARCHITECTURE.md
index fb5c25ffba..ec6712920b 100644
--- a/docs/i18n/ar/docs/ARCHITECTURE.md
+++ b/docs/i18n/ar/docs/ARCHITECTURE.md
@@ -4,257 +4,291 @@
---
-_آخر تحديث: 2026-03-28_## الملخص التنفيذي
-OmniRoute عبارة عن بوابة توجيه نقطة تعمل بالذكاء الاصطناعي ولوحة معلومات مبنية على Next.js.
-وهو يوفر نقطة نهاية واحدة متوافقة مع OpenAI (`/v1/*`) ويوجه حركة المرور عبر العديد من الخدمات الموفري الأولية مع الترجمة والاحتياط وتحديث الرمز المميز وتتبع الاستخدام.
-التان الأساسية:
+_Last updated: 2026-03-28_
-- سطح API متوافق مع OpenAI لـ CLI/الأدوات (28 منتجًا)
-- ترجمة الطلب/الاستجابة عبر التنسيقات الموفر
-- نموذج بناء التحرير والسرد (سلسلة الارتباطات المتعددة)
-- موازنة حساب الحساب (حسابات متعددة لكل شخص)
-- إدارة اتصال موفر OAuth + API-key
-- إنشاء التضمين عبر `/v1/embeddings` (6 مقدمي خدمات، 9 نماذج)
-- إنشاء الصور عبر `/v1/images/Generation` (4 مقدمي خدمات، 9 نماذج)
-- فكر في تحليل العلامات (`
...`) لنماذج الاستدلال
-- تحديد القيمة للتوافق مع OpenAI SDK
-- تطبيع الدور (المطور → النظام، النظام → المستخدم) للتوافق بين الموفرين
-- تحويل المنتج منظم (json_schema → Gemini ResponseSchema)
-- الثبات المحلي لمقدمي الخدمات والمفاتيح والأسماء المستعارة والمجموعات والإعدادات والتسعير
-- تتبع تكلفة/التكلفة وتسجيل الطلب
-- نوبات سحابية اختيارية للأجهزة/الحالة الثابتة
-- القائمة الخاصة بها/القائمة المحظورة لـ IP للتحكم في الوصول إلى واجهة برمجة التطبيقات
-- التفكير في إدارة الميزانية (العبور / التلقائي / المقصود / التكيفي)
-- هيكل البناء العالمي
-- تتبع البصمات
-- تحديد المحسن لكل حساب مع الملفات الشخصية الخاصة بالمزود
-- تقطع فاصل لمرونة المورد
-- حماية القطيع ضد الرعد مع موتكس
-- ذاكرة التخزين المؤقتة لإلغاء البيانات المكررة للطلبة المستندية للتوقيع
-- المجال: توفر النموذج، وقواعد التكلفة، والسياسة الاحتياطية، وسياسة فك الضغط
-- فرانسيسكوية المجال المجال (ذاكرة التخزين المؤقتة للكتاب في SQLite للاحتياطيات والميزانيات وفتح قواطع الضوء)
-- السياسة التي تحدد الطلب المركزي (التأمين → الميزانية → الاحتياطي)
-- طلب القياس عن بعد مع تجميع الكمون ص50/ص95/ص99
-- معرف الارتباط (X-Request-Id) للتتبع الشامل
-- تسجيل تدقيق كامل مع إلغاء الاشتراك لمفتاح API
-- إطار تقييمي وجودة LLM
-- لوحة تحكم واجهة المستخدم المرنة مع فاصل زمني في العمل
-- مفري OAuth المطاطيون (12 وحدة ضمن `src/lib/oauth/providers/`)
+## Executive Summary
-وقت نموذج التشغيل الأساسي:
+OmniRoute is a local AI routing gateway and dashboard built on Next.js.
+It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking.
-- تقوم مسارات تطبيق Next.js ضمن `src/app/api/*` ولتتمكن كل من واجهات تطبيقات برمجة لوحة المعلومات وواجهات برمجة تطبيقات التوافق
-- نواة توجيه/SSE اشترك في `src/sse/*` + `open-sse/*` تمويل مع تنفيذ الموفر والترجمة والتدفق والرجوع والاستخدام## النطاق والحدود### In Scope
+Core capabilities:
-- وقت تشغيل البوابة المحلية
-- واجهات برمجة التطبيقات المبتكرة للوحة المعلومات
-- مصادقة الموفر وتحديث الرمز المميز
-- طلب الترجمة و التدفق SSE
-- الحالة المحلية + استمرارية الاستخدام
-- نوبات سحابية اختيارية### خارج النطاق
+- OpenAI-compatible API surface for CLI/tools (28 providers)
+- Request/response translation across provider formats
+- Model combo fallback (multi-model sequence)
+- Account-level fallback (multi-account per provider)
+- OAuth + API-key provider connection management
+- Embedding generation via `/v1/embeddings` (6 providers, 9 models)
+- Image generation via `/v1/images/generations` (4 providers, 9 models)
+- Think tag parsing (`
...`) for reasoning models
+- Response sanitization for strict OpenAI SDK compatibility
+- Role normalization (developer→system, system→user) for cross-provider compatibility
+- Structured output conversion (json_schema → Gemini responseSchema)
+- Local persistence for providers, keys, aliases, combos, settings, pricing
+- Usage/cost tracking and request logging
+- Optional cloud sync for multi-device/state sync
+- IP allowlist/blocklist for API access control
+- Thinking budget management (passthrough/auto/custom/adaptive)
+- Global system prompt injection
+- Session tracking and fingerprinting
+- Per-account enhanced rate limiting with provider-specific profiles
+- Circuit breaker pattern for provider resilience
+- Anti-thundering herd protection with mutex locking
+- Signature-based request deduplication cache
+- Domain layer: model availability, cost rules, fallback policy, lockout policy
+- Context Relay: session handoff summaries for account rotation continuity
+- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers)
+- Policy engine for centralized request evaluation (lockout → budget → fallback)
+- Request telemetry with p50/p95/p99 latency aggregation
+- Correlation ID (X-Request-Id) for end-to-end tracing
+- Compliance audit logging with opt-out per API key
+- Eval framework for LLM quality assurance
+- Resilience UI dashboard with real-time circuit breaker status
+- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`)
-- تنفيذ خدمة السحابية خلف `NEXT_PUBLIC_CLOUD_URL`
-- مستوى تحرير السودان/مستوى التحكم خارج نطاق العمل
-- ثنائيات CLI الخارجية نفسها (Claude CLI، Codex CLI، وما إلى ذلك) ## سطح لوحة القيادة (الحالي)
+Primary runtime model:
-الصفحة الرئيسية ضمن `src/app/(dashboard)/dashboard/`:
+- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs
+- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage
-- `/dashboard` - بداية سريعة + نظرة عامة على الموفر
-- `/dashboard/endpoint` - وكيل نقطة النهاية + علامات نهاية نقطة النهاية MCP + A2A + API
-- `/dashboard/providers` - اتصالات الموفر وبيانات الاعتماد
-- `/dashboard/combos` - إستراتيجيات التحرير والسرد والقوالب وقواعد توجيه التطورات
-- `/dashboard/costs` - تجميع الأسعار ورؤية الأسعار
-- `/dashboard/analytics` - تحليلات تعاطيات البناء
-- `/dashboard/limits` - ضوابط الحصص/المعدلات
-- `/dashboard/cli-tools` - إعداد واجهة سطر مودم، والكشف عن وقت التشغيل، ويشمل ذلك
-- `/dashboard/agents` — تم ابتكار عملاء ACP + تسجيل عميل مخصص
-- `/dashboard/media` — ساحة لعب الصور/الفيديو/الموسيقى
-- `/dashboard/search-tools` - اختبار خريطة البحث
-- `/dashboard/health` - وقت التشغيل، قواطع الدائرة، حدود المعدل
-- `/dashboard/logs` - سجلات الطلب/الوكيل/التدقيق/وحدة التحكم
-- `/dashboard/settings` - علامات إعدادات النظام (عامة، توجيه، إعدادات التحرير والإعدادات البرمجية، إلخ.)
-- `/dashboard/api-manager` - دورة حياة مفتاح برمجة برمجة التطبيقات والأذونات النموذجية## سياق النظام عالي المستوى```mermaid
- flowchart LR
- subgraph Clients[Developer Clients]
- C1[Claude Code]
- C2[Codex CLI]
- C3[OpenClaw / Droid / Cline / Continue / Roo]
- C4[Custom OpenAI-compatible clients]
- BROWSER[Browser Dashboard]
- end
+## Scope and Boundaries
- subgraph Router[OmniRoute Local Process]
- API[V1 Compatibility API\n/v1/*]
- DASH[Dashboard + Management API\n/api/*]
- CORE[SSE + Translation Core\nopen-sse + src/sse]
- DB[(storage.sqlite)]
- UDB[(usage tables + log artifacts)]
- end
+### In Scope
- subgraph Upstreams[Upstream Providers]
- P1[OAuth Providers\nClaude/Codex/Gemini/Qwen/Qoder/GitHub/Kiro/Cursor/Antigravity]
- P2[API Key Providers\nOpenAI/Anthropic/OpenRouter/GLM/Kimi/MiniMax\nDeepSeek/Groq/xAI/Mistral/Perplexity\nTogether/Fireworks/Cerebras/Cohere/NVIDIA]
- P3[Compatible Nodes\nOpenAI-compatible / Anthropic-compatible]
- end
+- Local gateway runtime
+- Dashboard management APIs
+- Provider authentication and token refresh
+- Request translation and SSE streaming
+- Local state + usage persistence
+- Optional cloud sync orchestration
- subgraph Cloud[Optional Cloud Sync]
- CLOUD[Cloud Sync Endpoint\nNEXT_PUBLIC_CLOUD_URL]
- end
+### Out of Scope
- C1 --> API
- C2 --> API
- C3 --> API
- C4 --> API
- BROWSER --> DASH
+- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL`
+- Provider SLA/control plane outside local process
+- External CLI binaries themselves (Claude CLI, Codex CLI, etc.)
- API --> CORE
- DASH --> DB
- CORE --> DB
- CORE --> UDB
+## Dashboard Surface (Current)
- CORE --> P1
- CORE --> P2
- CORE --> P3
+Main pages under `src/app/(dashboard)/dashboard/`:
- DASH --> CLOUD
+- `/dashboard` — quick start + provider overview
+- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs
+- `/dashboard/providers` — provider connections and credentials
+- `/dashboard/combos` — combo strategies, templates, model routing rules
+- `/dashboard/costs` — cost aggregation and pricing visibility
+- `/dashboard/analytics` — usage analytics and evaluations
+- `/dashboard/limits` — quota/rate controls
+- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation
+- `/dashboard/agents` — detected ACP agents + custom agent registration
+- `/dashboard/media` — image/video/music playground
+- `/dashboard/search-tools` — search provider testing and history
+- `/dashboard/health` — uptime, circuit breakers, rate limits
+- `/dashboard/logs` — request/proxy/audit/console logs
+- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.)
+- `/dashboard/api-manager` — API key lifecycle and model permissions
-````
+## High-Level System Context
+
+```mermaid
+flowchart LR
+ subgraph Clients[Developer Clients]
+ C1[Claude Code]
+ C2[Codex CLI]
+ C3[OpenClaw / Droid / Cline / Continue / Roo]
+ C4[Custom OpenAI-compatible clients]
+ BROWSER[Browser Dashboard]
+ end
+
+ subgraph Router[OmniRoute Local Process]
+ API[V1 Compatibility API\n/v1/*]
+ DASH[Dashboard + Management API\n/api/*]
+ CORE[SSE + Translation Core\nopen-sse + src/sse]
+ DB[(storage.sqlite)]
+ UDB[(usage tables + log artifacts)]
+ end
+
+ subgraph Upstreams[Upstream Providers]
+ P1[OAuth Providers\nClaude/Codex/Gemini/Qwen/Qoder/GitHub/Kiro/Cursor/Antigravity]
+ P2[API Key Providers\nOpenAI/Anthropic/OpenRouter/GLM/Kimi/MiniMax\nDeepSeek/Groq/xAI/Mistral/Perplexity\nTogether/Fireworks/Cerebras/Cohere/NVIDIA]
+ P3[Compatible Nodes\nOpenAI-compatible / Anthropic-compatible]
+ end
+
+ subgraph Cloud[Optional Cloud Sync]
+ CLOUD[Cloud Sync Endpoint\nNEXT_PUBLIC_CLOUD_URL]
+ end
+
+ C1 --> API
+ C2 --> API
+ C3 --> API
+ C4 --> API
+ BROWSER --> DASH
+
+ API --> CORE
+ DASH --> DB
+ CORE --> DB
+ CORE --> UDB
+
+ CORE --> P1
+ CORE --> P2
+ CORE --> P3
+
+ DASH --> CLOUD
+```
## Core Runtime Components
## 1) API and Routing Layer (Next.js App Routes)
-الدلائل الرئيسية:
+Main directories:
-- `src/app/api/v1/*` و `src/app/api/v1beta/*` لواجهات برمجة التطبيقات المتوافقة
-- `src/app/api/*` لواجهات برمجة تطبيقات للإدارة/التكوين
-- إعادة الكتابة التالية في الخريطة `next.config.mjs` `/v1/*` إلى `/api/v1/*`
+- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs
+- `src/app/api/*` for management/configuration APIs
+- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*`
-طرق التوافق:
+Important compatibility routes:
- `src/app/api/v1/chat/completions/route.ts`
- `src/app/api/v1/messages/route.ts`
- `src/app/api/v1/responses/route.ts`
-- `src/app/api/v1/models/route.ts` - تشمل نماذج مخصصة ذات `مخصصة: صحيح`
-- `src/app/api/v1/embeddings/route.ts` - إنشاء التضمين (6 مفري)
-- `src/app/api/v1/images/ Generations/route.ts` - إنشاء الصور (4+ موفري خدمات بما في ذلك Antigravity/Nebius)
+- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true`
+- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers)
+- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius)
- `src/app/api/v1/messages/count_tokens/route.ts`
-- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` - دردشة مخصصة لكل المرشحين
-- `src/app/api/v1/providers/[provider]/embeddings/route.ts` - عمليات التضمين المخصصة لكل المجالات
-- `src/app/api/v1/providers/[provider]/images/ Generations/route.ts` - صور مخصصة لكل إطار
+- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat
+- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings
+- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images
- `src/app/api/v1beta/models/route.ts`
- `src/app/api/v1beta/models/[...path]/route.ts`
-الفترات الإدارية:
+Management domains:
-- المصادقة/الإعدادات: `src/app/api/auth/*`، `src/app/api/settings/*`
-- مقدمو الخدمة/الاتصالات: `src/app/api/providers*`
-- عقد الموفر: `src/app/api/provider-nodes*`
-- الروابط ذات الصلة: `src/app/api/provider-models` (GET/POST/DELETE)
-- كتالوج الارتباطات: `src/app/api/models/route.ts` (GET)
-- الوكيل التنفيذي: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
+- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*`
+- Providers/connections: `src/app/api/providers*`
+- Provider nodes: `src/app/api/provider-nodes*`
+- Custom models: `src/app/api/provider-models` (GET/POST/DELETE)
+- Model catalog: `src/app/api/models/route.ts` (GET)
+- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
- OAuth: `src/app/api/oauth/*`
--لوحة المفاتيح/الأسماء المستعارة/المجموعات/التسعير: `src/app/api/keys*`، `src/app/api/models/alias`، `src/app/api/combos*`، `src/app/api/pricing`
--استخدام: `src/app/api/usage/*`
-- الناقلات/السحابة: `src/app/api/sync/*`، `src/app/api/cloud/*`
-- مساعدي أدوات CLI: `src/app/api/cli-tools/*`
-- مرشح IP: `src/app/api/settings/ip-filter` (GET/PUT)
-- تكلفة التفكير: `src/app/api/settings/thinking-budget` (GET/PUT)
-- متشوق النظام: `src/app/api/settings/system-prompt` (GET/PUT)
-- الجلسات: `src/app/api/sessions` (GET)
-- النطاق المعدل: `src/app/api/rate-limits` (GET)
-- معطف: `src/app/api/resilience` (GET/PATCH) - ملفات تعريف الموفر، التفاضل والتكامل، حالة لا يمكن تعديلها
-- إعادة ضبط ضبط: `src/app/api/resilience/reset` (POST) - إعادة ضبط القواطع + تخفيف التهدئة
-- إحصائيات ذاكرة تخزين مؤقتة: `src/app/api/cache/stats` (GET/DELETE)
-- توفر النموذج: `src/app/api/models/availability` (GET/POST)
-- القياس عن بعد: `src/app/api/telemetry/summary` (GET)
-- الميزانية: `src/app/api/usage/budget` (GET/POST)
-- السلاسل الاحتياطية: `src/app/api/fallback/chains` (GET/POST/DELETE)
-- تدقيق تماما: `src/app/api/compliance/audit-log` (GET)
-- التقييمات: `src/app/api/evals` (GET/POST)، `src/app/api/evals/[suiteId]` (GET)
-- للمزيد: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core
+- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing`
+- Usage: `src/app/api/usage/*`
+- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*`
+- CLI tooling helpers: `src/app/api/cli-tools/*`
+- IP filter: `src/app/api/settings/ip-filter` (GET/PUT)
+- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT)
+- System prompt: `src/app/api/settings/system-prompt` (GET/PUT)
+- Sessions: `src/app/api/sessions` (GET)
+- Rate limits: `src/app/api/rate-limits` (GET)
+- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state
+- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns
+- Cache stats: `src/app/api/cache/stats` (GET/DELETE)
+- Model availability: `src/app/api/models/availability` (GET/POST)
+- Telemetry: `src/app/api/telemetry/summary` (GET)
+- Budget: `src/app/api/usage/budget` (GET/POST)
+- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE)
+- Compliance audit: `src/app/api/compliance/audit-log` (GET)
+- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET)
+- Policies: `src/app/api/policies` (GET/POST)
-وحدات السرعة الرئيسية:- الإدخال: `src/sse/handlers/chat.ts`
-- أريد الأساسي: `open-sse/handlers/chatCore.ts`
-- محولات تنفيذ الموفر: `open-sse/executors/*`
-- الاكتشاف الجديد/تكوين الموفر: `open-sse/services/provider.ts`
-- تحليل/حل النموذج: `src/sse/services/model.ts`، `open-sse/services/model.ts`
-- الحساب الاحتياطي للحساب: `open-sse/services/accountFallback.ts`
-- سجل الترجمة: `open-sse/translator/index.ts`
-- تحويلات الدفق: `open-sse/utils/stream.ts`، `open-sse/utils/streamHandler.ts`
--الطلب/تطبيع الاستخدام: `open-sse/utils/usageTracking.ts`
-- فكر في محلل العناوين: `open-sse/utils/thinkTagParser.ts`
--معالج التضمين: `open-sse/handlers/embeddings.ts`
-- سجل موفر التضمين: open-sse/config/embeddingRegistry.ts
--معالج إنشاء الصور: `open-sse/handlers/imageGeneration.ts`
-- سجل موفر الصور: `open-sse/config/imageRegistry.ts`
-- تعريف القيمة: `open-sse/handlers/responseSanitizer.ts`
-- تطبيع الدور: `open-sse/services/roleNormalizer.ts`
+## 2) SSE + Translation Core
-الخدمات (منطقة الأعمال):
+Main flow modules:
-- اختيار الحساب/تسجيل النقاط: `open-sse/services/accountSelector.ts`
-- إدارة دورة حياة السياق: `open-sse/services/contextManager.ts`
-- فرض مرشح IP: `open-sse/services/ipFilter.ts`
-- تعقيب النظر: `open-sse/services/sessionManager.ts`
--طلب إلغاء البيانات المكررة: `open-sse/services/signatureCache.ts`
-- البناء الكامل: `open-sse/services/systemPrompt.ts`
-- التفكير في إدارة الميزانية: `open-sse/services/thinkingBudget.ts`
-- توجيه نموذج حرف البدل: `open-sse/services/wildcardRouter.ts`
-- إدارة إلى حد التعديل: `open-sse/services/rateLimitManager.ts`
-- قاطع الدائرة: `open-sse/services/circuitBreaker.ts`
+- Entry: `src/sse/handlers/chat.ts`
+- Core orchestration: `open-sse/handlers/chatCore.ts`
+- Provider execution adapters: `open-sse/executors/*`
+- Format detection/provider config: `open-sse/services/provider.ts`
+- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts`
+- Account fallback logic: `open-sse/services/accountFallback.ts`
+- Translation registry: `open-sse/translator/index.ts`
+- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
+- Usage extraction/normalization: `open-sse/utils/usageTracking.ts`
+- Think tag parser: `open-sse/utils/thinkTagParser.ts`
+- Embedding handler: `open-sse/handlers/embeddings.ts`
+- Embedding provider registry: `open-sse/config/embeddingRegistry.ts`
+- Image generation handler: `open-sse/handlers/imageGeneration.ts`
+- Image provider registry: `open-sse/config/imageRegistry.ts`
+- Response sanitization: `open-sse/handlers/responseSanitizer.ts`
+- Role normalization: `open-sse/services/roleNormalizer.ts`
-وحدات المجال:
+Services (business logic):
-- توفر النموذج: `src/lib/domain/modelAvailability.ts`
-- متطلبات/ميزانيات التكلفة: `src/lib/domain/costRules.ts`
-- السياسة الافتراضية: `src/lib/domain/fallbackPolicy.ts`
-- محلل التحرير والسرد: `src/lib/domain/comboResolver.ts`
-- تأمين التأمين: `src/lib/domain/lockoutPolicy.ts`
-- محرك السياسة: `src/domain/policyEngine.ts` - القفل المركزي ← الميزانية ← التقييم الاحتياطي
-- كتالوج الرموز لسبب: `src/lib/domain/errorCodes.ts`
-- معرف الطلب: `src/lib/domain/requestId.ts`
-- مهلة الجلب: `src/lib/domain/fetchTimeout.ts`
--طلب القياس عن بعد: `src/lib/domain/requestTelemetry.ts`
-- شامل/الدقيق: `src/lib/domain/compliance/index.ts`
-- عداء التقييم: `src/lib/domain/evalRunner.ts`
-- دونية المجال المجال: `src/lib/db/domainState.ts` - SQLite CRUD للسلاسل الاحتياطية، والميزانيات، خسر التكلفة، وحالة القفل، وقواطع الضوء
+- Account selection/scoring: `open-sse/services/accountSelector.ts`
+- Context lifecycle management: `open-sse/services/contextManager.ts`
+- IP filter enforcement: `open-sse/services/ipFilter.ts`
+- Session tracking: `open-sse/services/sessionManager.ts`
+- Request deduplication: `open-sse/services/signatureCache.ts`
+- System prompt injection: `open-sse/services/systemPrompt.ts`
+- Thinking budget management: `open-sse/services/thinkingBudget.ts`
+- Wildcard model routing: `open-sse/services/wildcardRouter.ts`
+- Rate limit management: `open-sse/services/rateLimitManager.ts`
+- Circuit breaker: `open-sse/services/circuitBreaker.ts`
+- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy
+- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions
-وحدات موفر OAuth (12 ملفًا فرديًا ضمن `src/lib/oauth/providers/`):
+Domain layer modules:
-- فهرس التسجيل: `src/lib/oauth/providers/index.ts`
-- مقدمو الخدمات الأشخاص: `claude.ts`، `codex.ts`، `gemini.ts`، `antigravity.ts`، `qode.ts`، `qwen.ts`، `kimi-coding.ts`، `github.ts`، `kiro.ts`، `cursor.ts`، `kilocode.ts`، `cline.ts`
-- طعام السباحة: `src/lib/oauth/providers.ts` - يُعاد تصديره من العناصر العناصر## 3) طبقة الثبات
+- Model availability: `src/lib/domain/modelAvailability.ts`
+- Cost rules/budgets: `src/lib/domain/costRules.ts`
+- Fallback policy: `src/lib/domain/fallbackPolicy.ts`
+- Combo resolver: `src/lib/domain/comboResolver.ts`
+- Lockout policy: `src/lib/domain/lockoutPolicy.ts`
+- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation
+- Error codes catalog: `src/lib/domain/errorCodes.ts`
+- Request ID: `src/lib/domain/requestId.ts`
+- Fetch timeout: `src/lib/domain/fetchTimeout.ts`
+- Request telemetry: `src/lib/domain/requestTelemetry.ts`
+- Compliance/audit: `src/lib/domain/compliance/index.ts`
+- Eval runner: `src/lib/domain/evalRunner.ts`
+- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers
-قاعدة بيانات الحالة الأساسية (SQLite):- المعرفة البشرية الأساسية: `src/lib/db/core.ts` (better-sqlite3، migrations، WAL)
-- واجهة إعادة التصدير: `src/lib/localDb.ts` (طبقة توافق مختلفة للمتصلين)
-- الملف: `${DATA_DIR}/storage.sqlite` (أو `$XDG_CONFIG_HOME/omniroute/storage.sqlite` عند الضرورة، وإلا `~/.omniroute/storage.sqlite`)
-- كيانات (الجداول + أسماء KV): ProvideConnections، وproviderNodes، وmodelAliases، والمجموعات، WapiKeys، والإعدادات، والتسعير،**customModels**،**proxyConfig**،**ipFilter**،**thinkingBudget**،**systemPrompt**
+OAuth provider modules (12 individual files under `src/lib/oauth/providers/`):
-بمرور الوقت الاستخدام:
+- Registry index: `src/lib/oauth/providers/index.ts`
+- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
+- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules
-- الواجهة: `src/lib/usageDb.ts` (وحدات متحللة في `src/lib/usage/*`)
-- جداول SQLite في `storage.sqlite`: `usage_history`، `call_logs`، `proxy_logs`
-- تبرز عناصر الملف الاختياري للتوافق/تصحيح سبب (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `
/logs/...`)
-- يتم رحيل ملفات JSON القديمة إلى SQLite عن طريق عمليات رحيل بدء التشغيل عند وجودها
+## 3) Persistence Layer
-قاعدة بيانات المجال (SQLite):
+Primary state DB (SQLite):
-- `src/lib/db/domainState.ts` - عمليات إنتاج CRUD لحالة المجال
-- الجداول (التي تم تحديدها في `src/lib/db/core.ts`): `domain_fallback_chains`، `domain_budgets`، `domain_cost_history`، `domain_lockout_state`، `domain_circuit_breakers`.
-- نمط ذاكرة التخزين المؤقت للكتابة: قرص الاتصال موجود في الذاكرة الموثوقة في وقت التشغيل؛ تتم كتابة الطفرات بشكل متزامن إلى SQLite؛ يتم استعادة حالة قاعدة البيانات عند البداية الباردة ## 4) المصادقة + الأسطح الأمنية
+- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL)
+- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers)
+- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`)
+- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt**
-- مصادقة ملف تعريف الارتباط في لوحة المعلومات: `src/proxy.ts`، `src/app/api/auth/login/route.ts`
-- إنشاء/التحقق من مفتاح واجهة برمجة التطبيقات: `src/shared/utils/apiKey.ts`
--أسرار الموفر في الخطوط "providerConnections".
-- دعم خارجي تمامًا عبر `open-sse/utils/proxyFetch.ts` (env vars) و`open-sse/utils/networkProxy.ts` (قابل للتكوين لكل المرشحين أو عالمي)## 5) Cloud Sync
+Usage persistence:
-- جدولة init: `src/lib/initCloudSync.ts`، `src/shared/services/initializeCloudSync.ts`، `src/shared/services/modelSyncScheduler.ts`
-- أهم الأحداث: `src/shared/services/cloudSyncScheduler.ts`
-- أهم الأحداث: `src/shared/services/modelSyncScheduler.ts`
-- التحكم في المسار: `src/app/api/sync/cloud/route.ts`## دورة حياة الطلب (`/v1/chat/completions`)```mermaid
+- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`)
+- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs`
+- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`)
+- legacy JSON files are migrated to SQLite by startup migrations when present
+
+Domain State DB (SQLite):
+
+- `src/lib/db/domainState.ts` — CRUD operations for domain state
+- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
+- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start
+
+## 4) Auth + Security Surfaces
+
+- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts`
+- API key generation/verification: `src/shared/utils/apiKey.ts`
+- Provider secrets persisted in `providerConnections` entries
+- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global)
+
+## 5) Cloud Sync
+
+- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
+- Periodic task: `src/shared/services/cloudSyncScheduler.ts`
+- Periodic task: `src/shared/services/modelSyncScheduler.ts`
+- Control route: `src/app/api/sync/cloud/route.ts`
+
+## Request Lifecycle (`/v1/chat/completions`)
+
+```mermaid
sequenceDiagram
autonumber
participant Client as CLI/SDK Client
@@ -297,7 +331,7 @@ sequenceDiagram
Stream-->>Client: SSE chunks / JSON response
Stream->>Usage: extract usage + persist history/log
-````
+```
## Combo + Account Fallback Flow
@@ -329,15 +363,19 @@ flowchart TD
Q -- No --> R[Return all unavailable]
```
-يتم اتخاذ القرار الاحتياطي بواسطة `open-sse/services/accountFallback.ts` باستخدام رموز الحالة للاستدلال على رسائل الخطأ. تسهيل توجيه التشغيل والتنسيق بين الطرفين طوعًا للمساعدة في تقديم الطلبات: يتم التعامل مع 400s على نطاق الموفر مثل كتلة المحتوى الأول وفشل التحقق من صحة الدور على أنها فشل رئيسي للنموذج، لذا لا يزال لا يزال مطلوبًا التحرير والسرد التالي.## OAuth Onboarding and Token Refresh Lifecycle```mermaid
+Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run.
+
+## OAuth Onboarding and Token Refresh Lifecycle
+
+```mermaid
sequenceDiagram
-autonumber
-participant UI as Dashboard UI
-participant OAuth as /api/oauth/[provider]/[action]
-participant ProvAuth as Provider Auth Server
-participant DB as localDb
-participant Test as /api/providers/[id]/test
-participant Exec as Provider Executor
+ autonumber
+ participant UI as Dashboard UI
+ participant OAuth as /api/oauth/[provider]/[action]
+ participant ProvAuth as Provider Auth Server
+ participant DB as localDb
+ participant Test as /api/providers/[id]/test
+ participant Exec as Provider Executor
UI->>OAuth: GET authorize or device-code
OAuth->>ProvAuth: create auth/device flow
@@ -355,10 +393,13 @@ participant Exec as Provider Executor
Exec-->>Test: valid or refreshed token info
Test->>DB: update status/tokens/errors
Test-->>UI: validation result
+```
-````
+Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`.
-يتم تنفيذ التحديث أثناء حركة التحرير المباشر داخل `open-sse/handlers/chatCore.ts` عبر المنفذ `refreshCredentials()`.## دورة حياة المزامنة السحابية (تمكين / مزامنة / تعطيل)```mermaid
+## Cloud Sync Lifecycle (Enable / Sync / Disable)
+
+```mermaid
sequenceDiagram
autonumber
participant UI as Endpoint Page UI
@@ -386,13 +427,17 @@ sequenceDiagram
Sync->>Cloud: DELETE /sync/{machineId}
Sync->>Claude: switch ANTHROPIC_BASE_URL back to local (if needed)
Sync-->>UI: disabled
-````
+```
-يتم تشغيل الدورية بواسطة "CloudSyncScheduler" عند السحابة.## نموذج البيانات وخريطة التخزين```mermaid
+Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled.
+
+## Data Model and Storage Map
+
+```mermaid
erDiagram
-SETTINGS ||--o{ PROVIDER_CONNECTION : controls
-PROVIDER_NODE ||--o{ PROVIDER_CONNECTION : backs_compatible_provider
-PROVIDER_CONNECTION ||--o{ USAGE_ENTRY : emits_usage
+ SETTINGS ||--o{ PROVIDER_CONNECTION : controls
+ PROVIDER_NODE ||--o{ PROVIDER_CONNECTION : backs_compatible_provider
+ PROVIDER_CONNECTION ||--o{ USAGE_ENTRY : emits_usage
SETTINGS {
boolean cloudEnabled
@@ -485,15 +530,18 @@ PROVIDER_CONNECTION ||--o{ USAGE_ENTRY : emits_usage
string prompt
string position
}
+```
-````
+Physical storage files:
-ملفات الوضع المالي:
+- primary runtime DB: `${DATA_DIR}/storage.sqlite`
+- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact)
+- structured call payload archives: `${DATA_DIR}/call_logs/`
+- optional translator/request debug sessions: `/logs/...`
-- قاعدة بيانات وقت التشغيل الأساسي: `${DATA_DIR}/storage.sqlite`
-- أسطر سجل الطلب: `${DATA_DIR}/log.txt` (أداة متوافقة/تصحيح سبب)
-- أرشيفات استضافة المؤتمرات التنظيمية: `${DATA_DIR}/call_logs/`
-- مجموعات تصحيح الأخطاء المترجم/الطلب الاختيارية: `/logs/...`## Deployment Topology```mermaid
+## Deployment Topology
+
+```mermaid
flowchart LR
subgraph LocalHost[Developer Host]
CLI[CLI Tools]
@@ -520,200 +568,255 @@ flowchart LR
Core --> UsageDB
Core --> Providers
Next --> SyncCloud
-````
+```
## Module Mapping (Decision-Critical)
### Route and API Modules
-- `src/app/api/v1/*`، `src/app/api/v1beta/*`: واجهات برمجة التطبيقات المتوافقة
-- `src/app/api/v1/providers/[provider]/*`: مسارات مخصصة لكل دليل (الدردشة والتضمينات والصور)
-- `src/app/api/providers*`: موفر CRUD، التحقق من الصحة، الاختبار
-- `src/app/api/provider-nodes*`: إدارة العقد المتوافقة المخصصة
-- `src/app/api/provider-models`: إدارة الارتباطات المخصصة (CRUD)
-- `src/app/api/models/route.ts`: برمجة تطبيقات كتالوج الارتباطات (الأسماء المستعارة + الارتباطات البديلة)
-- `src/app/api/oauth/*`: تدفقات رمز OAuth/الجهاز
-- `src/app/api/keys*`: دورة حياة مفتاح برمجة التطبيقات المحلية
-- `src/app/api/models/alias`: إدارة الأسماء المستعارة
-- `src/app/api/combos*`: إدارة التحرير والسرد الاحتياطي
-- `src/app/api/pricing`: تجاوزات التسعير لحساب التكلفة
-- `src/app/api/settings/proxy`: الصارم المعتمد (GET/PUT/DELETE)
-- `src/app/api/settings/proxy/test`: اختبار تشغيل الوكيل (POST)
-- `src/app/api/usage/*`: واجهات برمجة تطبيقات الاستخدام والسجلات
-- `src/app/api/sync/*` + `src/app/api/cloud/*`: نوبات السحابية والمساعدون الذين يتحملون السحابة
-- `src/app/api/cli-tools/*`: كاتب/أداة الدما لتكوين CLI المحلي
-- `src/app/api/settings/ip-filter`: قائمة IP مخصصة لها/القائمة المبتكرة (GET/PUT)
-- `src/app/api/settings/thinking-budget`: الاختيار المناسب رمز التفكير (GET/PUT)
-- `src/app/api/settings/system-prompt`: موجه النظام العام (GET/PUT)
-- `src/app/api/sessions`: قائمة العناصر العضوية (GET)
-- `src/app/api/rate-limits`: حالة لا يمكن تعديلها لكل حساب (GET)### التوجيه والتنفيذ الأساسي
+- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs
+- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images)
+- `src/app/api/providers*`: provider CRUD, validation, testing
+- `src/app/api/provider-nodes*`: custom compatible node management
+- `src/app/api/provider-models`: custom model management (CRUD)
+- `src/app/api/models/route.ts`: model catalog API (aliases + custom models)
+- `src/app/api/oauth/*`: OAuth/device-code flows
+- `src/app/api/keys*`: local API key lifecycle
+- `src/app/api/models/alias`: alias management
+- `src/app/api/combos*`: fallback combo management
+- `src/app/api/pricing`: pricing overrides for cost calculation
+- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE)
+- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST)
+- `src/app/api/usage/*`: usage and logs APIs
+- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers
+- `src/app/api/cli-tools/*`: local CLI config writers/checkers
+- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT)
+- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT)
+- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT)
+- `src/app/api/sessions`: active session listing (GET)
+- `src/app/api/rate-limits`: per-account rate limit status (GET)
-- `src/sse/handlers/chat.ts`: تحليل الطلب، ومعالجة التحرير والسرد، حلقة الحساب
-- `open-sse/handlers/chatCore.ts`: الترجمة، المنفذ، إعادة المحاولة/التحديث، إعداد الدفق
-- `open-sse/executors/*`: التحكم الشبكة والتنسيق الخاص بالموفر### سجل الترجمة ومحولات التنسيق
+### Routing and Execution Core
-- `open-sse/translator/index.ts`: تسجيل المترجم وتنسيقه
- -طلب المترجمين: `open-sse/translator/request/*`
-- مترجمو المصدر: `open-sse/translator/response/*`
-- ثوابت عادة: `open-sse/translator/formats.ts`### Persistence
+- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop
+- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup
+- `open-sse/executors/*`: provider-specific network and format behavior
-- `src/lib/db/*`: تفعيل/الحالة الفعالة واستمرارية المجال على SQLite
-- `src/lib/localDb.ts`: إعادة تصدير التوافق لوحدات قاعدة البيانات
-- `src/lib/usageDb.ts`: واجهة سجل/سجلات استخدامات المكالمات أعلى جداول SQLite## Provider Executor Coverage (Strategy Pattern)
+### Translation Registry and Format Converters
-| يحتوي على كل موفر على منفذ تنفيذي متخصص لعدة `BaseExecutor` (في `open-sse/executors/base.ts`)، والذي يوفر بيانات إنشاء عنوان URL، ولكنه، جاهز المحاولة مع الأسيي، ومآثر تحديث الاعتماد، وطريقة استمرار `execute()`. | المنفذ | المزود (المقدمون) | التعامل الخاص |
-| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------- | ------------- |
-| `المنفذ الافتراضي` | أوبن إيه آي، كلود، جيميني، كوين، كيودر، أوبن روتر، جي إل إم، كيمي، ميني ماكس، ديب سيك، جروك، إكس آي آي، ميسترال، بيربليكسيتي، توغا، فاير ووركس، سيريبراس، كوهير، نفيديا | الاختيارية عنوان URL/الرأس الكيميائي لكل |
-| `منفذ مضاد للجغرافيا` | جوجل مكافحة الجاذبية | معرفات المشروع/الجلسة المخصصة، إعادة المحاولة بعد التحليل |
-| `منفذ الكودكس` | OpenAI Codex | يحقن تعليمات النظام، ويفرض جهدًا منطقيًا |
-| `منفذ مفصل` | بيئة تطوير متكاملة للمؤشر | البروتوكول ConnectRPC، ترجمة Protobuf، طلب التوقيع عبر الفصول الاختباري |
-| `GithubExecutor` | جيثب مساعد الطيار | تحديث الرمز المميز لـ Copilot، ورؤوس محاكاة VSCode |
-| `KiroExecutor` | AWS CodeWhisperer/كيرو | يتغير الثنائي لـ AWS EventStream → تحويل SSE |
-| `الجوزاءCLIEExecutor` | الجوزاء CLI | دورة تحديث رمز OAuth المميز لـ Google |
+- `open-sse/translator/index.ts`: translator registry and orchestration
+- Request translators: `open-sse/translator/request/*`
+- Response translators: `open-sse/translator/response/*`
+- Format constants: `open-sse/translator/formats.ts`
-| يستخدم جميع الموفرين الآخرين (بما في ذلك العقد المتوافق المخصص) "DefaultExecutor".## مصفوفة توافق الموفرين | مقدم | التنسيق | مصادقة | تيار مستمر | غير دفق | تحديث الرمز المميز | برمجة تطبيقات الاستخدام |
-| ---------------------------------------------------------------------------------------------------------- | ---------------- | -------------------------------------- | --------------- | ---------- | ------- | ----------------------- | ----------------------- |
-| كلود | كلود | واجهة برمجة التطبيقات الرئيسية / OAuth | ✅ | ✅ | ✅ | ⚠️ المشرف فقط |
-| الجوزاء | الجوزاء | واجهة برمجة التطبيقات الرئيسية / OAuth | ✅ | ✅ | ✅ | ⚠️ وحدة التحكم السحابية |
-| الجوزاء CLI | الجوزاء-cli | أووث | ✅ | ✅ | ✅ | ⚠️ وحدة التحكم السحابية |
-| مكافحة الجاذبية | ضد الجاذبية | أووث | ✅ | ✅ | ✅ | ✅ الحصة الكاملة API |
-| أوبن آي | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ |
-| الدستور الغذائي | openai-responses | أووث | ✅ مجبور | ❌ | ✅ | ✅الحدود المعدلة |
-| جيثب مساعد الطيار | أوبيناي | OAuth + رمز مساعد الطيار | ✅ | ✅ | ✅ | ✅ لقطات الحصص |
-| | مؤثر | مؤثر الاستطلاع المفضل | ✅ | ✅ | ❌ | ❌ |
-| كيرو | كيرو | AWS SSO OIDC | ✅(ايفنت ستريم) | ❌ | ✅ | ✅ حدود الاستخدام |
-| كوين | أوبيناي | أووث | ✅ | ✅ | ✅ | ⚠️ طلب حسب الطلب |
-| قدير | أوبيناي | OAuth (أساسي) | ✅ | ✅ | ✅ | ⚠️ طلب حسب الطلب |
-| اوبن راوتر | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ |
-| جي إل إم/كيمي/ميني ماكس | كلود | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ |
-| ديب سيك | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ |
-| جروك | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ |
-| xAI (جروك) | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ |
-| ميسترال | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ |
-| الحيرة | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ |
-| منظمة العفو الدولية | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ |
-| منظمة العفو الدولية للعبة | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ |
-| الشيخ | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ |
-| كوهير | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ |
-| نفيديا نيم | أوبيناي | مفتاح واجهة برمجة التطبيقات | ✅ | ✅ | ❌ | ❌ | ## تنسيق تغطية الترجمة |
+### Persistence
-تتضمن التنسيقات المصدر المكتشفة ما يلي:
+- `src/lib/db/*`: persistent config/state and domain persistence on SQLite
+- `src/lib/localDb.ts`: compatibility re-export for DB modules
+- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables
-- `أوبيني`
-- `الردود المفتوحة`
-- "كلود".
-- "الجوزاء".
+## Provider Executor Coverage (Strategy Pattern)
-تتضمن الواردات التفصيلية ما يلي:
+Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method.
-- دردشة/ردود OpenAI
-- كلود
- -الجوزاء/الجوزاء-CLI/الظرف للجاذبية
-- كيرو
-- مرض
+| Executor | Provider(s) | Special Handling |
+| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- |
+| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider |
+| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing |
+| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort |
+| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum |
+| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers |
+| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion |
+| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle |
-استخدم الترجمات**OpenAI كتنسيق مركزي**— جرب جميع التحويلات عبر OpenAI كتنسيق وسيط:`
-تنسيق المصدر → OpenAI (المحور) → التنسيق المستهدف`
+All other providers (including custom compatible nodes) use the `DefaultExecutor`.
-يتم تحديد الترجمات ديناميكيًا استنادًا إلى شكل حمولة المصدر والتنسيق المستهدف للموفر.
+## Provider Compatibility Matrix
-طبقات معالجة إضافية في مسار الترجمة:
+| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API |
+| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ |
+| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only |
+| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API |
+| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits |
+| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots |
+| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ |
+| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits |
+| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request |
+| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request |
+| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ |
+| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ |
--**تطهير الاستجابة**— يزيل الحقول غير القياسية من استجابات تنسيق OpenAI (سواء المتدفقة أو غير المتدفقة) لضمان الامتثال الصارم لـ SDK -**تطبيع الدور**— تحويل `المطور` ← `النظام` للأهداف غير التابعة لـ OpenAI؛ يدمج "النظام" → "المستخدم" للنماذج التي ترفض دور النظام (GLM، ERNIE) -**استخراج علامة التفكير**— يوزع كتل `...` من المحتوى إلى حقل `reasoning_content` -**الإخراج المنظم**— يحول OpenAI `response_format.json_schema` إلى `responseMimeType` + `responseSchema` الخاص بـ Gemini## Supported API Endpoints
+## Format Translation Coverage
-| نقطة النهاية | تنسيق | معالج |
-| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------------- | ----------------- |
-| `POST /v1/chat/completions` | دردشة OpenAI | `src/sse/handlers/chat.ts` |
-| `POST /v1/messages` | رسائل كلود | نفس المعالج (تم اكتشافه تلقائيًا) |
-| `POST /v1/responses` | ردود OpenAI | `open-sse/handlers/responsesHandler.ts` |
-| `POST /v1/embeddings` | تضمينات OpenAI | `open-sse/handlers/embeddings.ts` |
-| `الحصول على /v1/embeddings` | قائمة النماذج | طريق API |
-| `POST /v1/images/أجيال` | صور OpenAI | `open-sse/handlers/imageGeneration.ts` |
-| `الحصول على /v1/images/أجيال` | قائمة النماذج | طريق API |
-| `POST /v1/providers/{provider}/chat/completions` | دردشة OpenAI | مخصص لكل مزود مع التحقق من صحة النموذج |
-| `POST /v1/providers/{provider}/embeddings` | تضمينات OpenAI | مخصص لكل مزود مع التحقق من صحة النموذج |
-| `POST /v1/providers/{provider}/images/generations` | صور OpenAI | مخصص لكل مزود مع التحقق من صحة النموذج |
-| `POST /v1/messages/count_tokens` | عدد كلود توكن | طريق API |
-| `الحصول على /v1/models` | قائمة نماذج OpenAI | مسار واجهة برمجة التطبيقات (الدردشة + التضمين + الصورة + النماذج المخصصة) |
-| `الحصول على /api/models/catalog` | كتالوج | جميع النماذج مجمعة حسب الموفر + النوع |
-| `POST /v1beta/models/*:streamGenerateContent` | مولود برج الجوزاء | طريق API |
-| `الحصول على/PUT/DELETE /api/settings/proxy` | تكوين الوكيل | تكوين وكيل الشبكة |
-| `POST /api/settings/proxy/test` | اتصال الوكيل | نقطة نهاية اختبار صحة الوكيل/الاتصال |
-| `الحصول على/النشر/الحذف /api/provider-models` | نماذج المزود | البيانات الوصفية لنموذج الموفر تدعم النماذج المتاحة المخصصة والمدارة | ## Bypass Handler |
+Detected source formats include:
-يعترض معالج التجاوز (`open-sse/utils/bypassHandler.ts`) طلبات "رمية سريعة" معروفة من Claude CLI - أصوات التمهيد، واستخراج العناوين، وعدد الرموز المميزة - ويعيد**استجابة زائفة**دون استهلاك الرموز المميزة للموفر الرئيسي. يتم تشغيل هذا فقط عندما يحتوي "User-Agent" على "clude-cli".## Request Logger Pipeline
+- `openai`
+- `openai-responses`
+- `claude`
+- `gemini`
-يوفر مسجل الطلب (`open-sse/utils/requestLogger.ts`) مسارًا لتسجيل تصحيح الأخطاء مكون من 7 مراحل، معطل افتراضيًا، وممكن عبر `ENABLE_REQUEST_LOGS=true`:```
+Target formats include:
+
+- OpenAI chat/Responses
+- Claude
+- Gemini/Gemini-CLI/Antigravity envelope
+- Kiro
+- Cursor
+
+Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate:
+
+```
+Source Format → OpenAI (hub) → Target Format
+```
+
+Translations are selected dynamically based on source payload shape and provider target format.
+
+Additional processing layers in the translation pipeline:
+
+- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance
+- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE)
+- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field
+- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema`
+
+## Supported API Endpoints
+
+| Endpoint | Format | Handler |
+| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- |
+| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` |
+| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) |
+| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` |
+| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` |
+| `GET /v1/embeddings` | Model listing | API route |
+| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` |
+| `GET /v1/images/generations` | Model listing | API route |
+| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation |
+| `POST /v1/messages/count_tokens` | Claude Token Count | API route |
+| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) |
+| `GET /api/models/catalog` | Catalog | All models grouped by provider + type |
+| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route |
+| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration |
+| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint |
+| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models |
+
+## Bypass Handler
+
+The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`.
+
+## Request Logger Pipeline
+
+The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`:
+
+```
1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json
→ 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt
-
```
-تتم كتابة الملفات إلى `/logs//` لكل جلسة طلب.## أوضاع الفشل والمرونة## 1) Account/Provider Availability
+Files are written to `/logs//` for each request session.
-- عبارة عن حساب الموفر عند أخطاء/معدل/مصادقة
-- إرجاع الحساب قبل فشل الطلب
-- نموذج التحرير والسرد الاحتياطي عند استنفاد مسار النموذج/المزود الحالي## 2) Token Expiry
+## Failure Modes and Resilience
-- ملفات التقدم والتحديث مع إعادة محاولة توفير خدمة موثوقة للتحديث
-- 401/403 إعادة المحاولة بعد محاولة التحديث في المسار الأساسي## 3) Stream Safety
+## 1) Account/Provider Availability
-- وحدة تحكم قطع الاتصال بالتيار المستمر
-- دفق الترجمة تدفق مع نهاية الدفق و `[تم]`
-- ترخيص للاستخدام عندما تكون البيانات الوصفية للاستخدام الموفر المفقود## 4) تدهور المزامنة السحابية
+- provider account cooldown on transient/rate/auth errors
+- account fallback before failing request
+- combo model fallback when current model/provider path is exhausted
-- أخطاء الأخطاء ولكن استمر تشغيلها محليًا
-- يحتوي على المجدول على منطقه قادر على إعادة المحاولة، ولكن التنفيذ الدوري يستدعي حاليا متزامنة التفعيل بشكل افتراضي## 5) Data Integrity
+## 2) Token Expiry
-- عمليات ترحيل مخطط SQLite وفواتير الترقية التلقائية عند بدء التشغيل
-- JSON القديم → مسار التوافق ترحيل SQLite## إمكانية المراقبة والإشارات التشغيلية
+- pre-check and refresh with retry for refreshable providers
+- 401/403 retry after refresh attempt in core path
-مصادر معرفة وقت التشغيل:
+## 3) Stream Safety
-- أرشيف وحدة التحكم من `src/sse/utils/logger.ts`
-- مجاميع الاستخدام لكل طلب في SQLite (`usage_history`، `call_logs`، `proxy_logs`)
-- التقاط التفاصيل الصافية الصافية على أربع مراحل في SQLite (`request_detail_logs`) عندما تكون `settings.detailed_logs_enabled=true`
-- سجل حالة الطلب النصي في "log.txt" (اختياري/متوافق)
-- سجلات الطلب/الترجمة المتخصصة الاختيارية ضمن `السجلات/` عندما يكون `ENABLE_REQUEST_LOGS=true`
-- نقاط نهاية استخدام معلومات اللوحة (`/api/usage/*`) لاستهلاك واجهة المستخدم
+- disconnect-aware stream controller
+- translation stream with end-of-stream flush and `[DONE]` handling
+- usage estimation fallback when provider usage metadata is missing
-يقوم بالتقاط تكتيكات متعددة بتخزين ما يصل إلى أربع مراحل من نشاطات JSON لكل ما يستقبل بصرية:
+## 4) Cloud Sync Degradation
-- الطلب الوارد من العميل
-- تم إرسال الطلب المترجم إلى المنبع
-- إعادة بناء الرابط الموفر JSON؛ يتم ضغط الاستجابات المتدفقة إلى الملخص النهائي بالإضافة إلى بيانات تعريف الدفق
--الرد النهائي الذي تم إرجاعه بواسطة OmniRoute؛ يتم تخزين الاستجابات المتدفقة في نفس النموذج الملخص المكون## الحدود الحساسة للأمان
+- sync errors are surfaced but local runtime continues
+- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default
-- يعمل سر JWT (`JWT_SECRET`) على تأمين المصادقة/التوقيع على ملف تعريف الارتباط لجلسة لوحة المعلومات
-- يجب الالتزام بالبراءة الأولية لكلمة المرور (`INITIAL_PASSWORD`) ووافق على الاعتراف بها لأول مرة
-- يعمل سر HMAC لمفتاح API (`API_KEY_SECRET`) على تنسيق تنسيق مفتاح API المحلي الذي تم التعاقد معه
-- تظلل أسرار الموفر (مفاتيح/رموز برمجة التطبيقات) موجودة في قاعدة البيانات الأصلية وحماتها على مستوى نظام الملفات
-- تعتمد نقاط نهاية الهجمات السحابية على مصادقة مفتاح API + دلالات معرف الجهاز## مصفوفة البيئة ووقت التشغيل
+## 5) Data Integrity
-تحريرات البيئة المستخدمة بشكل نشط بواسطة تعليمات الحظر:- التطبيق/المصادقة: `JWT_SECRET`، `INITIAL_PASSWORD`
-- التخزين: `DATA_DIR`
-- العقدة المتوافقة: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
-- تجاوز قاعدة الاختيار الاختيارية (Linux/macOS عند إلغاء تعيين `DATA_DIR`): `XDG_CONFIG_HOME`
-- التجزئة الأمنية: `API_KEY_SECRET`، `MACHINE_ID_SALT`
-- التسجيل: `ENABLE_REQUEST_LOGS`
-- عناوين URL للاستقبال/السحابة: `NEXT_PUBLIC_BASE_URL`، `NEXT_PUBLIC_CLOUD_URL`
-- الوكيل الشامل: `HTTP_PROXY`، `HTTPS_PROXY`، `ALL_PROXY`، `NO_PROXY` ومتغيرات الصغيرة الصغيرة
-- علامات ميزات SOCKS5: `ENABLE_SOCKS5_PROXY`، `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
-- مساعدو النظام الأساسي/وقت التشغيل (وليس تفعيل الخاص بالتطبيق): `APPDATA`، `NODE_ENV`، `PORT`، `HOSTNAME`## الملاحظات المعمارية المعروفة
+- SQLite schema migrations and auto-upgrade hooks at startup
+- legacy JSON → SQLite migration compatibility path
-1. تشارك `usageDb` و`localDb` في نفس الدليل الأساسي (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> ``~/.omniroute`) مع ترحيل الملفات القديمة.
-2. يفوض `/api/v1/route.ts` إلى نفس منشئ الكتالوج الموحد الذي يستخدمه `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) العلم الانحراف الدلالي.
-3. يقوم بطلب تسجيل بكتابة الرؤوس/النص الكامل عند جاكسونه؛ التعامل مع سجل الدليل على أنه حساسية.
-4. يعتمد حماية السحابة على `NEXT_PUBLIC_BASE_URL` صحيح وإمكانية الوصول إلى نقطة نهاية السحابة.
-5. تم نشر الدليل `open-sse/` باسم `@omniroute/open-sse`**حزمة مساحة العمل npm**. يقوم بكود المصدر باستيراده عبر `@omniroute/open-sse/...` (تم حله بواسطة Next.js `transpilePackages`). لا تسلك الطرق المستمرة في هذا المستند استخدم اسم الدليل `open-sse/` للاتساق.
-6. نستخدم الكائنات الموجودة في لوحة المعلومات**Recharts**(المستندة إلى SVG) لتصورات التحليلات التفاعلية التي يمكن الوصول إليها (المخططات الشريطية للاستخدام للنموذج، والجرافيك المستخدمة للمخرجين مع النجاح).
-7.استخدام السيولة E2E**Playwright**(`tests/e2e/`)، ويمكنها عبر `npm run test:e2e`. المستخدمة في الوحدة**Node.js test runner**(`tests/unit/`)، ويمكن تشغيلها عبر `npm run test:unit`. كود المصدر ضمن `src/` هو**TypeScript**(`.ts`/`.tsx`)؛ تختلف مساحة العمل `open-sse/` JavaScript (`.js`).
-8. تم ضبط صفحة الإعدادات في 5 علامات: الأمان، التوجيه (6 إستراتيجيات عالمية: التعبئة العامة، جولة روبن، p2c، تنظيم غير محدد لاستخدامًا، تحسين التكلفة)، اشتراك (حدود الرسوم المتحركة للتحرير، قطع الدقة، إبداع)، الذكاء الاصطناعي (ميزانية التفكير، متشوق للنظام، ذاكرة التخزين المؤقت السريع)، المتقدمة (الوكيل).## قائمة التحقق من التشغيل
+## Observability and Operational Signals
-- البناء من المصدر: ``npm run build``
-- إنشاء صورة Docker: `docker build -t omniroute .`
-- بدء الخدمة والتحقق:
-- `الحصول على /api/settings`
-- `الحصول على /api/v1/models`
-- يجب أن يكون عنوان URL الأساسي لهدف واجهة سطر اللاسلكي هو `http://:20128/v1` عندما يكون `PORT=20128`
-```
+Runtime visibility sources:
+
+- console logs from `src/sse/utils/logger.ts`
+- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`)
+- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true`
+- textual request status log in `log.txt` (optional/compat)
+- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true`
+- dashboard usage endpoints (`/api/usage/*`) for UI consumption
+
+Detailed request payload capture stores up to four JSON payload stages per routed call:
+
+- raw request received from the client
+- translated request actually sent upstream
+- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata
+- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form
+
+## Security-Sensitive Boundaries
+
+- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing
+- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning
+- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format
+- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level
+- Cloud sync endpoints rely on API key auth + machine id semantics
+
+## Environment and Runtime Matrix
+
+Environment variables actively used by code:
+
+- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD`
+- Storage: `DATA_DIR`
+- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
+- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME`
+- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT`
+- Logging: `ENABLE_REQUEST_LOGS`
+- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
+- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants
+- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
+- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`
+
+## Known Architectural Notes
+
+1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration.
+2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift.
+3. Request logger writes full headers/body when enabled; treat log directory as sensitive.
+4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability.
+5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency.
+6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates).
+7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`).
+8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy).
+9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed.
+10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22.
+11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible.
+
+## Operational Verification Checklist
+
+- Build from source: `npm run build`
+- Build Docker image: `docker build -t omniroute .`
+- Start service and verify:
+- `GET /api/settings`
+- `GET /api/v1/models`
+- CLI target base URL should be `http://:20128/v1` when `PORT=20128`
diff --git a/docs/i18n/ar/docs/FEATURES.md b/docs/i18n/ar/docs/FEATURES.md
index b975bc2f06..0db3f5758e 100644
--- a/docs/i18n/ar/docs/FEATURES.md
+++ b/docs/i18n/ar/docs/FEATURES.md
@@ -4,70 +4,168 @@
---
-دليل مرئي لكل قسم من معلومات لوحة OmniRoute.---## 🔌 Providers
-إدارة اتصالات الذكاء الصناعي: موفري OAuth (Claude Code وCodex وGemini CLI) وموفري مفاتيح API (Groq وDeepSeek وOpenRouter) ومقدمي خدمات العيد (Qoder وQwen وKiro). لحسابات كيرو على تتبع الاعتماد الائتماني - الأرصدة النهائية لإجمالي استطلاعات الرأي المتخصصة في لوحة التحكم → استخدام.---
+
+Visual guide to every section of the OmniRoute dashboard.
+
+---
+
+## 🔌 Providers
+
+Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage.
+
+
+
+---
## 🎨 Combos
-أنشئ مجموعات التوجيه باستخدام 6 إستراتيجيات: نأمل، والمتزايدة، والدورية، والعشوائية، وأقل استخدامًا، والمُحسّن من حيث التكلفة. وخاصة مجموعة نماذج متعددة مع اختلافات سريعة وفحوصات للجاهزية.---
+Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks.
+
+
+
+---
## 📊 Analytics
-تحليلات استخدام شاملة مع الرمز المميز، وتقديرات التكلفة، وخرائط، ومخططات التوزيع الأسبوعية، والتفاصيل لكل محمية.---
+Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns.
+
+
+
+---
## 🏥 System Health
-التسجيل في الوقت الفعلي: وقت العمل، والذاكرة، والإصدار، والنسب لزمن الوصول (p50/p95/p99)، وإحصائيات ذاكرة التخزين المؤقتة، وحالات منع دائرة الموفر.---
+Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states.
+
+
+
+---
## 🔧 Translator Playground
-أدوات لتصحيح أخطاء ترجمات برمجة التطبيقات:**ساحة اللعب**(محول أربعة نجاح)،**اختبار الدردشة**(الطلب المباشر)،**منصة الاختبار**(اختبارات الدفعة)، و**المراقب المباشر**(بث الوقت في العمل).---
+Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream).
+
+
+
+---
## 🎮 Model Playground _(v2.0.9+)_
-اختبر أي نموذج مباشرة من لوحة القيادة. حدد الموفر والطراز والنقطة النهائية، وكتب المطالبات باستخدام محرر موناكو، وقم بتفعيل الاستثناءات في المنتج الفعلي، وإلغاء منتصف الدفق، والمعايرة التقليدية مرة.---## 🎨 Themes _(v2.0.5+)_
+Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics.
-ألوان قابلة للتخصيص لمعلومات لوحة المفاتيح بأكملها. اختر من بين 7 ألوان محددة ليمين (مرجاني، أزرق، أخضر، بنفسجي، لون أحمر، سماوي) أو قم باختيار سمة مخصصة عن طريق اختيار أي سداسي عشري. يدعم وضع الضوء والظلام النظام.---## ⚙️ Settings
+---
-لوحة الإعدادات شاملة مع علامات التبويب:
+## 🎨 Themes _(v2.0.5+)_
--**عام**— تخزين النظام، وإدارة النسخ الاحتياطي (قاعدة بيانات التصدير/الاستيراد) -**المظهر**— محدد السماعة (داكن/فاتح/نظام)، الإعدادات المسبقة لموضوع الألوان والألوان المخصصة، ورؤية السجل الصحي، وعناصر التحكم في رؤية عنصر الشريط الجانبي -**الأمان**— حماية نقطة نهاية واجهة برمجة التطبيقات، وحظر الموفر المخصص، وتصفية IP، ومعلومات الاتصال -**التوجيه**— الأسماء المستعارة للنماذج، و الابتكارات الخلفية -**المرونة**— ونتيجة لذلك الحد الأقصى للمعدل، وضبط القيود، والتعطيل التلقائي للحسابات المحظورة، وانتهاء صلاحية الموفر -**متقدم**— تجاوز، ومسار تدقيق فقط، وتطبيق التدمير الاحتياطي---
+Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode.
+
+---
+
+## ⚙️ Settings
+
+Comprehensive settings panel with tabs:
+
+- **General** — System storage, backup management (export/import database)
+- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls
+- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info
+- **Routing** — Model aliases, background task degradation
+- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration
+- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode
+
+
+
+---
## 🔧 CLI Tools
-ختمة واحدة لأدوات تميز الذكاء الصناعي: Claude Code، وCodex CLI، وGemini CLI، وOpenClaw، وKilo Code، وAntigravity، وCline، وContinue، وCursor، وFactory Droid. تم تفعيل/إعادة ضبط تلقائي، فقط تعريف الاتصال، والنتائج المباشرة.---
+One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping.
+
+
+
+---
## 🤖 CLI Agents _(v2.0.11+)_
-لوحة معلومات للتحكم في وكلاء CLI. تم عرض شبكة مكونة من 14 وكيلًا مدمجًا (Codex وClaude وGoose وGemini CLI وOpenClaw وAider وOpenCode وCline وQwen Code وForgeCode وAmazon Q وOpen Interpreter وCursor CLI وWarp) مع:
+Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with:
--**حالة التثبيت**— تم التثبيت/لم يتم العثور عليه باستخدام اكتشاف الإصدار -**توصيات المذكورة**— stdio، HTTP، وما إلى ذلك. -**الوكلاء يستهدفون**— هل هناك أي أداة لواجهة سطر الوكيل (CLI) عبر النموذج (الاسم، ثنائي، أمر الإصدار، وسيط النشر) -**مطابقة بصمة CLI**— التبديل لكل المرشحين لمطابقة توقيعات طلب CLI الأصلية، مما سيقدر من المبدع بالفعل مع ضمان عنوان IP الوكيل---## 🖼️ Media _(v2.0.3+)_
+- **Installation status** — Installed / Not Found with version detection
+- **Protocol badges** — stdio, HTTP, etc.
+- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args)
+- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP
-موجود في الصور ومقاطع الفيديو والموسيقى من لوحة التحكم. يدعم OpenAI وxAI وTogether وHyperbolic وSD WebUI وComfyUI وAnimateDiff وStable Audio Open وMusicGen.---## 📝 Request Logs
+---
-تسجيل طلبات الإنتاج في الواقع باستخدام التصفية حسب الموفر والطراز والحساب ومفتاح واجهة برمجة التطبيقات. معلمات القيمة الناتجة عن التعويض الطبيعي ووقت التعويض وتفاصيل التعويض.---
+## 🔗 Context Relay _(v3.5.5+)_
+
+A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context.
+
+Configurable via combo-level or global settings:
+- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%)
+- **Max Messages For Summary** — How much recent history to condense
+- **Summary Model** — Optional override model for generating the handoff summary
+
+Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md).
+
+---
+
+## 🛡️ Proxy Hardening _(v3.5.5+)_
+
+Comprehensive proxy configuration enforcement across the entire request pipeline:
+
+- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments
+- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings
+- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22
+- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS
+
+---
+
+## 🖼️ Media _(v2.0.3+)_
+
+Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen.
+
+---
+
+## 📝 Request Logs
+
+Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details.
+
+
+
+---
## 🌐 API Endpoint
-نقطة نهاية واجهة برمجة التطبيقات الموحدة الخاصة بك مع تفاصيل التفاصيل: عمليات التسجيل، وواجهة برمجة تطبيقات الاستجابات، والتضمينات، وأي الصور، إلى الإعداد، والنسخة الصوتية، تحويل النص إلى كلام، والإشراف، ومفاتيح واجهة برمجة التطبيقات المفقودة. تكامل Cloudflare Quick Tunnel للتواصل مع وكيل السحابي للوصول إليه بعد.---
+Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access.
+
+
+
+---
## 🔑 API Key Management
-إنشاء مفاتيح API ونطاقها لربط وإلها. يمكن أن يكون هناك كل المفاتيح الرئيسية على/موفري خدمات محددة لهم حق الوصول الكامل أو أذونات القراءة فقط. إدارة المفاتيح المرئية مع تكرار الاستخدام.---## 📋 Audit Log
+Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking.
-متابعة الإجراءات الإدارية بالتصفية حسب نوع الإجراء والممثل والهدف وعنوان IP والطابع الزمني. سجل الأحداث الأمنية الكاملة.---## 🖥️ Desktop Application
+---
-تطبيق Native Electron لسطح المكتب لأنظمة التشغيل Windows وmacOS وLinux. قم بالموافقة على OmniRoute كتطبيق مستقل مع نظام متكامل للنظام والدعم دون الاتصال والتحديث التلقائي والتثبيت بنقرة واحدة.
+## 📋 Audit Log
-الميزات الرئيسية:
+Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history.
-- استقصاء جاهزية الضيوف (لا توجد شاشة عند التشغيل البارد)
-- نظام إدارة المنافذ
-- اتخاذ القرار بشأن المحتوى
-- مثال واحد
-- التحديث التلقائي عند إعادة التشغيل
-- واجهة المستخدم مشروطة بالكامل (إشارات المرور لنظام التشغيل MacOS، وشريط العنوان الإلكتروني لنظام التشغيل Windows/Linux)
-- بناء الإلكترون المقوى - يتم إبتكار "وحدات_العقدة" وتشهد بالرمز في المقترحات ورفضها قبل قبولها، مما يمنع الاعتماد في وقت التشغيل على البناء (الإصدار 2.5.5+)
+---
-📖 راجع [`electron/README.md`](../electron/README.md) للحصول على التوثيق الكامل.
+## 🖥️ Desktop Application
+
+Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install.
+
+Key features:
+
+- Server readiness polling (no blank screen on cold start)
+- System tray with port management
+- Content Security Policy
+- Single-instance lock
+- Auto-update on restart
+- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar)
+- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+)
+
+📖 See [`electron/README.md`](../electron/README.md) for full documentation.
diff --git a/docs/i18n/ar/docs/TROUBLESHOOTING.md b/docs/i18n/ar/docs/TROUBLESHOOTING.md
index 0d1086b79c..a56950cd98 100644
--- a/docs/i18n/ar/docs/TROUBLESHOOTING.md
+++ b/docs/i18n/ar/docs/TROUBLESHOOTING.md
@@ -4,65 +4,148 @@
---
-المشاكل والحلول الشائعة لـ OmniRoute.---## Quick Fixes
-| مشكلة | الحل |
-| ------------------------------------- | --------------------------------------------------------------------- | --------------------- |
-| تسجيل الدخول الأول لا يعمل | قم بزيارة `INITIAL_PASSWORD` في `.env` (بدون ترميز افتراضي) |
-| بدأت لوحة المعلومات على المنفذ الخاطئ | قم بزيارة `PORT=20128` و`NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
-| لا توجد سجلات للطلب ضمن `السجلات/` | اضبط `ENABLE_REQUEST_LOGS=true` |
-| EACCES: تم رفض الإذن | اضبط `DATA_DIR=/path/to/writable/dir` لتجاوز `~/.omniroute` |
-| استراتيجية لا تنقذ | التحديث إلى الإصدار 1.4.11+ (إصلاح مخطط Zod لاستمرارية الإعدادات) | ---## Provider Issues |
+
+Common problems and solutions for OmniRoute.
+
+---
+
+## Quick Fixes
+
+| Problem | Solution |
+| ----------------------------- | ------------------------------------------------------------------ |
+| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) |
+| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
+| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` |
+| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` |
+| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) |
+| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below |
+| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below |
+
+---
+
+## Node.js Compatibility
+
+
+
+### Login page crashes or shows "Module self-registration" error
+
+**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database.
+
+**Symptoms:**
+- Login page shows a blank screen or a server error
+- Console shows `Error: Module did not self-register` or similar native binding errors
+- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected
+
+**Fix:**
+
+1. Install Node.js 22 LTS (recommended):
+ ```bash
+ nvm install 22
+ nvm use 22
+ ```
+2. Verify your version: `node --version` should show `v22.x.x`
+3. Reinstall OmniRoute: `npm install -g omniroute`
+4. Restart: `omniroute`
+
+> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**.
+
+---
+
+## Proxy Issues
+
+
+
+### Provider validation shows "fetch failed"
+
+**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing.
+
+**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically.
+
+### Token health check fails with "fetch failed"
+
+**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection.
+
+**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+.
+
+### SOCKS5 proxy returns "invalid onRequestStart method"
+
+**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation.
+
+**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+.
+
+---
+
+## Provider Issues
### "Language model did not provide messages"
-**السبب:**استنفدت حصة الموفر.
+**Cause:** Provider quota exhausted.
-**الإصلاح:**
+**Fix:**
-1. تحقق من تعقب الحصص في لوحة القيادة
-2. استخدم المجموعة من المستويات التجريبية
-3. قم بالبديل إلى اللغة اللاتينية الأرخص/المجانية### تحديد المعدل
+1. Check dashboard quota tracker
+2. Use a combo with fallback tiers
+3. Switch to cheaper/free tier
-**السبب:**استنفدت حصة الاشتراك.
+### Rate Limiting
-**الإصلاح:**
+**Cause:** Subscription quota exhausted.
-- إضافة بيع: `cc/clude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-- استخدم GLM/MiniMax كنسخة بيعية بسعر رخيص### OAuth Token منتهي الصلاحية
+**Fix:**
-يقوم OmniRoute بكتابة الشعارات المميزة. إذا كانت هناك مشاكل:
+- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Use GLM/MiniMax as cheap backup
-1. لوحة المعلومات → الموفر → إعادة الاتصال
-2. قم بإلغاء الحذف وإضافة اتصال الموفر---## Cloud Issues
+### OAuth Token Expired
+
+OmniRoute auto-refreshes tokens. If issues persist:
+
+1. Dashboard → Provider → Reconnect
+2. Delete and re-add the provider connection
+
+---
+
+## Cloud Issues
### Cloud Sync Errors
-1. تحقق من نقاط `BASE_URL` لمثيلك قيد التشغيل (على سبيل المثال، `http://localhost:20128`)
-2. تحقق من نقاط `CLOUD_URL` إلى نقطة نهاية السحابة الخاصة بك (على سبيل المثال، `https://omniroute.dev`)
-3. حافظ على قيم `NEXT_PUBLIC_*` مع قيم من جانب العمال### Cloud `stream=false` Returns 500
+1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`)
+2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`)
+3. Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**العلامة:**`الرمز المميز 'd'...' غير متوقع في نقطة نهاية السحابة للمكالمات غير المتدفقة.
+### Cloud `stream=false` Returns 500
-**السبب:**يقوم المنبع بإرجاع حمولة SSE أثناء العميل JSON.
+**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls.
-**الحل البديل:**استخدم `stream=true` للمكالمات السحابية المباشرة. قم بتضمين SSE المحلي → JSON الاحتياطي.### السحابة تقول أنها متصلة ولكن "مفتاح API غير صالح"
+**Cause:** Upstream returns SSE payload while client expects JSON.
-1. أنشئ مفتاحًا جديدًا من لوحة التحكم المحلية (`/api/keys`)
-2. قم بتشغيل البروتوكولات السحابية: قم بتمكين السحابة → النوبات الآن
-3. لا يزال بإمكانها المفاتيح القديمة/غير المتزامنة إرجاع "401" على السحابة---## Docker Issues
+**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback.
+
+### Cloud Says Connected but "Invalid API key"
+
+1. Create a fresh key from local dashboard (`/api/keys`)
+2. Run cloud sync: Enable Cloud → Sync Now
+3. Old/non-synced keys can still return `401` on cloud
+
+---
+
+## Docker Issues
### CLI Tool Shows Not Installed
-1. تحقق من استهلاك وقت التشغيل: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
-2. بالنسبة لوضع الهاتف المحمول: استخدم الصورة الهدف `runner-cli` (CLIs المجمعة)
-3. بالنسبة لصلاحية التثبيت المحلية: قم بـ `CLI_EXTRA_PATHS` وتثبيت دليل المضيفة للقراءة فقط
-4. إذا كان "تم التثبيت = صحيح" و"قابل للتشغيل = خطأ": تم العثور على الملف الثنائي ولكن فشل التحقق من الصحة### Quick Runtime Validation```bash
- curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
- curl -s http://localhost:20128/api/cli-tools/claude-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
- curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
+1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
+2. For portable mode: use image target `runner-cli` (bundled CLIs)
+3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only
+4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck
-````
+### Quick Runtime Validation
+
+```bash
+curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
+curl -s http://localhost:20128/api/cli-tools/claude-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
+curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
+```
---
@@ -70,106 +153,160 @@
### High Costs
-1. التحقق من إحصائيات استخدام لوحة المعلومات →
-2. قم باستبدال النموذج الأساسي بـ GLM/MiniMax
-3. استخدم طلب التقديم (Gemini CLI، Qoder) للمهام غير المرغوب فيه
-4. قم بإنشاء اقتصاديات التكلفة لكل مفتاح برمجة التطبيقات: لوحة المعلومات ← مفاتيح برمجة التطبيقات ← الميزانية---## Debugging
+1. Check usage stats in Dashboard → Usage
+2. Switch primary model to GLM/MiniMax
+3. Use free tier (Gemini CLI, Qoder) for non-critical tasks
+4. Set cost budgets per API key: Dashboard → API Keys → Budget
+
+---
+
+## Debugging
### Enable Request Logs
-قم بزيارة `ENABLE_REQUEST_LOGS=true` في ملف `.env` الخاص بك. تسجيل سجلات ضمن دليل "السجلات/".### التحقق من صحة المزود```bash
+Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory.
+
+### Check Provider Health
+
+```bash
# Health dashboard
http://localhost:20128/dashboard/health
# API health check
curl http://localhost:20128/api/monitoring/health
-````
+```
### Runtime Storage
-- الحالة الرئيسية: `${DATA_DIR}/storage.sqlite` (الموفرون، المجموعات، الأسماء المستعارة، المفاتيح، الإعدادات)
- -استخدام: جداول SQLite في `storage.sqlite` (`usage_history`، `call_logs`، `proxy_logs`) + اختياري `${DATA_DIR}/log.txt` و`${DATA_DIR}/call_logs/`
-- أرشيف الطلب: `/logs/...` (عندما يكون `ENABLE_REQUEST_LOGS=true`)---## Circuit Breaker Issues
+- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings)
+- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/`
+- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`)
+
+---
+
+## Circuit Breaker Issues
### Provider stuck in OPEN state
-عندما يكون حظر دائرة الموفر مفتوحًا، يتم حظره حتى نهاية فترة التهدئة.
+When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires.
-**الإصلاح:**
+**Fix:**
-1. انتقل إلى**لوحة التحكم ← الإعدادات ← طيران**
-2. التحقق من بطاقة القاطع الكهربائي الخاصة بالمزود المتأثر
-3. انقر فوق**إعادة تعيين الكل**لمسح جميع القواطع، أو انتظر حتى انتهاء فترة التهدئة
-4. التحقق من أن الموفر فعلياً قبل العودة### مقدم الخدمة يستمر في قطع قاطع الدائرة الكهربائية
+1. Go to **Dashboard → Settings → Resilience**
+2. Check the circuit breaker card for the affected provider
+3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire
+4. Verify the provider is actually available before resetting
-إذا وصلت الخدمة بشكل عام في الحالة المفتوحة:
+### Provider keeps tripping the circuit breaker
-1. تحقق من**لوحة التحكم ← الصحة ← صحة مقدم الخدمة**نمط المعرفة تبني
-2. انتقل إلى**الإعدادات → اختلاف → ملفات تعريف الموفر**وكم حتى حد ما
-3. تحقق مما إذا كان الموفر قد قام بتغيير حدود واجهة برمجة التطبيقات (API) أو طلب إعادة المصادقة
-4. قم بمراجعة القياس عن بعد لزمن التعرض - قد يتسبب في حدوث التأثيرات الناتجة في سبب واحد بسبب انتهاء المهلة---## Audio Transcription Issues
+If a provider repeatedly enters OPEN state:
+
+1. Check **Dashboard → Health → Provider Health** for the failure pattern
+2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold
+3. Check if the provider has changed API limits or requires re-authentication
+4. Review latency telemetry — high latency may cause timeout-based failures
+
+---
+
+## Audio Transcription Issues
### "Unsupported model" error
-- تأكد من أنك تستخدم القواعد الصحيحة: `deepgram/nova-3` أو `assemblyai/best`
-- تحقق من أن الموفر متصل في**لوحة التحكم ← الموفرون**### يعود النسخ فارغًا أو فاشلًا
+- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best`
+- Verify the provider is connected in **Dashboard → Providers**
-- التحقق من تنسيقات الصوت المدعومة: `mp3`، `wav`، `m4a`، `flac`، `ogg`، `webm`
-- التحقق من أن حجم الملف يقع ضمن حدود الموفر (عادةً أقل من 25 ميجابايت)
-- التحقق من صلاحية مفتاح API الخاص بالموفر في البطاقة المزودة---## Translator Debugging
+### Transcription returns empty or fails
-استخدم**لوحة المعلومات → المترجم**لتصحيح المناسب لرغبتك:
+- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm`
+- Verify file size is within provider limits (typically < 25MB)
+- Check provider API key validity in the provider card
-| الوضع | متى تستخدم |
-| ------------------ | ------------------------------------------------------------------------------ | -------------------------- |
-| **ساحة اللعب** | قارن ملفات الإدخال/الإخراج جنباً إلى جنب — لا لصق طلباً فاشلاً لترى كيف ترجمته |
-| **اختبار الدردشة** | أرسل الرسائل مباشرة وافحص فاعلية الطلب/الاستجابة الكاملة بما في ذلك الرؤوس |
-| **الاختبار** | قم بإنهاء الفقرات المجمعة عبر مجموعات محددة على الترجمات المعطلة |
-| **مراقبة حية** | شاهد تدفق الطلبات في التنسيق المطلوب على الترجمة المتقطعة | ### مشكلات التنسيق الشائعة |
+---
--**لا تضع علامات التفكير**— تحقق مما إذا كان الموفر المستهدف مقبولاً يمكن توقعه -**استدعاءات مساعدة**— قد توفر بعض الترجمات بشكل فعال لحذف الأشخاص غير المساعدين؛ تحقق في وضع الملعب -**مطالبة النظام مفقودة**— نظام Claude وGemini مع المطالبات المختلفة؛ التحقق من إخراج الترجمة -**ترجع سلسلة SDK أولية أخرى من**- تم إصلاح ذلك في الإصدار 1.1.0: تقوم أداة التأثير الفوري الآن وأنواعها غير الممتازة (`x_groq`، و`usage_breakdown`، وما إلى ذلك) التي فشلت في التحقق من صحة OpenAI SDK Pydantic -**GLM/ERNIE يرفض دور `النظام`**- تم إصلاحه في الإصدار 1.1.0: يقوم بـ«تطبيع الدور التنفيذي برسائل مدمجة في النظام في رسائل المستخدم للنماذج غير المتوافقة» -**لم يتم التعرف على دور "المطور"**- تم إصلاحه في الإصدار 1.1.0: تم تحويله تلقائياً إلى "نظام" لتقديم الخدمات غير التابعة لـ OpenAI -**`json_schema` لا يعمل مع Gemini**— تم إصلاحه في الإصدار 1.1.0: تم الآن تحويل `response_format` إلى `responseMimeType` + `responseSchema` الخاص بـ Gemini---## Resilience Settings
+## Translator Debugging
+
+Use **Dashboard → Translator** to debug format translation issues:
+
+| Mode | When to Use |
+| ---------------- | -------------------------------------------------------------------------------------------- |
+| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates |
+| **Chat Tester** | Send live messages and inspect the full request/response payload including headers |
+| **Test Bench** | Run batch tests across format combinations to find which translations are broken |
+| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues |
+
+### Common format issues
+
+- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting
+- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode
+- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output
+- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures
+- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models
+- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers
+- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema`
+
+---
+
+## Resilience Settings
### Auto rate-limit not triggering
-- يُطبق حتى يتم التعديل تلقائيًا فقط على لوحة مفاتيح برمجة التطبيقات (وليس OAuth/الاشتراك)
-- تحقق من أن**الإعدادات → ← ملفات تعريف الموفر**تم وأيضا تعديلها بشكل تلقائي
-- تحقق مما إذا كان الموفر يعرض رموز الحالة "429" أو الذاكرة "إعادة المحاولة بعد".### ضبط التراجع الأسي
+- Auto rate-limit only applies to API key providers (not OAuth/subscription)
+- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled
+- Check if the provider returns `429` status codes or `Retry-After` headers
-تدعم ملفات تعريف الموفر هذه الإعدادات:
+### Tuning exponential backoff
--**التأخير الأساسي**— وقت الانتظار الأول بعد الأول (الافتراضي: 1 ثانية) -**الحد الأقصى للتأخير**— الحد الأقصى لوقت الانتظار (الافتراضي: 30 ثانية) -**المضاعف**— كمية الزيادة الشاملة لكل فشل متتالي (الافتراضي: 2x)### قطيع مضاد الرعد
+Provider profiles support these settings:
-عندما تصل العديد من الطلبات المتزامنة إلى وفرة السعر، يستخدم تقنية OmniRoute تقنية Mutex + تحديد موعد مباشر لتنزيل الطلبات مباشرة من أجل توقف الحالات المتتالية. وهذا تلقائي لموفري مفاتيح API.---## Optional RAG / LLM failure taxonomy (16 problems)
+- **Base delay** — Initial wait time after first failure (default: 1s)
+- **Max delay** — Maximum wait time cap (default: 30s)
+- **Multiplier** — How much to increase delay per consecutive failure (default: 2x)
-يقوم بعض مستخدمي OmniRoute بالتحرك أمام RAG أو مكدسات الوكيل. في هذه الإعدادات، من الشائع رؤية نمط غريب: يبدو OmniRoute سليمًا (مقدمو خدمة في وضع جيد، وإصدار الأحكام الشخصية على ما بعد، ولا توجد تنبيهات بخلاف حدود القضاء) ولكن الإجابة لا تزال لا تزال صحيحة.
+### Anti-thundering herd
-ومن ثم، يأتي هذا الذي يأتي من خط الأنابيب النهائي RAG، وليس من نفسه.
+When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers.
-إذا كنت تريد مفردات الحصول على وصف لتلك الإخفاقات، فيمكنك استخدام WFGY IssueMap، وهو مصدر ترخيص MIT الخارجي يحدد ستة عشرة نمطًاًا لفشل RAG / LLM. على مستوى عال يغطي:
+---
-- الانجراف استرجاع وحدود السياقة المكسورة
-- الفهارس الفارغة أو القديمة ومخازن المتجهات
-- التضمين مقابل عدم التطابق الدلالي
-- رخص السياقة ورافعة السياقة
-- مجموعة واسعة من الإجابات الاستخدام في التجارة الحرة
-- خلل في النص بين النص والوكيل
-- ذاكرة متعددة للعامل والمؤثرات
-- مشاكل النشر والتمهيد
+## Optional RAG / LLM failure taxonomy (16 problems)
-فكرة بسيطة:
+Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong.
-1. عندما تقوم بالتحقق من خلل في حسابك، قم بالقاطع:
- - مهمة المستخدم وطلبه
- - مجموعة الطريق أو المورد في OmniRoute
- - أي مؤتمر RAG في المراحل النهائية (المستندات المستردة، وأدوات الأدوات، وما إلى ذلك)
-2. قم بتخطيط الحادث لواحد أو من أرقام WFGY IssueMap (`رقم 1`...`رقم 16`).
-3. قم بتخزين الرقم في لوحة المعلومات الخاصة بك، أو دليل التشغيل، أو أداة التعقب بجوار سجلات OmniRoute.
-4. استخدم صفحة WFGY لتقرر ما إذا كنت تريد تغيير مكدس RAG أو المسترد أو استراتيجية التوجيه.
+In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself.
-النص الكامل والوصفات الملموسة موجودة هنا (ترخيص معهد ماساتشوستس فارس، النص فقط):
+If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers:
-[الملف التمهيدي لخريطة مشاكل WFGY](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md)
+- retrieval drift and broken context boundaries
+- empty or stale indexes and vector stores
+- embedding versus semantic mismatch
+- prompt assembly and context window issues
+- logic collapse and overconfident answers
+- long chain and agent coordination failures
+- multi agent memory and role drift
+- deployment and bootstrap ordering problems
-ستتجاهل هذا القسم إذا لم تسمح لـ RAG أو خطوط الأنابيب الخارجية خلف OmniRoute.---## Still Stuck?
+The idea is simple:
--**مشكلات GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**الهندسة الداخلية**: راجع [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) للحصول على التفاصيل -**مرجع واجهة برمجة التطبيقات**: راجع [`docs/API_REFERENCE.md`](API_REFERENCE.md) -**لوحة معلومات الصحة**: التحقق من**معلومات اللوحة ← صحة**معرفة النظام في الوقت الفعلي -**المترجم**: استخدم**لوحة المعلومات ← المترجم**ل التصحيح المناسب لك
+1. When you investigate a bad response, capture:
+ - user task and request
+ - route or provider combo in OmniRoute
+ - any RAG context used downstream (retrieved documents, tool calls, etc)
+2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`).
+3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs.
+4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy.
+
+Full text and concrete recipes live here (MIT license, text only):
+
+[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md)
+
+You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute.
+
+---
+
+## Still Stuck?
+
+- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details
+- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints
+- **Health Dashboard**: Check **Dashboard → Health** for real-time system status
+- **Translator**: Use **Dashboard → Translator** to debug format issues
diff --git a/docs/i18n/ar/llm.txt b/docs/i18n/ar/llm.txt
new file mode 100644
index 0000000000..e9a26be0e3
--- /dev/null
+++ b/docs/i18n/ar/llm.txt
@@ -0,0 +1,476 @@
+# OmniRoute (العربية)
+
+🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt)
+
+---
+
+
+> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app.
+
+## نظرة عامة
+
+OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free.
+
+**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost.
+
+**Current version:** 3.5.5
+
+## Tech Stack
+
+- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`)
+- **Framework:** Next.js 16 (App Router) with TypeScript 5.9
+- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations)
+- **State management:** Zustand (client), SQLite (server persistence)
+- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons
+- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth
+- **Schemas:** Zod v4 for all API / MCP input validation
+- **Background jobs:** Custom token health check scheduler, 24h model auto-sync
+- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses
+- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine
+- **i18n:** next-intl with 30 languages
+- **Desktop:** Electron (cross-platform: Windows, macOS, Linux)
+- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`)
+
+## Project Structure
+
+```
+/
+├── src/ # Main application source
+│ ├── app/ # Next.js App Router pages and API routes
+│ │ ├── (dashboard)/ # Dashboard UI pages
+│ │ │ └── dashboard/
+│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents)
+│ │ │ ├── analytics/ # Usage analytics and charts
+│ │ │ ├── api-manager/ # API key management
+│ │ │ ├── audit/ # Audit logs
+│ │ │ ├── auto-combo/ # Auto-combo engine dashboard
+│ │ │ ├── cache/ # Cache dashboard (semantic cache stats)
+│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.)
+│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates)
+│ │ │ ├── costs/ # Cost tracking per provider/model
+│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs
+│ │ │ ├── health/ # System health (uptime, circuit breakers, latency)
+│ │ │ ├── limits/ # Rate limits dashboard
+│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed)
+│ │ │ ├── media/ # Image/video/music generation + transcription
+│ │ │ ├── memory/ # Memory system dashboard
+│ │ │ ├── onboarding/ # Onboarding wizard
+│ │ │ ├── playground/ # Model playground (Monaco editor, streaming)
+│ │ │ ├── providers/ # Provider management (OAuth + API key + free)
+│ │ │ ├── search-tools/ # Search tools configuration
+│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced)
+│ │ │ ├── skills/ # Skills system dashboard
+│ │ │ ├── translator/ # Format translator + debug tools
+│ │ │ └── usage/ # Usage history
+│ │ ├── api/ # REST API endpoints (51 route directories)
+│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings,
+│ │ │ │ # images, audio, videos, music, moderations, rerank, search,
+│ │ │ │ # responses, messages, registered-keys, quotas, accounts)
+│ │ │ ├── v1beta/ # Gemini-compatible API
+│ │ │ ├── a2a/ # A2A agent management API
+│ │ │ ├── acp/ # ACP agent management API
+│ │ │ ├── oauth/ # OAuth flows per provider
+│ │ │ ├── providers/ # Provider CRUD and batch testing
+│ │ │ ├── models/ # Dashboard model listing and aliases
+│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains)
+│ │ │ ├── memory/ # Memory system API
+│ │ │ ├── skills/ # Skills system API
+│ │ │ ├── evals/ # Eval runner API
+│ │ │ ├── mcp/ # MCP HTTP transport API
+│ │ │ ├── search/ # Search provider API
+│ │ │ ├── webhooks/ # Webhook management
+│ │ │ ├── tunnels/ # Cloudflare tunnel management
+│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.)
+│ │ ├── landing/ # Landing page
+│ │ ├── login/ # Login page
+│ │ ├── forgot-password/ # Password recovery
+│ │ ├── status/ # Status page
+│ │ └── docs/ # In-app documentation
+│ ├── domain/ # Domain types and policy engine
+│ │ ├── policyEngine.ts # Central policy engine
+│ │ ├── comboResolver.ts # Combo resolution logic
+│ │ ├── costRules.ts # Cost calculation rules
+│ │ ├── degradation.ts # Graceful degradation
+│ │ ├── fallbackPolicy.ts # Fallback behavior
+│ │ ├── lockoutPolicy.ts # Account lockout logic
+│ │ ├── modelAvailability.ts # Model availability checks
+│ │ ├── providerExpiration.ts # Provider credential expiration
+│ │ ├── quotaCache.ts # Quota caching layer
+│ │ ├── configAudit.ts # Configuration auditing
+│ │ └── responses.ts # Domain response types
+│ ├── i18n/ # Internationalization
+│ │ └── messages/ # 30 language JSON files
+│ ├── lib/ # Core libraries
+│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server
+│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting)
+│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup
+│ │ │ └── streaming.ts # SSE streaming for A2A
+│ │ ├── acp/ # Agent Communication Protocol registry and manager
+│ │ ├── compliance/ # Compliance policy engine
+│ │ ├── db/ # SQLite database layer (21 modules + migrations)
+│ │ │ ├── core.ts # Database initialization, connection, schema
+│ │ │ ├── providers.ts # Provider connection CRUD
+│ │ │ ├── models.ts # Model catalog management
+│ │ │ ├── combos.ts # Combo configuration
+│ │ │ ├── apiKeys.ts # API key management
+│ │ │ ├── settings.ts # Settings persistence
+│ │ │ ├── backup.ts # Database backup/restore
+│ │ │ ├── proxies.ts # Proxy registry
+│ │ │ ├── prompts.ts # Prompt templates
+│ │ │ ├── webhooks.ts # Webhook subscriptions
+│ │ │ ├── detailedLogs.ts # Detailed request logging
+│ │ │ ├── domainState.ts # Domain state persistence
+│ │ │ ├── registeredKeys.ts # Registered API keys with quotas
+│ │ │ ├── quotaSnapshots.ts # Quota snapshot history
+│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings
+│ │ │ ├── cliToolState.ts # CLI tool state tracking
+│ │ │ ├── encryption.ts # Data encryption
+│ │ │ ├── readCache.ts # Read-through cache layer
+│ │ │ ├── secrets.ts # Secrets management
+│ │ │ ├── stateReset.ts # State reset utilities
+│ │ │ ├── migrationRunner.ts # Schema migration runner
+│ │ │ └── migrations/ # 16 SQL migration files
+│ │ ├── evals/ # Eval runner and scheduler
+│ │ ├── memory/ # Persistent conversational memory
+│ │ │ ├── extraction.ts # Memory extraction from conversations
+│ │ │ ├── injection.ts # Memory injection into context
+│ │ │ ├── retrieval.ts # Memory retrieval/search
+│ │ │ ├── store.ts # Memory persistence layer
+│ │ │ └── summarization.ts # Memory summarization
+│ │ ├── oauth/ # OAuth providers, services, and utilities
+│ │ │ ├── constants/ # Default OAuth credentials (overridable via env)
+│ │ │ ├── providers/ # Provider-specific OAuth configs
+│ │ │ ├── services/ # Provider-specific token exchange logic
+│ │ │ └── utils/ # PKCE, callback server, token helpers
+│ │ ├── plugins/ # Plugin system
+│ │ ├── skills/ # Extensible skill framework
+│ │ │ ├── registry.ts # Skill registration
+│ │ │ ├── executor.ts # Skill execution engine
+│ │ │ ├── sandbox.ts # Skill sandbox environment
+│ │ │ ├── builtin/ # Built-in skills
+│ │ │ ├── interception.ts # Skill request interception
+│ │ │ └── injection.ts # Skill context injection
+│ │ ├── usage/ # Usage tracking system
+│ │ │ ├── callLogs.ts # Call log persistence
+│ │ │ ├── costCalculator.ts # Cost calculation engine
+│ │ │ └── usageHistory.ts # Usage history queries
+│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers
+│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management
+│ │ ├── pricingSync.ts # LiteLLM pricing data sync
+│ │ ├── semanticCache.ts # Semantic caching layer
+│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler
+│ │ ├── webhookDispatcher.ts # Webhook event dispatcher
+│ │ └── localDb.ts # Unified re-export layer for all DB modules
+│ ├── middleware/ # Request middleware
+│ │ └── promptInjectionGuard.ts # Prompt injection detection
+│ ├── mitm/ # MITM proxy capability
+│ │ ├── cert/ # Certificate management
+│ │ ├── dns/ # DNS handling
+│ │ ├── targets/ # Target routing
+│ │ └── manager.ts # MITM proxy manager
+│ ├── shared/ # Shared utilities, components, and constants
+│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.)
+│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes
+│ │ ├── contracts/ # Shared API contracts
+│ │ ├── hooks/ # React hooks
+│ │ ├── middleware/ # Shared middleware utilities
+│ │ ├── schemas/ # Shared Zod schemas
+│ │ ├── services/ # Shared services
+│ │ ├── types/ # Shared TypeScript types
+│ │ ├── validation/ # Zod schemas (settings, providers, routes)
+│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID)
+│ ├── sse/ # SSE proxy pipeline
+│ │ ├── services/ # Auth resolution, format translation, response handling
+│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency
+│ ├── store/ # Zustand client-side stores (theme, providers, etc.)
+│ └── types/ # TypeScript type definitions
+├── open-sse/ # Standalone SSE server (npm workspace)
+│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video,
+│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models)
+│ ├── executors/ # Provider-specific request executors (14 executors)
+│ │ ├── base.ts # Base executor with shared logic
+│ │ ├── default.ts # Default OpenAI-compatible executor
+│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum)
+│ │ ├── codex.ts # OpenAI Codex CLI
+│ │ ├── antigravity.ts # Antigravity IDE
+│ │ ├── github.ts # GitHub Copilot
+│ │ ├── gemini-cli.ts # Gemini CLI
+│ │ ├── kiro.ts # Kiro AI
+│ │ ├── qoder.ts # Qoder AI
+│ │ ├── vertex.ts # Vertex AI (Service Account JSON)
+│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI
+│ │ ├── opencode.ts # OpenCode Zen/Go
+│ │ ├── pollinations.ts # Pollinations AI
+│ │ └── puter.ts # Puter AI
+│ ├── handlers/ # Request handlers per API type (11 handlers)
+│ │ ├── chatCore.ts # Main chat completions handler
+│ │ ├── responsesHandler.ts # OpenAI Responses API handler
+│ │ ├── embeddings.ts # Embedding generation
+│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.)
+│ │ ├── videoGeneration.ts # Video generation
+│ │ ├── musicGeneration.ts # Music generation
+│ │ ├── audioSpeech.ts # Text-to-speech
+│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI)
+│ │ ├── moderations.ts # Content moderation
+│ │ ├── rerank.ts # Reranking API
+│ │ └── search.ts # Web search API
+│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP)
+│ │ ├── server.ts # MCP server core (tool registration, scope enforcement)
+│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools)
+│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a)
+│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes)
+│ │ ├── audit.ts # Tool call audit logging
+│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat
+│ │ └── httpTransport.ts # HTTP transport handler
+│ ├── services/ # 36+ service modules
+│ │ ├── combo.ts # Core routing engine
+│ │ ├── usage.ts # Usage tracking
+│ │ ├── tokenRefresh.ts # OAuth token refresh
+│ │ ├── rateLimitManager.ts # Rate limit management
+│ │ ├── accountFallback.ts # Multi-account fallback
+│ │ ├── sessionManager.ts # Session management
+│ │ ├── wildcardRouter.ts # Wildcard model routing
+│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration)
+│ │ ├── intentClassifier.ts # Request intent classification
+│ │ ├── taskAwareRouter.ts # Task-aware routing
+│ │ ├── thinkingBudget.ts # Thinking budget management
+│ │ ├── contextManager.ts # Context window management
+│ │ ├── modelDeprecation.ts # Model deprecation handling
+│ │ ├── modelFamilyFallback.ts # Intra-family model fallback
+│ │ ├── emergencyFallback.ts # Emergency fallback
+│ │ ├── workflowFSM.ts # Workflow state machine
+│ │ ├── backgroundTaskDetector.ts # Background task detection
+│ │ ├── ipFilter.ts # IP-based access control
+│ │ ├── signatureCache.ts # CLI signature caching
+│ │ ├── volumeDetector.ts # Request volume detection
+│ │ ├── contextHandoff.ts # Context relay handoff generation and injection
+│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay
+│ │ └── ... # Additional services (14 more modules)
+│ ├── transformer/ # Responses API transformer
+│ │ └── responsesTransformer.ts
+│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek)
+│ │ ├── request/ # Request translators per provider
+│ │ ├── response/ # Response translators per provider
+│ │ ├── helpers/ # Translation helpers
+│ │ └── image/ # Image format translation
+│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.)
+├── electron/ # Electron desktop app (cross-platform)
+│ ├── main.js # Electron main process
+│ ├── preload.js # Preload script (IPC bridge)
+│ └── assets/ # App icons and assets
+├── tests/ # Test suites
+│ ├── unit/ # 122 unit test files
+│ ├── integration/ # Integration tests
+│ ├── e2e/ # Playwright E2E tests
+│ ├── security/ # Security tests
+│ ├── translator/ # Translator-specific tests
+│ └── load/ # Load tests
+├── docs/ # Documentation
+│ ├── i18n/ # 30-language translated docs
+│ ├── ARCHITECTURE.md # Full architecture documentation
+│ ├── API_REFERENCE.md # API reference
+│ ├── USER_GUIDE.md # User guide
+│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview
+│ ├── CLI-TOOLS.md # CLI tools integration guide
+│ ├── A2A-SERVER.md # A2A agent protocol documentation
+│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring)
+│ ├── MCP-SERVER.md # MCP server (25 tools)
+│ ├── TROUBLESHOOTING.md # Troubleshooting guide
+│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide
+│ ├── openapi.yaml # OpenAPI specification
+│ └── screenshots/ # Dashboard screenshots
+├── bin/ # CLI entry points (omniroute, reset-password)
+├── scripts/ # Build and utility scripts
+└── .env.example # Environment variable template
+```
+
+## Key Features (v3.5.5)
+
+### Core Proxy
+- **60+ AI providers** with automatic format translation
+- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible)
+- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay
+- **4-tier fallback**: Subscription → API Key → Cheap → Free
+- **Context Relay strategy**: Session handoff summaries on account rotation for continuity
+- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown
+- **Semantic caching** with cache hit/miss headers
+- **Idempotency** with configurable dedup window
+- **Circuit breaker** per provider with configurable thresholds
+- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback
+- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers
+- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement
+- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization
+- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills
+- **Prompt Injection Guard**: Middleware-level prompt injection detection
+- **MITM Proxy**: Certificate management, DNS handling, and target routing
+- **Cloudflare Tunnels**: Managed tunnel creation for remote access
+- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches)
+
+### الأمان
+- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs)
+- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()`
+- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()`
+- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection
+- **CLI Fingerprint Matching** — Per-provider request signature matching
+- **Prompt injection guard** — Request middleware detection
+- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`)
+- **PII sanitizer** — Sensitive data scrubbing in logs
+
+### Dashboard Pages (23 sections)
+- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons
+- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies
+- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics
+- **Analytics** — Token consumption, cost, heatmaps, distributions
+- **Health** — Uptime, memory, latency percentiles, circuit breakers
+- **Logs** — Request, Proxy, Audit, Console (tabbed)
+- **Audit** — Audit trail and compliance logging
+- **Costs** — Cost tracking per provider/model
+- **Limits** — Rate limit monitoring
+- **Cache** — Semantic cache statistics and management
+- **CLI Tools** — One-click configuration for 10+ AI CLI tools
+- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration
+- **Playground** — Test any model with Monaco editor, streaming responses
+- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files)
+- **Search Tools** — Search provider configuration and testing
+- **Memory** — Memory system management and visualization
+- **Skills** — Skills framework management and execution
+- **Translator** — Format debugging: playground, chat tester, test bench, live monitor
+- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced
+- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed)
+- **Onboarding** — Setup wizard for new users
+- **Usage** — Usage history and analytics
+- **API Manager** — API key management with scoped permissions
+
+### Protocol Support
+- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations`
+- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens`
+- **OpenAI Responses** — `/v1/responses`
+- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}`
+- **Ollama** — `/v1/api/chat`, `/api/tags`
+- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily)
+- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP)
+- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills)
+- **ACP** — Agent Communication Protocol registry and manager
+
+### MCP Server (25 Tools)
+| Category | Tools |
+|-----------|-------|
+| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` |
+| Memory (3) | `memory_search`, `memory_add`, `memory_clear` |
+| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` |
+
+**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience`
+
+### Provider Categories
+
+**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI
+
+**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline
+
+**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan
+
+**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs
+
+### Internationalization
+- 30 languages for UI (all dashboard pages)
+- 30 translated documentation sets in docs/i18n/
+- Language switcher in documentation
+
+## Key Architectural Decisions
+
+1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints.
+
+2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently.
+
+3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation.
+
+4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity.
+
+5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client.
+
+6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes.
+
+7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`).
+
+8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages.
+
+9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations.
+
+10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`.
+
+11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation.
+
+12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline.
+
+13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast.
+
+## Main Flows
+
+### Proxy Request Flow
+1. Client sends OpenAI-format request to `/v1/chat/completions`
+2. API key validation
+3. Model resolution: direct model or combo lookup
+4. For combos: iterate through models with selected strategy
+5. Auth resolution: get credentials for the target provider
+6. Format translation: OpenAI → provider native format
+7. CLI fingerprint matching (if enabled for provider)
+8. Upstream request with circuit breaker and rate limiting
+9. Response translation: provider → OpenAI format
+10. omniModel tag sanitization (strip internal tags)
+11. SSE streaming back to client
+12. Memory extraction (if memory system enabled)
+13. Usage logging and cost calculation
+
+### OAuth Flow
+1. Dashboard initiates `/api/oauth/[provider]/authorize`
+2. User completes OAuth login in browser
+3. Callback hits `/api/oauth/[provider]/exchange`
+4. Tokens stored as a provider connection in SQLite
+5. Background job refreshes tokens before expiry
+
+## Important Notes for LLMs
+
+1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only).
+
+2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`).
+
+3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services.
+
+4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`.
+
+5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module.
+
+6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`).
+
+7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes.
+
+8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB.
+
+9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown.
+
+10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130).
+
+11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux).
+
+12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint.
+
+13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API.
+
+14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API.
+
+15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time.
+
+16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`.
+
+17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22.
+
+18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false.
+
+## Links
+
+- Repository: https://github.com/diegosouzapw/OmniRoute
+- Website: https://omniroute.online
+- npm: https://www.npmjs.com/package/omniroute
+- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute
+- Documentation: See `/docs/` directory
diff --git a/docs/i18n/bg/README.md b/docs/i18n/bg/README.md
index ec9265e506..3445919355 100644
--- a/docs/i18n/bg/README.md
+++ b/docs/i18n/bg/README.md
@@ -4,11 +4,14 @@
---
+
### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback.
-_Вашият универсален API прокси — една крайна точка, 60+ доставчици, нулев престой. Сега с**MCP сървър (25 инструмента)**,**A2A протокол**,**Системи за памет/умения**и**Electron Desktop App**._
+_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._
-**Завършвания на чат • Вграждания • Генериране на изображения • Видео • Музика • Аудио • Прекласиране •**Уеб търсене**• MCP сървър • A2A протокол • 100% TypeScript**---
+**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript**
+
+---
@@ -39,9 +42,13 @@ _Вашият универсален API прокси — една крайна
[](https://omniroute.online)
[](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-[🌐 Уебсайт](https://omniroute.online) • [🚀 Бърз старт](#-бърз старт) • [💡 Функции](#-ключови-функции) • [📖 Документи](#-документация) • [💰 Ценообразуване](#-ценообразуване с един поглед) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-🌐**Налично на:**🇺🇸 [английски](README.md) | 🇧🇷 [Португалски (Бразилия)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [италиански](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Нидерландия](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Португалия)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Полски](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [филипински](docs/i18n/phi/README.md) | 🇨🇿 [Чещина](docs/i18n/cs/README.md)---
+
+
+🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)
+
+---
## 🖼️ Main Dashboard
@@ -53,555 +60,629 @@ _Вашият универсален API прокси — една крайна
## 📸 Dashboard Preview
-<подробности>
+
+Click to see dashboard screenshots
-Щракнете, за да видите екранни снимки на таблото за управление
+| Page | Screenshot |
+| -------------- | ------------------------------------------------- |
+| **Providers** |  |
+| **Combos** |  |
+| **Analytics** |  |
+| **Health** |  |
+| **Translator** |  |
+| **Settings** |  |
+| **CLI Tools** |  |
+| **Usage Logs** |  |
+| **Endpoints** |  |
-| Страница | Екранна снимка |
-| -------------------------- | ----------------------------------------------------- | ---------- |
-| **Доставчици** |  |
-| **Комбота** |  |
-| **Анализ** |  |
-| **Здраве** |  |
-| **Преводач** |  |
-| **Настройки** |  |
-| **CLI инструменти** |  |
-| **Дневници за използване** |  |
-| **Крайни точки** |  | |
+
---
### 🤖 Free AI Provider for your favorite coding agents
-_Свържете всеки базиран на AI IDE или CLI инструмент чрез OmniRoute — безплатен API шлюз за неограничено кодиране._
-
-<таблица>
-
-
-
-
-OpenClaw
-
-⭐ 205K
- |
-
-
-
-NanoBot
-
-⭐ 20,9K
- |
-
-
-
-PicoClaw
-
-⭐ 14,6K
- |
-
-
-
-ZeroClaw
-
-⭐ 9,9K
- |
-
-
-
-Железен нокът
-
-⭐ 2,1K
- |
-
-
-
-
-
-OpenCode
-
-⭐ 106K
- |
-
-
-
-Codex CLI
-
-⭐ 60,8K
- |
-
-
-
-Клод Код
-
-⭐ 67,3K
- |
-
-
-
-Gemini CLI
-
-⭐ 94,7K
- |
-
-
-
-Код на килограм
-
-⭐ 15,5K
- |
-
+_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._
+
-📡 Всички агенти се свързват чрез http://localhost:20128/v1 или http://cloud.omniroute.online/v1 — една конфигурация, неограничени модели и квота---
+📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota
+
+---
## 🤔 Why OmniRoute?
-**Спрете да пилеете пари и да достигате лимити:**
+**Stop wasting money and hitting limits:**
--
Абонаментната квота изтича неизползвана всеки месец
--
Ограниченията на скоростта ви спират да кодирате по средата
--
Скъпи API ($20-50/месец на доставчик)
--
Ръчно превключване между доставчици
+-
Subscription quota expires unused every month
+-
Rate limits stop you mid-coding
+-
Expensive APIs ($20-50/month per provider)
+-
Manual switching between providers
-**OmniRoute решава това:**
+**OmniRoute solves this:**
-- ✅**Увеличете максимално абонаментите**- Проследете квотата, използвайте всеки бит преди нулиране
-- ✅**Автоматичен резервен режим**- Абонамент → API ключ → Евтини → Безплатно, нулев престой
-- ✅**Множество акаунти**- Кръгови сметки между акаунти на доставчик
-- ✅**Универсален**- Работи с Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, всеки CLI инструмент---
+- ✅ **Maximize subscriptions** - Track quota, use every bit before reset
+- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime
+- ✅ **Multi-account** - Round-robin between accounts per provider
+- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool
+
+---
## 📧 Support
-> 💬**Присъединете се към нашата общност!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Получавайте помощ, споделяйте съвети и бъдете в течение.
+> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated.
--**Уебсайт**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Проблеми**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Група на общността](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Принос**: Вижте [CONTRIBUTING.md](CONTRIBUTING.md), отворете PR или изберете „добър първи брой“ -**Оригинален проект**: [9router от decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug?
+- **Website**: [omniroute.online](https://omniroute.online)
+- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute)
+- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue`
+- **Original Project**: [9router by decolua](https://github.com/decolua/9router)
-Когато отваряте проблем, моля, изпълнете командата system-info и прикачете генерирания файл:```bash
+### 🐛 Reporting a Bug?
+
+When opening an issue, please run the system-info command and attach the generated file:
+
+```bash
npm run system-info
-
```
-Това генерира `system-info.txt` с вашата версия на Node.js, версия на OmniRoute, подробности за операционната система, инсталирани CLI инструменти (qoder, gemini, claude, codex, antigravity, droid и т.н.), състояние на Docker/PM2 и системни пакети – всичко, от което се нуждаем, за да възпроизведем бързо проблема ви. Прикачете файла директно към вашия проблем с GitHub.---
+This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue.
+
+---
## 🔄 How It Works
```
-
┌─────────────┐
-│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
-│ Tool │
+│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
+│ Tool │
└──────┬──────┘
-│ http://localhost:20128/v1
-↓
+ │ http://localhost:20128/v1
+ ↓
┌─────────────────────────────────────────┐
-│ OmniRoute (Smart Router) │
-│ • Format translation (OpenAI ↔ Claude) │
-│ • Quota tracking + Embeddings + Images │
-│ • Auto token refresh │
+│ OmniRoute (Smart Router) │
+│ • Format translation (OpenAI ↔ Claude) │
+│ • Quota tracking + Embeddings + Images │
+│ • Auto token refresh │
└──────┬──────────────────────────────────┘
-│
-├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
-│ ↓ quota exhausted
-├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
-│ ↓ budget limit
-├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
-│ ↓ budget limit
-└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
+ │
+ ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
+ │ ↓ quota exhausted
+ ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
+ │ ↓ budget limit
+ ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
+ │ ↓ budget limit
+ └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
Result: Never stop coding, minimal cost
-
-````
+```
---
## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases
->**Всеки разработчик, използващ AI инструменти, се сблъсква с тези проблеми всеки ден.**OmniRoute е създаден, за да разреши всички тях — от преразход на разходите до регионални блокове, от повредени OAuth потоци до операции на протоколи и корпоративна наблюдаемост.
+> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability.
-<подробности>
-💸 1. „Плащам за скъп абонамент, но все още ме прекъсват ограниченията“
+
+💸 1. "I pay for an expensive subscription but still get interrupted by limits"
-Разработчиците плащат $20–200/месец за Claude Pro, Codex Pro или GitHub Copilot. Дори и да плащате, квотата има таван — 5 часа използване, седмични лимити или лимити на цените на минута. По средата на сесията на кодиране, доставчикът спира да отговаря и разработчикът губи поток и производителност.
+Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity.
-**Как OmniRoute го решава:**
+**How OmniRoute solves it:**
--**Smart 4-Tier Fallback**— Ако квотата за абонамент се изчерпи, автоматично пренасочва към API Key → Евтино → Безплатно с нулева ръчна намеса
--**Проследяване на ограниченията на доставчика**— Кешираните моментни снимки на квотата се опресняват по график от страна на сървъра (по подразбиране `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) с ръчно опресняване, налично в потребителския интерфейс
--**Поддръжка на множество акаунти**— Множество акаунти на доставчик с автоматичен кръгов режим — когато единият свърши, превключва към следващия
--**Персонализирани комбинации**— Резервни вериги с възможност за персонализиране с 9 стратегии за балансиране (приоритетни, претеглени, първо запълване, кръгови, P2C, произволни, най-малко използвани, оптимизирани по отношение на разходите, строго произволни)
--**Codex Business Quotas**— Мониторинг на квотите на работното пространство на бизнеса/екипа директно в таблото за управление
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention
+- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI
+- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next
+- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**)
+- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard
-<подробности>
-🔌 2. „Трябва да използвам няколко доставчика, но всеки има различен API“
+
-OpenAI използва един формат, Claude (Anthropic) използва друг, Gemini още един. Ако разработчикът иска да тества модели от различни доставчици или резервен вариант между тях, той трябва да преконфигурира SDK, да промени крайните точки, да се справи с несъвместими формати. Персонализираните доставчици (FriendLI, NIM) имат крайни точки на нестандартен модел.
+
+🔌 2. "I need to use multiple providers but each has a different API"
-**Как OmniRoute го решава:**
+OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints.
--**Unified Endpoint**— Един `http://localhost:20128/v1` служи като прокси за всички 60+ доставчици
--**Превод на формати**— Автоматично и прозрачно: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
--**Response Sanitization**— Премахва нестандартните полета (`x_groq`, `usage_breakdown`, `service_tier`), които нарушават OpenAI SDK v1.83+
--**Нормализиране на ролята**— Преобразува `developer` → `system` за доставчици, които не са OpenAI; `система` → `потребител` за GLM/ERNIE
--**Think Tag Extraction**— Извлича `` блокове от модели като DeepSeek R1 в стандартизирано `reasoning_content`
--**Структуриран изход за Gemini**— `json_schema` → `responseMimeType`/`responseSchema` автоматично преобразуване
--**`stream` по подразбиране е `false`**— Подравнява се със спецификацията на OpenAI, като се избягват неочаквани SSE в SDK на Python/Rust/Go
+**How OmniRoute solves it:**
-<подробности>
-🌐 3. „Моят доставчик на AI блокира моя регион/държава“
+- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers
+- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
+- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+
+- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE
+- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content`
+- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion
+- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs
-Доставчици като OpenAI/Codex блокират достъпа от определени географски региони. Потребителите получават грешки като `unsupported_country_region_territory` по време на OAuth и API връзки. Това е особено разочароващо за разработчиците от развиващите се страни.
+
-**Как OmniRoute го решава:**
+
+🌐 3. "My AI provider blocks my region/country"
--**3-Level Proxy Config**— Конфигурируем прокси на 3 нива: глобално (цял трафик), на доставчик (само един доставчик) и на връзка/ключ
--**Цветно кодирани прокси значки**— Визуални индикатори: 🟢 глобален прокси, 🟡 прокси на доставчик, 🔵 прокси за връзка, винаги показващ IP
--**OAuth обмен на токени през прокси**— OAuth потокът също минава през проксито, решавайки `unsupported_country_region_territory`
--**Тестове за връзка чрез прокси**— Тестовете за връзка използват конфигурирания прокси (без повече директен байпас)
--**SOCKS5 Support**— Пълна SOCKS5 прокси поддръжка за изходящо маршрутизиране
--**TLS Fingerprint Spoofing**— подобен на браузър TLS пръстов отпечатък чрез `wreq-js` за заобикаляне на откриването на ботове
--**🔏 Съпоставяне на пръстови отпечатъци на CLI**— Пренарежда заглавките и полетата на основния текст, за да съответстват на собствените двоични подписи на CLI, драстично намалявайки риска от маркиране на акаунта. Прокси IP адресът се запазва — получавате едновременно стелт**и**IP маскиране
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries.
-<подробности>
-🆓 4. „Искам да използвам AI за кодиране, но нямам пари“
+**How OmniRoute solves it:**
-Не всеки може да плаща $20-200/месец за абонаменти за AI. Студенти, разработчици от развиващи се страни, любители и фрийлансъри се нуждаят от достъп до качествени модели на нулева цена.
+- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key
+- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP
+- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory`
+- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass)
+- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection
+- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously
-**Как OmniRoute го решава:**
+
--**Вградени доставчици на безплатни нива**— Вградена поддръжка за 100% безплатни доставчици: Qoder (5 неограничени модела чрез OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 неограничени модела: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID безплатно), Gemini CLI (180K токена/месец безплатно)
--**Ollama Cloud**— Хоствани в облака Ollama модели на `api.ollama.com` с безплатно ниво „Light usage“; използвайте префикса `ollamacloud/<модел>`
--**Безплатни само комбинации**— Верига `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/месец с нулев престой
--**NVIDIA NIM безплатен достъп**— ~40 RPM dev-вечно безплатен достъп до 70+ модела на build.nvidia.com (преход от кредити към чисти лимити на скоростта)
--**Стратегия за оптимизиране на разходите**— Стратегия за маршрутизиране, която автоматично избира най-евтиния наличен доставчик
+
+🆓 4. "I want to use AI for coding but I have no money"
-<подробности>
-🔒 5. „Трябва да защитя своя AI шлюз от неоторизиран достъп“
+Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost.
-При излагане на AI шлюз към мрежата (LAN, VPS, Docker), всеки с адреса може да използва токените/квотата на разработчика. Без защита приложните програмни интерфейси (API) са уязвими за злоупотреба, незабавно инжектиране и злоупотреба.
+**How OmniRoute solves it:**
-**Как OmniRoute го решава:**
+- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free)
+- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix
+- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime
+- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits)
+- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider
--**API Key Management**— Генериране, ротация и обхват за всеки доставчик със специална страница `/dashboard/api-manager`
--**Разрешения на ниво модел**— Ограничете API ключовете до конкретни модели (`openai/*`, шаблони със заместващи символи), с превключвател Разрешаване на всички/Ограничаване
--**API Endpoint Protection**— Изискване на ключ за `/v1/models` и блокиране на определени доставчици от списъка
--**Auth Guard + CSRF Protection**— Всички маршрути на таблото са защитени с мидълуер `withAuth` + CSRF токени
--**Ограничител на скоростта**— Ограничаване на скоростта на IP с конфигурируеми прозорци
--**IP Filtering**— Списък с разрешени/списък с блокирани за контрол на достъпа
--**Prompt Injection Guard**— Дезинфекция срещу злонамерени бързи модели
--**AES-256-GCM криптиране**— Идентификационните данни са криптирани в покой
+
-<подробности>
-🛑 6. „Доставчикът ми се срина и загубих потока на кодиране“
+
+🔒 5. "I need to protect my AI gateway from unauthorized access"
-Доставчиците на AI могат да станат нестабилни, да върнат грешки 5xx или да достигнат временни лимити на скоростта. Ако разработчикът зависи от един доставчик, той е прекъснат. Без прекъсвачи многократните повторни опити могат да сринат приложението.
+When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse.
-**Как OmniRoute го решава:**
+**How OmniRoute solves it:**
--**Прекъсвач за всеки модел**— Автоматично отваряне/затваряне с конфигурируеми прагове и изчакване (затворен/отворен/полуотворен), обхват за всеки модел, за да се избегнат каскадни блокове
--**Exponential Backoff**— Прогресивни закъснения при повторен опит
--**Anti-Thundering Herd**— Mutex + семафорна защита срещу едновременни повторни бури
--**Combo Fallback Chains**— Ако основният доставчик се провали, автоматично преминава през веригата без намеса
--**Combo Circuit Breaker**— Автоматично деактивира неуспешни доставчици в рамките на комбинирана верига
--**Health Dashboard**— Мониторинг на времето на работа, състояния на прекъсвачи, блокировки, статистика на кеша, латентност на p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page
+- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle
+- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing
+- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens
+- **Rate Limiter** — Per-IP rate limiting with configurable windows
+- **IP Filtering** — Allowlist/blocklist for access control
+- **Prompt Injection Guard** — Sanitization against malicious prompt patterns
+- **AES-256-GCM Encryption** — Credentials encrypted at rest
-<подробности>
-🔧 7. „Конфигурирането на всеки AI инструмент е досадно и повтарящо се“
+
-Разработчиците използват Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Всеки инструмент се нуждае от различна конфигурация (крайна точка на API, ключ, модел). Преконфигурирането при смяна на доставчик или модел е загуба на време.
+
+🛑 6. "My provider went down and I lost my coding flow"
-**Как OmniRoute го решава:**
+AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application.
--**CLI Tools Dashboard**— Специална страница с настройка с едно кликване за Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
--**GitHub Copilot Config Generator**— Генерира `chatLanguageModels.json` за VS код с масов избор на модел
--**Onboarding Wizard**— Насочвана настройка в 4 стъпки за потребители за първи път
--**Една крайна точка, всички модели**— Конфигурирайте `http://localhost:20128/v1` веднъж, достъп до 60+ доставчици
+**How OmniRoute solves it:**
-<подробности>
-🔑 8. „Управлението на OAuth токени от множество доставчици е истински ад“
+- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks
+- **Exponential Backoff** — Progressive retry delays
+- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms
+- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention
+- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain
+- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency
-Claude Code, Codex, Gemini CLI, Copilot — всички използват OAuth 2.0 с изтичащи токени. Разработчиците трябва постоянно да се удостоверяват отново, да се справят с „client_secret липсва“, „redirect_uri_mismatch“ и повреди на отдалечени сървъри. OAuth на LAN/VPS е особено проблематичен.
+
-**Как OmniRoute го решава:**
+
+🔧 7. "Configuring each AI tool is tedious and repetitive"
--**Auto Token Refresh**— OAuth токените се опресняват във фонов режим преди изтичане
--**OAuth 2.0 (PKCE) Вграден**— Автоматичен поток за Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
--**Multi-Account OAuth**— Множество акаунти на доставчик чрез JWT/ID извличане на токени
--**OAuth LAN/Remote Fix**— Частно IP откриване за `redirect_uri` + ръчен URL режим за отдалечени сървъри
--**OAuth зад Nginx**— Използва `window.location.origin` за обратна прокси съвместимост
--**Отдалечено ръководство за OAuth**— Ръководство стъпка по стъпка за идентификационни данни на Google Cloud на VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time.
-<подробности>
-📊 9. „Не знам колко харча или къде“
+**How OmniRoute solves it:**
-Разработчиците използват множество платени доставчици, но нямат унифициран поглед върху разходите. Всеки доставчик има собствено табло за таксуване, но няма консолидиран изглед. Неочакваните разходи могат да се натрупат.
+- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
+- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection
+- **Onboarding Wizard** — Guided 4-step setup for first-time users
+- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers
-**Как OmniRoute го решава:**
+
--**Табло за анализ на разходите**— Проследяване на разходите за токени и управление на бюджета за доставчик
--**Бюджетни ограничения за ниво**— Таван на разходите за ниво, което задейства автоматичен резервен вариант
--**Конфигурация на ценообразуване за модел**— Конфигурируеми цени за модел
--**Статистика на използването на API ключ**— Брой заявки и последно използвано клеймо за всеки ключ
--**Табло за управление на анализи**— Статистически карти, диаграма на използването на модела, таблица на доставчика с проценти на успех и закъснение
+
+🔑 8. "Managing OAuth tokens from multiple providers is hell"
-<подробности>
-🐛 10. „Не мога да диагностицирам грешки и проблеми в обажданията с AI“
+Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic.
-Когато обаждането е неуспешно, разработчикът не знае дали е ограничение на скоростта, изтекъл токен, грешен формат или грешка на доставчика. Фрагментирани регистрационни файлове в различни терминали. Без възможност за наблюдение отстраняването на грешки е метод проба-грешка.
+**How OmniRoute solves it:**
-**Как OmniRoute го решава:**
+- **Auto Token Refresh** — OAuth tokens refresh in background before expiration
+- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
+- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction
+- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers
+- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility
+- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker
--**Табло за управление на унифицирани регистрационни файлове**— 4 раздела: регистрационни файлове за заявки, регистрационни файлове за прокси, регистрационни файлове за одит, конзола
--**Console Log Viewer**— Преглед в стил терминал в реално време с цветно кодирани нива, автоматично превъртане, търсене, филтър
--**SQLite Proxy Logs**— Постоянни регистрационни файлове, които оцеляват при рестартиране на сървъра
--**Translator Playground**— 4 режима за отстраняване на грешки: Playground (превод на формат), Chat Tester (обиколно пътуване), Test Bench (партида), Live Monitor (в реално време)
--**Заявка за телеметрия**— p50/p95/p99 латентност + проследяване на X-Request-Id
--**Регистриране на базата на файлове с ротация**— регистрационните файлове на приложението се редуват по размер, дни на съхранение и брой архиви; артефактите в регистъра на повикванията се редуват по дни на задържане и брой файлове
--**Отчет за системна информация**— `npm run system-info` генерира `system-info.txt` с вашата пълна среда (версия на възел, версия на OmniRoute, OS, CLI инструменти, състояние на Docker/PM2). Прикачете го, когато докладвате за проблеми за незабавно сортиране.
+
-<подробности>
-🏗️ 11. „Внедряването и поддържането на шлюза е сложно“
+
+📊 9. "I don't know how much I'm spending or where"
-Инсталирането, конфигурирането и поддържането на AI прокси в различни среди (локални, VPS, Docker, облак) е трудоемко. Проблеми като твърдо кодирани пътища, `EACCES` в директории, конфликти на портове и междуплатформени компилации добавят триене.
+Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up.
-**Как OmniRoute го решава:**
+**How OmniRoute solves it:**
--**npm global install**— `npm install -g omniroute && omniroute` — готово
--**Docker Multi-Platform**— роден AMD64 + ARM64 (Apple Silicon, AWS Graviton, Raspberry Pi)
--**Docker Compose Profiles**— `base` (без CLI инструменти) и `cli` (с Claude Code, Codex, OpenClaw)
--**Electron Desktop App**— родно приложение за Windows/macOS/Linux със системна област, автоматично стартиране, офлайн режим
--**Split-Port Mode**— API и табло за управление на отделни портове за разширени сценарии (обратен прокси, контейнерна мрежа)
--**Cloud Sync**— Конфигуриране на синхронизиране между устройства чрез Cloudflare Workers
--**DB Backups**— Автоматично архивиране, възстановяване, експортиране и импортиране на всички настройки, с `DISABLE_SQLITE_AUTO_BACKUP` за външно управлявани архиви
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider
+- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback
+- **Per-Model Pricing Configuration** — Configurable prices per model
+- **Usage Statistics Per API Key** — Request count and last-used timestamp per key
+- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency
-<подробности>
-🌍 12. „Интерфейсът е само на английски и екипът ми не говори английски“
+
-Екипите в неанглоговорящите страни, особено в Латинска Америка, Азия и Европа, се затрудняват с интерфейси само на английски. Езиковите бариери намаляват приемането и увеличават грешките в конфигурацията.
+
+🐛 10. "I can't diagnose errors and problems in AI calls"
-**Как OmniRoute го решава:**
+When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error.
--**Dashboard i18n — 30 езика**— Всички 500+ преведени клавиша, включително арабски, български, датски, немски, испански, фински, френски, иврит, хинди, унгарски, индонезийски, италиански, японски, корейски, малайски, холандски, норвежки, полски, португалски (PT/BR), румънски, руски, словашки, шведски, тайландски, украински, виетнамски, китайски, филипински, английски
--**RTL Support**— Поддръжка отдясно наляво за арабски и иврит
--**Многоезични READMEs**— 30 пълни превода на документация
--**Избор на език**— Икона на глобус в заглавката за превключване в реално време
+**How OmniRoute solves it:**
-<подробности>
-🔄 13. „Имам нужда от повече от чат — имам нужда от вграждания, изображения, аудио“
+- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console
+- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter
+- **SQLite Proxy Logs** — Persistent logs that survive server restarts
+- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time)
+- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing
+- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count
+- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage.
-AI не е просто завършване на чат. Разработчиците трябва да генерират изображения, да транскрибират аудио, да създават вграждания за RAG, да прекласират документи и да модерират съдържание. Всеки API има различна крайна точка и формат.
+
-**Как OmniRoute го решава:**
+
+🏗️ 11. "Deploying and maintaining the gateway is complex"
--**Вграждания**— `/v1/вграждания` с 6 доставчика и 9+ модела
--**Генериране на изображения**— `/v1/images/generations` с 10 доставчика и 20+ модела (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
--**Текст към видео**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) и SD WebUI
--**Текст към музика**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
--**Аудио транскрипция**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
--**Текст-към-говор**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + съществуващи доставчици
--**Модерации**— `/v1/moderations` — Проверки за безопасност на съдържанието
--**Прекласиране**— `/v1/rerank` — Прекласиране на уместността на документа
--**API за отговори**— Пълна поддръжка на `/v1/responses` за Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction.
-<подробности>
-🧪 14. „Нямам начин да тествам и сравнявам качеството между моделите“
+**How OmniRoute solves it:**
-Разработчиците искат да знаят кой модел е най-подходящ за техния случай на употреба – код, превод, разсъждения – но ръчното сравняване е бавно. Не съществуват интегрирани инструменти за оценка.
+- **npm global install** — `npm install -g omniroute && omniroute` — done
+- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi)
+- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw)
+- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode
+- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking)
+- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers
+- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups
-**Как OmniRoute го решава:**
+
--**Оценки на LLM**— Тестване със златен комплект с 10 предварително заредени случая, обхващащи поздрави, математика, география, генериране на код, съответствие с JSON, превод, маркдаун, отказ за безопасност
--**4 стратегии за съвпадение**— `exact`, `contains`, `regex`, `custom` (JS функция)
--**Translator Playground Test Bench**— Пакетно тестване с множество входове и очаквани изходи, сравнение между доставчици
--**Chat Tester**— Пълно двупосочно пътуване с визуално изобразяване на отговора
--**Монитор на живо**— Поток в реално време на всички заявки, преминаващи през проксито
+
+🌍 12. "The interface is English-only and my team doesn't speak English"
-<подробности>
-📈 15. „Трябва да мащабирам, без да губя производителност“
+Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors.
-Тъй като обемът на заявките нараства, без кеширане едни и същи въпроси генерират дублиращи се разходи. Без идемпотентност, дубликат иска обработка на отпадъци. Трябва да се спазват ограниченията за тарифите за всеки доставчик.
+**How OmniRoute solves it:**
-**Как OmniRoute го решава:**
+- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English
+- **RTL Support** — Right-to-left support for Arabic and Hebrew
+- **Multi-Language READMEs** — 30 complete documentation translations
+- **Language Selector** — Globe icon in header for real-time switching
--**Семантичен кеш**— Двуслоен кеш (подпис + семантичен) намалява разходите и забавянето
--**Request Idempotency**— 5s прозорец за дедупликация за идентични заявки
--**Rate Limit Detection**— RPM на доставчик, минимална разлика и максимално едновременно проследяване
--**Редактируеми ограничения на скоростта**— Конфигурируеми настройки по подразбиране в Настройки → Устойчивост с постоянство
--**API Key Validation Cache**— 3-степенен кеш за производствена производителност
--**Здравно табло с телеметрия**— p50/p95/p99 латентност, статистика на кеша, ъптайм
+
-<подробности>
-🤖 16. „Искам да контролирам поведението на модела глобално“
+
+🔄 13. "I need more than chat — I need embeddings, images, audio"
-Разработчици, които искат всички отговори на конкретен език, със специфичен тон или искат да ограничат токените за мотивиране. Конфигурирането на това във всеки инструмент/заявка е непрактично.
+AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format.
-**Как OmniRoute го решава:**
+**How OmniRoute solves it:**
--**Инжектиране на системна подкана**— Глобална подкана, приложена към всички заявки
--**Thinking Budget Validation**— Разсъждаващ контрол на разпределението на токени за всяка заявка (преминаване, автоматично, персонализирано, адаптивно)
--**9 стратегии за маршрутизиране**— Глобални стратегии, които определят как се разпределят заявките
--**Wildcard Router**— моделите `provider/*` маршрутизират динамично към всеки доставчик
--**Combo Enable/Disable Toggle**— Превключвайте комбинации директно от таблото за управление
--**Превключване на доставчика**— Активирайте/деактивирайте всички връзки за доставчик с едно щракване
--**Блокирани доставчици**— Изключете определени доставчици от списъка `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models
+- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
+- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI
+- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
+- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
+- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers
+- **Moderations** — `/v1/moderations` — Content safety checks
+- **Reranking** — `/v1/rerank` — Document relevance reranking
+- **Responses API** — Full `/v1/responses` support for Codex
-<подробности>
-🧰 17. „Имам нужда от MCP инструменти като първокласни продуктови възможности“
+
-Много AI шлюзове разкриват MCP само като скрит детайл за изпълнение. Екипите се нуждаят от видим, управляем оперативен слой.
+
+🧪 14. "I have no way to test and compare quality across models"
-**Как OmniRoute го решава:**
+Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist.
-- MCP се появява в раздела за навигация на таблото за управление и протокол на крайна точка
-- Специализирана страница за управление на MCP с процес, инструменти, обхвати и одит
-- Вграден бърз старт за `omniroute --mcp` и включване на клиента
+**How OmniRoute solves it:**
-<подробности>
-🧠 18. „Имам нужда от A2A оркестрация със синхронизиране + пътеки на задачи за поток“
+- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal
+- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function)
+- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison
+- **Chat Tester** — Full round-trip with visual response rendering
+- **Live Monitor** — Real-time stream of all requests flowing through the proxy
-Работните процеси на агентите се нуждаят както от директни отговори, така и от дълготрайно поточно изпълнение с контрол на жизнения цикъл.
+
-**Как OmniRoute го решава:**
+
+📈 15. "I need to scale without losing performance"
-- A2A JSON-RPC крайна точка (`POST /a2a`) с `message/send` и `message/stream`
-- SSE поточно предаване с разпространение на състоянието на терминала
-- API на жизнения цикъл на задачите за „tasks/get“ и „tasks/cancel“.
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected.
-<подробности>
-🛰️ 19. „Имам нужда от истинско състояние на MCP процес, а не от познат статус“
+**How OmniRoute solves it:**
-Оперативните екипи трябва да знаят дали MCP действително е жив, а не само дали API е достъпен.
+- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency
+- **Request Idempotency** — 5s deduplication window for identical requests
+- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking
+- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence
+- **API Key Validation Cache** — 3-tier cache for production performance
+- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime
-**Как OmniRoute го решава:**
+
-- Сърдечен файл по време на изпълнение с PID, времеви отпечатъци, транспорт, брой инструменти и режим на обхват
-- API за състояние на MCP, комбиниращ сърдечен ритъм + скорошна активност
-- Карти за състояние на потребителския интерфейс за свежест на процеса/време на работа/пулс
+
+🤖 16. "I want to control model behavior globally"
-<подробности>
-<резюме>📋 20. „Имам нужда от изпълнение на MCP инструмент с възможност за проверка“резюме>
+Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical.
-Когато инструментите променят конфигурацията или задействат оперативни действия, екипите се нуждаят от криминалистична проследимост.
+**How OmniRoute solves it:**
-**Как OmniRoute го решава:**
+- **System Prompt Injection** — Global prompt applied to all requests
+- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive)
+- **9 Routing Strategies** — Global strategies that determine how requests are distributed
+- **Wildcard Router** — `provider/*` patterns route dynamically to any provider
+- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard
+- **Provider Toggle** — Enable/disable all connections for a provider with one click
+- **Blocked Providers** — Exclude specific providers from `/v1/models` listing
-- Поддържано от SQLite одитно регистриране за извиквания на MCP инструмент
-- Филтрира по инструмент, успех/неуспех, API ключ и пагинация
-- Таблица за одит на таблото + статистически крайни точки за автоматизация
+
-<подробности>
-🔐 21. „Имам нужда от MCP разрешения с обхват за интеграция“
+
+🧰 17. "I need MCP tools as first-class product capabilities"
-Различните клиенти трябва да имат най-малко привилегирован достъп до категории инструменти.
+Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer.
-**Как OmniRoute го решава:**
+**How OmniRoute solves it:**
-- 10 гранулирани MCP обхвата за контролиран достъп до инструмента
-- Налагане на обхват и видимост в потребителския интерфейс за управление на MCP
-- Безопасна поза по подразбиране за оперативни инструменти
+- MCP appears in the dashboard navigation and endpoint protocol tab
+- Dedicated MCP management page with process, tools, scopes, and audit
+- Built-in quick-start for `omniroute --mcp` and client onboarding
-<подробности>
-⚙️ 22. „Имам нужда от оперативни контроли без пренасочване“
+
-Екипите се нуждаят от бързи промени във времето на изпълнение по време на инциденти или разходни събития.
+
+🧠 18. "I need A2A orchestration with sync + stream task paths"
-**Как OmniRoute го решава:**
+Agent workflows need both direct replies and long-running streamed execution with lifecycle control.
-- Превключете комбо активирането директно от таблото за управление на MCP
-- Прилагайте профили на устойчивост от предварително дефинирани пакети с правила
-- Нулирайте състоянието на прекъсвача от същия операционен панел
+**How OmniRoute solves it:**
-<подробности>
-🔄 23. „Имам нужда от видимост и анулиране на жизнения цикъл на задачите A2A на живо“
+- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream`
+- SSE streaming with terminal state propagation
+- Task lifecycle APIs for `tasks/get` and `tasks/cancel`
-Без видимост на жизнения цикъл инцидентите със задачи стават трудни за сортиране.
+
-**Как OmniRoute го решава:**
+
+🛰️ 19. "I need real MCP process health, not guessed status"
-- Списък със задачи/филтриране по състояние/умение с пагинация
-- Разбивка на метаданни, събития и артефакти на задачи
-- Крайна точка за анулиране на задача и действие на потребителския интерфейс с потвърждение
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable.
-<подробности>
-🌊 24. „Имам нужда от активни показатели на потока за A2A натоварване“
+**How OmniRoute solves it:**
-Поточните работни потоци изискват оперативно вникване в паралелността и живите връзки.
+- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode
+- MCP status API combining heartbeat + recent activity
+- UI status cards for process/uptime/heartbeat freshness
-**Как OmniRoute го решава:**
+
-- Броячи на активни потоци, интегрирани в статуса A2A
-- Времево клеймо на последната задача и брой на състоянието
-- A2A карти на таблото за наблюдение на операциите в реално време
+
+📋 20. "I need auditable MCP tool execution"
-<подробности>
-🪪 25. „Имам нужда от стандартно откриване на агент за клиенти“
+When tools mutate config or trigger ops actions, teams need forensic traceability.
-Външните клиенти и оркестраторите се нуждаят от машинночетими метаданни за включване.
+**How OmniRoute solves it:**
-**Как OmniRoute го решава:**
+- SQLite-backed audit logging for MCP tool calls
+- Filters by tool, success/failure, API key, and pagination
+- Dashboard audit table + stats endpoints for automation
-- Карта на агент, изложена в `/.well-known/agent.json`
-- Възможности и умения, показани в потребителския интерфейс за управление
-- API за състоянието на A2A включва метаданни за откриване за автоматизация
+
-<подробности>
-🧭 26. „Имам нужда от откриваемост на протокола в UX на продукта“
+
+🔐 21. "I need scoped MCP permissions per integration"
-Ако потребителите не могат да открият повърхности на протокола, качеството на приемане и поддръжка пада.
+Different clients should have least-privilege access to tool categories.
-**Как OmniRoute го решава:**
+**How OmniRoute solves it:**
-- Консолидирана страница**Крайни точки**с раздели за прокси, MCP, A2A и API крайни точки
-- Превключва състоянието на вградената услуга (онлайн/офлайн) за MCP и A2A
-- Връзки от преглед към специални раздели за управление
+- 10 granular MCP scopes for controlled tool access
+- Scope enforcement and visibility in MCP management UI
+- Safe default posture for operational tooling
-<подробности>
-🧪 27. „Имам нужда от валидиране на протокол от край до край с реални клиенти“
+
-Фалшивите тестове не са достатъчни за валидиране на съвместимостта на протокола преди пускане.
+
+⚙️ 22. "I need operational controls without redeploying"
-**Как OmniRoute го решава:**
+Teams need quick runtime changes during incidents or cost events.
-- E2E пакет, който зарежда приложение и използва реален MCP SDK клиентски транспорт
-- Клиент A2A тества за потоци откриване, изпращане, поточно предаване, получаване и отмяна
-- Кръстосана проверка на твърдения срещу MCP одит и API на A2A задачи
+**How OmniRoute solves it:**
-<подробности>
-📡 28. „Имам нужда от унифицирана видимост във всички интерфейси“
+- Switch combo activation directly from MCP dashboard
+- Apply resilience profiles from pre-defined policy packs
+- Reset circuit breaker state from the same operations panel
-Разделянето на наблюдаемостта по протокол създава слепи зони и по-дълъг MTTR.
+
-**Как OmniRoute го решава:**
+
+🔄 23. "I need live A2A task lifecycle visibility and cancellation"
-- Унифицирани табла за управление/логове/аналитика в един продукт
-- Здраве + одит + заявка за телеметрия в OpenAI, MCP и A2A слоеве
-- Оперативни API за статус и автоматизация
+Without lifecycle visibility, task incidents become hard to triage.
-<подробности>
-💼 29. „Имам нужда от една среда за изпълнение за прокси + инструменти + оркестрация на агенти“
+**How OmniRoute solves it:**
-Изпълнението на много отделни услуги увеличава оперативните разходи и режимите на отказ.
+- Task listing/filtering by state/skill with pagination
+- Drill-down on task metadata, events, and artifacts
+- Task cancellation endpoint and UI action with confirmation
-**Как OmniRoute го решава:**
+
-- OpenAI-съвместим прокси, MCP сървър и A2A сървър в един стек
-- Споделено удостоверяване, устойчивост, съхранение на данни и възможност за наблюдение
-- Последователен модел на политика във всички повърхности на взаимодействие
+
+🌊 24. "I need active stream metrics for A2A load"
-<подробности>
-🚀 30. „Трябва да изпращам агентски работни потоци без разрастване на лепен код“
+Streaming workflows require operational insight into concurrency and live connections.
-Екипите губят скорост, когато свързват множество ad-hoc услуги и скриптове.
+**How OmniRoute solves it:**
-**Как OmniRoute го решава:**
+- Active stream counters integrated into A2A status
+- Last task timestamp and per-state counts
+- A2A dashboard cards for real-time ops monitoring
-- Единна стратегия за крайни точки за клиенти и агенти
-- Вграден потребителски интерфейс за управление на протоколи и пътеки за проверка на дим
-- Готови за производство основи (сигурност, регистриране, устойчивост, архивиране)
+
+
+
+🪪 25. "I need standard agent discovery for clients"
+
+External clients and orchestrators need machine-readable metadata for onboarding.
+
+**How OmniRoute solves it:**
+
+- Agent Card exposed at `/.well-known/agent.json`
+- Capabilities and skills shown in management UI
+- A2A status API includes discovery metadata for automation
+
+
+
+
+🧭 26. "I need protocol discoverability in the product UX"
+
+If users cannot discover protocol surfaces, adoption and support quality drop.
+
+**How OmniRoute solves it:**
+
+- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints
+- Inline service status toggles (Online/Offline) for MCP and A2A
+- Links from overview to dedicated management tabs
+
+
+
+
+🧪 27. "I need end-to-end protocol validation with real clients"
+
+Mock tests are not enough to validate protocol compatibility before release.
+
+**How OmniRoute solves it:**
+
+- E2E suite that boots app and uses real MCP SDK client transport
+- A2A client tests for discovery, send, stream, get, and cancel flows
+- Cross-check assertions against MCP audit and A2A tasks APIs
+
+
+
+
+📡 28. "I need unified observability across all interfaces"
+
+Splitting observability by protocol creates blind spots and longer MTTR.
+
+**How OmniRoute solves it:**
+
+- Unified dashboards/logs/analytics in one product
+- Health + audit + request telemetry across OpenAI, MCP, and A2A layers
+- Operational APIs for status and automation
+
+
+
+
+💼 29. "I need one runtime for proxy + tools + agent orchestration"
+
+Running many separate services increases operational cost and failure modes.
+
+**How OmniRoute solves it:**
+
+- OpenAI-compatible proxy, MCP server, and A2A server in one stack
+- Shared auth, resilience, data store, and observability
+- Consistent policy model across all interaction surfaces
+
+
+
+
+🚀 30. "I need to ship agentic workflows without glue-code sprawl"
+
+Teams lose velocity when stitching multiple ad-hoc services and scripts.
+
+**How OmniRoute solves it:**
+
+- Unified endpoint strategy for clients and agents
+- Built-in protocol management UIs and smoke validation paths
+- Production-ready foundations (security, logging, resilience, backup)
+
+
### Example Playbooks (Integrated Use Cases)
-**Playbook A: Увеличете максимално платения абонамент + евтино архивиране**```txt
+**Playbook A: Maximize paid subscription + cheap backup**
+
+```txt
Combo: "maximize-claude"
1. cc/claude-opus-4-6
2. glm/glm-4.7
@@ -609,21 +690,23 @@ Combo: "maximize-claude"
Monthly cost: $20 + small backup spend
Outcome: higher quality, near-zero interruption
-````
+```
-**Playbook B: Стек за кодиране с нулеви разходи**```txt
+**Playbook B: Zero-cost coding stack**
+
+```txt
Combo: "free-forever"
-
-1. gc/gemini-3-flash
-2. if/kimi-k2-thinking
-3. qw/qwen3-coder-plus
+ 1. gc/gemini-3-flash
+ 2. if/kimi-k2-thinking
+ 3. qw/qwen3-coder-plus
Monthly cost: $0
Outcome: stable free coding workflow
+```
-````
+**Playbook C: 24/7 always-on fallback chain**
-**Playbook C: 24/7 винаги включена резервна верига**```txt
+```txt
Combo: "always-on"
1. cc/claude-opus-4-6
2. cx/gpt-5.2-codex
@@ -632,122 +715,134 @@ Combo: "always-on"
5. if/kimi-k2-thinking
Outcome: deep fallback depth for deadline-critical workloads
-````
+```
-**Playbook D: Операции на агент с MCP + A2A**```txt
+**Playbook D: Agent ops with MCP + A2A**
-1. Start MCP transport (`omniroute --mcp`) for tool-driven operations
-2. Run A2A tasks via `message/send` and `message/stream`
-3. Observe via /dashboard/endpoint (MCP and A2A tabs)
-4. Toggle services via inline status controls
-
-````
+```txt
+1) Start MCP transport (`omniroute --mcp`) for tool-driven operations
+2) Run A2A tasks via `message/send` and `message/stream`
+3) Observe via /dashboard/endpoint (MCP and A2A tabs)
+4) Toggle services via inline status controls
+```
---
## 🆓 Start Free — Zero Configuration Cost
-> Настройте AI кодиране за минути при**$0/месец**. Свържете тези безплатни акаунти и използвайте вградената комбинация**Free Stack**.
+> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo.
-| Стъпка | Действие | Отключени доставчици |
+| Step | Action | Providers Unlocked |
| ---- | -------------------------------------------------- | ------------------------------------------------------------------ |
-| 1 | Свържете**Kiro**(AWS Builder ID OAuth) | Клод Сонет 4.5, Хайку 4.5 —**неограничен**|
-| 2 | Свържете**Qoder**(Google OAuth) | kimi-k2-мислене, qwen3-coder-plus, deepseek-r1... —**неограничен**|
-| 3 | Свържете**Qwen**(Код на устройството) | qwen3-coder-plus, qwen3-coder-flash... —**неограничен**|
-| 4 | Свържете**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/мес безплатно**|
-| 5 | `/dashboard/combos` →**Безплатен стек ($0)**шаблон | Кръгово обвързване на всички безплатни доставчици автоматично |
+| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** |
+| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** |
+| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** |
+| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** |
+| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically |
-**Насочете всяка IDE/CLI към:**`http://localhost:20128/v1` · API ключ: `any-string` · Готово.
+**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done.
->**Допълнително покритие по избор (също безплатно):**Groq API ключ (30 RPM безплатно), NVIDIA NIM (40 RPM безплатно, 70+ модела), Cerebras (1M tok/ден), LongCat API ключ (50M tokens/ден!), Cloudflare Workers AI (10K Neurons/ден, 50+ модела).## Бърз старт
+> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models).
+
+## Бърз старт
### 1) Install and run
```bash
npm install -g omniroute
omniroute
-````
+```
-> **pnpm потребители:**Стартирайте `pnpm approve-builds -g` след инсталирането, за да активирате собствените скриптове за изграждане, изисквани от `better-sqlite3` и `@swc/core`:
+> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`:
>
-> ```баш
+> ```bash
> pnpm install -g omniroute
-> pnpm approve-builds -g # Изберете всички пакети → одобри
+> pnpm approve-builds -g # Select all packages → approve
> omniroute
> ```
-Таблото за управление се отваря на `http://localhost:20128`, а основният URL адрес на API е `http://localhost:20128/v1`.
+Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`.
-| Команда | Описание |
-| ----------------------- | ------------------------------------------------------------------------ |
-| `omniroute` | Стартов сървър (`PORT=20128`, API и таблото за управление на същия порт) |
-| `omniroute --порт 3000` | Задайте каноничен/API порт на 3000 |
-| `omniroute --mcp` | Стартирайте MCP сървър (stdio транспорт) |
-| `omniroute --no-open` | Без автоматично отваряне на браузъра |
-| `omniroute --help` | Показване на помощ |
+| Command | Description |
+| ----------------------- | ----------------------------------------------------------- |
+| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) |
+| `omniroute --port 3000` | Set canonical/API port to 3000 |
+| `omniroute --mcp` | Start MCP server (stdio transport) |
+| `omniroute --no-open` | Don't auto-open browser |
+| `omniroute --help` | Show help |
-Допълнителен режим на разделен порт:```bash
+Optional split-port mode:
+
+```bash
PORT=20128 DASHBOARD_PORT=20129 omniroute
-
-# API: http://localhost:20128/v1
-
+# API: http://localhost:20128/v1
# Dashboard: http://localhost:20129
-
-````
+```
### Long-Running Streaming Timeouts
-За повечето внедрявания се нуждаете само от:
+For most deployments, you only need:
-| Променлива | По подразбиране | Цел |
-| ------------------------ | ----------------------------- | -------------------------------------------------------------------------------------------------------------------------------- |
-| `REQUEST_TIMEOUT_MS` | `600000` | Споделена базова линия за извличане нагоре по веригата, скрити изчаквания на Undici, заявки за пръстови отпечатъци на TLS и изчаквания на заявка/прокси за мост на API |
-| `STREAM_IDLE_TIMEOUT_MS` | наследява `REQUEST_TIMEOUT_MS` | Максимална празнина между поточно предаване, преди OmniRoute да прекрати SSE потока |
+| Variable | Default | Purpose |
+| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- |
+| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts |
+| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream |
-Обратната съвместимост се запазва: съществуващите `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` и други променливи за изчакване на слой все още работят и заместват споделената базова линия.
+Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline.
-Налични са разширени настройки, ако имате нужда от по-фин контрол:| Променлива | По подразбиране | Цел |
-| ---------------------------------------------- | ---------------------------------------------- | -------------------------------------------------------------------- |
-| `FETCH_TIMEOUT_MS` | наследява `REQUEST_TIMEOUT_MS` | Общо време за изчакване на заявка нагоре по веригата, използвано от основния сигнал за прекъсване на извличането |
-| `FETCH_HEADERS_TIMEOUT_MS` | наследява `FETCH_TIMEOUT_MS` | Времево ограничение Undici за получаване на заглавки на отговор нагоре |
-| `FETCH_BODY_TIMEOUT_MS` | наследява `FETCH_TIMEOUT_MS` | Времево ограничение на Undici между частите на тялото нагоре (`0` го деактивира) |
-| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Време за изчакване на Undici TCP връзка |
-| `FETCH_KEEPALIVE_TIMEOUT_MS` | „4000“ | Времето за изчакване на сокета за неактивен поддържащ живот |
-| `TLS_CLIENT_TIMEOUT_MS` | наследява `FETCH_TIMEOUT_MS` | Време за изчакване за TLS заявки за пръстови отпечатъци, направени чрез `wreq-js` |
-| `API_BRIDGE_PROXY_TIMEOUT_MS` | наследява `REQUEST_TIMEOUT_MS` или `30000` | Време за изчакване за пренасочване на прокси `/v1` от API порт към порт на таблото |
-| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Време за изчакване на входящата заявка на мостовия сървър на API |
-| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Времето за изчакване на входящата заглавка на мостовия сървър на API |
-| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | „5000“ | Изчакване за поддържане на активност на мостовия сървър на API |
-| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | „0“ | Времето за изчакване на неактивност на сокета на мостовия сървър на API (`0` го деактивира) |
+Advanced overrides are available if you need finer control:
-Ако стартирате OmniRoute зад Nginx, Caddy, Cloudflare или друг обратен прокси, уверете се, че проксито
-таймаутите също са по-високи от вашите таймаути за поток/извличане на OmniRoute.### 2) Connect providers and create your API key
+| Variable | Default | Purpose |
+| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- |
+| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal |
+| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers |
+| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) |
+| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout |
+| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout |
+| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` |
+| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port |
+| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server |
+| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server |
+| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server |
+| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) |
-1. Отворете таблото за управление → `Доставчици` и свържете поне един доставчик (OAuth или API ключ).
-2. Отворете таблото за управление → `Крайни точки` и създайте API ключ.
-3. (По избор) Отворете таблото за управление → `Комбота` и задайте вашата резервна верига.### 3) Point your coding tool to OmniRoute
+If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy
+timeouts are also higher than your OmniRoute stream/fetch timeouts.
+
+### 2) Connect providers and create your API key
+
+1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key).
+2. Open Dashboard → `Endpoints` and create an API key.
+3. (Optional) Open Dashboard → `Combos` and set your fallback chain.
+
+### 3) Point your coding tool to OmniRoute
```txt
Base URL: http://localhost:20128/v1
API Key: [copy from Endpoint page]
Model: if/kimi-k2-thinking (or any provider/model prefix)
-````
+```
-Работи с Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode и OpenAI-съвместими SDK.### 4) Enable and validate protocols (v2.0)
+Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs.
-**MCP (за операции, управлявани от инструмент):**```bash
+### 4) Enable and validate protocols (v2.0)
+
+**MCP (for tool-driven operations):**
+
+```bash
omniroute --mcp
+```
-````
-
-След това свържете вашия MCP клиент през `stdio` и тествайте инструменти като:
+Then connect your MCP client over `stdio` and test tools like:
- `omniroute_get_health`
- `omniroute_list_combos`
-**A2A (за работни процеси от агент към агент):**```bash
+**A2A (for agent-to-agent workflows):**
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
```bash
curl -X POST http://localhost:20128/a2a \
@@ -761,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \
npm run test:protocols:e2e
```
-Този пакет валидира реални MCP и A2A клиентски потоци срещу работещо приложение.### Alternative: run from source
+This suite validates real MCP and A2A client flows against a running app.
+
+### Alternative: run from source
```bash
cp .env.example .env
@@ -769,14 +866,13 @@ npm install
PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev
```
-<подробности>
+
+Void Linux (`xbps-src` template)
-Void Linux (шаблон `xbps-src`)
-
-За потребители на Void Linux можете да изградите собствен пакет с помощта на `xbps-src`. Запазете този блок като `srcpkgs/omniroute/template`:```bash
+For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`:
+```bash
# Template file for 'omniroute'
-
pkgname=omniroute
version=3.4.1
revision=1
@@ -788,7 +884,7 @@ license="MIT"
homepage="https://github.com/diegosouzapw/OmniRoute"
distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz"
checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b
-system_accounts="\_omniroute"
+system_accounts="_omniroute"
omniroute_homedir="/var/lib/omniroute"
export NODE_ENV=production
export npm_config_engine_strict=false
@@ -796,71 +892,70 @@ export npm_config_loglevel=error
export npm_config_fund=false
export npm_config_audit=false
-do_build() { # Determine target CPU arch for node-gyp
-local \_gyp_arch
-case "$XBPS_TARGET_MACHINE" in
-aarch64*) \_gyp_arch=arm64 ;;
-armv7*|armv6*) \_gyp_arch=arm ;;
-i686*) \_gyp_arch=ia32 ;;
-\*) \_gyp_arch=x64 ;;
-esac
+do_build() {
+ # Determine target CPU arch for node-gyp
+ local _gyp_arch
+ case "$XBPS_TARGET_MACHINE" in
+ aarch64*) _gyp_arch=arm64 ;;
+ armv7*|armv6*) _gyp_arch=arm ;;
+ i686*) _gyp_arch=ia32 ;;
+ *) _gyp_arch=x64 ;;
+ esac
- # 1) Install all deps – skip scripts (no network in do_build, native modules
- # compiled separately below; better-sqlite3 is serverExternalPackage so
- # Next.js does not execute it during next build)
- NODE_ENV=development npm ci --ignore-scripts
+ # 1) Install all deps – skip scripts (no network in do_build, native modules
+ # compiled separately below; better-sqlite3 is serverExternalPackage so
+ # Next.js does not execute it during next build)
+ NODE_ENV=development npm ci --ignore-scripts
- # 2) Build the Next.js standalone bundle
- npm run build
+ # 2) Build the Next.js standalone bundle
+ npm run build
- # 3) Copy static assets into standalone
- cp -r .next/static .next/standalone/.next/static
- [ -d public ] && cp -r public .next/standalone/public || true
+ # 3) Copy static assets into standalone
+ cp -r .next/static .next/standalone/.next/static
+ [ -d public ] && cp -r public .next/standalone/public || true
- # 4) Compile better-sqlite3 native binding for the target architecture.
- # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
- # without npm altering them.
- local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
- (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
+ # 4) Compile better-sqlite3 native binding for the target architecture.
+ # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
+ # without npm altering them.
+ local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
+ (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
- # 5) Place the compiled binding into the standalone bundle
- local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
- mkdir -p "$_bs3_release"
- cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
+ # 5) Place the compiled binding into the standalone bundle
+ local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
+ mkdir -p "$_bs3_release"
+ cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
- # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
- # so sharp is not used at runtime; x64 .so files would break aarch64 strip
- rm -rf .next/standalone/node_modules/@img
-
- # 7) Copy pino runtime deps omitted by Next.js static analysis:
- # pino-abstract-transport – required by pino's worker thread
- # split2 – dep of pino-abstract-transport
- # process-warning – dep of pino itself
- for _mod in pino-abstract-transport split2 process-warning; do
- cp -r "node_modules/$_mod" .next/standalone/node_modules/
- done
+ # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
+ # so sharp is not used at runtime; x64 .so files would break aarch64 strip
+ rm -rf .next/standalone/node_modules/@img
+ # 7) Copy pino runtime deps omitted by Next.js static analysis:
+ # pino-abstract-transport – required by pino's worker thread
+ # split2 – dep of pino-abstract-transport
+ # process-warning – dep of pino itself
+ for _mod in pino-abstract-transport split2 process-warning; do
+ cp -r "node_modules/$_mod" .next/standalone/node_modules/
+ done
}
do_check() {
-npm run test:unit
+ npm run test:unit
}
do_install() {
-vmkdir usr/lib/omniroute/.next
+ vmkdir usr/lib/omniroute/.next
- vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
+ vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
- # Prevent removal of empty Next.js app router dirs by the post-install hook
- for _d in \
- .next/standalone/.next/server/app/dashboard \
- .next/standalone/.next/server/app/dashboard/settings \
- .next/standalone/.next/server/app/dashboard/providers; do
- touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
- done
-
- cat > "${WRKDIR}/omniroute" <<'EOF'
+ # Prevent removal of empty Next.js app router dirs by the post-install hook
+ for _d in \
+ .next/standalone/.next/server/app/dashboard \
+ .next/standalone/.next/server/app/dashboard/settings \
+ .next/standalone/.next/server/app/dashboard/providers; do
+ touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
+ done
+ cat > "${WRKDIR}/omniroute" <<'EOF'
#!/bin/sh
export PORT="${PORT:-20128}"
export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}"
@@ -872,10 +967,9 @@ EOF
}
post_install() {
-vlicense LICENSE
+ vlicense LICENSE
}
-
-````
+```
@@ -883,9 +977,11 @@ vlicense LICENSE
## 🐳 Docker
-OmniRoute е наличен като публично изображение на Docker в [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
+OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
-**Бързо бягане:**```bash
+**Quick run:**
+
+```bash
docker run -d \
--name omniroute \
--restart unless-stopped \
@@ -893,85 +989,96 @@ docker run -d \
-p 20128:20128 \
-v omniroute-data:/app/data \
diegosouzapw/omniroute:latest
-````
+```
-**С файл на средата:**```bash
+**With environment file:**
+```bash
# Copy and edit .env first
-
cp .env.example .env
docker run -d \
- --name omniroute \
- --restart unless-stopped \
- --stop-timeout 40 \
- --env-file .env \
- -p 20128:20128 \
- -v omniroute-data:/app/data \
- diegosouzapw/omniroute:latest
+ --name omniroute \
+ --restart unless-stopped \
+ --stop-timeout 40 \
+ --env-file .env \
+ -p 20128:20128 \
+ -v omniroute-data:/app/data \
+ diegosouzapw/omniroute:latest
+```
-````
+**Using Docker Compose:**
-**Използване на Docker Compose:**```bash
+```bash
# Base profile (no CLI tools)
docker compose --profile base up -d
# CLI profile (Claude Code, Codex, OpenClaw built-in)
docker compose --profile cli up -d
-````
+```
-Поддръжката на таблото за внедряване на Docker вече включва**Cloudflare Quick Tunnel**с едно щракване на `Табло → Крайни точки`. Първият активира изтеглянията `cloudflare` само когато е необходимо, стартира временен тунел към текущата ви крайна точка `/v1` и показва генерирания URL `https://*.trycloudflare.com/v1` директно под нормалния ви обществен URL адрес.
+Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL.
-Бележки:
+Notes:
-- URL адресите за бърз тунел са временни и се променят след всяко рестартиране.
-- Бързите тунели не се възстановяват автоматично след рестартиране на OmniRoute или контейнер. Активирайте ги отново от таблото за управление, когато е необходимо.
-- Управляваната инсталация в момента поддържа Linux, macOS и Windows на `x64` / `arm64`.
-- Управляваните бързи тунели по подразбиране са HTTP/2 транспорт, за да се избегнат шумни QUIC UDP буферни предупреждения в ограничени контейнерни среди. Задайте `CLOUDFLARED_PROTOCOL=quic` или `auto`, ако искате различен транспорт.
-- Изображенията на Docker обединяват системни CA корени и ги предават на управляван `cloudflared`, което избягва грешки в TLS доверието, когато тунелът стартира вътре в контейнера.
-- SQLite работи в режим WAL. `docker stop` трябва да бъде позволено да завърши, така че OmniRoute да може да провери последните промени обратно в `storage.sqlite`.
-- Пакетът Compose файлове вече задава гратисен период от 40 секунди. Ако стартирате изображението директно, запазете `--stop-timeout 40` (или подобно), така че ръчните спирания да не прекъсват почистването при изключване.
-- Задайте `CLOUDFLARED_BIN=/absolute/path/to/cloudflared`, ако искате OmniRoute да използва съществуващ двоичен файл, вместо да изтегля такъв.
+- Quick Tunnel URLs are temporary and change after every restart.
+- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed.
+- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`.
+- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport.
+- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container.
+- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`.
+- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup.
+- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one.
-**Използване на Docker Compose с Caddy (HTTPS Auto-TLS):**
+**Using Docker Compose with Caddy (HTTPS Auto-TLS):**
-OmniRoute може да бъде сигурно изложен чрез автоматичното SSL осигуряване на Caddy. Уверете се, че DNS A записът на вашия домейн сочи към IP адреса на вашия сървър.```yaml
+OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP.
+
+```yaml
services:
-omniroute:
-image: diegosouzapw/omniroute:latest
-container_name: omniroute
-restart: unless-stopped
-volumes: - omniroute-data:/app/data
-environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com
+ omniroute:
+ image: diegosouzapw/omniroute:latest
+ container_name: omniroute
+ restart: unless-stopped
+ volumes:
+ - omniroute-data:/app/data
+ environment:
+ - PORT=20128
+ - NEXT_PUBLIC_BASE_URL=https://your-domain.com
-caddy:
-image: caddy:latest
-container_name: caddy
-restart: unless-stopped
-ports: - "80:80" - "443:443"
-command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
+ caddy:
+ image: caddy:latest
+ container_name: caddy
+ restart: unless-stopped
+ ports:
+ - "80:80"
+ - "443:443"
+ command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
volumes:
-omniroute-data:
+ omniroute-data:
+```
-````
-
-| Изображение | Етикет | Размер | Описание |
+| Image | Tag | Size | Description |
| ------------------------ | -------- | ------ | --------------------- |
-| `diegosouzapw/omniroute` | `последно` | ~250MB | Най-новата стабилна версия |
-| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Текуща версия |---
+| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release |
+| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version |
+
+---
## 🖥️ Desktop App — Offline & Always-On
-> 🆕**НОВО!**OmniRoute вече е наличен като**стандартно настолно приложение**за Windows, macOS и Linux.
+> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux.
-Стартирайте OmniRoute като самостоятелно настолно приложение — без терминал, без браузър, без интернет, необходим за локалните модели. Базираното на Electron приложение включва:
+Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes:
-- 🖥️**Собствен прозорец**— Специален прозорец на приложението с интеграция в системната област
-- 🔄**Автоматично стартиране**— Стартирайте OmniRoute при влизане в системата
-- 🔔**Нативни известия**— Получавайте сигнали за изчерпване на квотата или проблеми с доставчика
-- ⚡**Инсталиране с едно кликване**— NSIS (Windows), DMG (macOS), AppImage (Linux)
-- 🌐**Офлайн режим**— Работи напълно офлайн с пакетния сървър### Бърз старт
+- 🖥️ **Native Window** — Dedicated app window with system tray integration
+- 🔄 **Auto-Start** — Launch OmniRoute on system login
+- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues
+- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux)
+- 🌐 **Offline Mode** — Works fully offline with bundled server
+
+### Бърз старт
```bash
# Development mode
@@ -982,308 +1089,370 @@ npm run electron:build # Current platform
npm run electron:build:win # Windows (.exe)
npm run electron:build:mac # macOS (.dmg) — x64 & arm64
npm run electron:build:linux # Linux (.AppImage)
-````
+```
### System Tray
-Когато е минимизиран, OmniRoute живее в системната област с бързи действия:
+When minimized, OmniRoute lives in your system tray with quick actions:
-- Отворете таблото
-- Промяна на сървърния порт
-- Излезте от приложението
+- Open dashboard
+- Change server port
+- Quit application
-📖 Пълна документация: [`electron/README.md`](electron/README.md)---
+📖 Full documentation: [`electron/README.md`](electron/README.md)
+
+---
## 💰 Pricing at a Glance
-| Ниво | Доставчик | Цена | Нулиране на квота | Най-добро за |
-| ------------------ | --------------------------- | -------------------------------- | ----------------------- | ----------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| **💳 АБОНАМЕНТ** | Claude Code (Pro) | $20/месец | 5 часа + седмично | Вече сте абонирани |
-| | Codex (Plus/Pro) | $20-200/месец | 5 часа + седмично | Потребители на OpenAI |
-| | Gemini CLI | **БЕЗПЛАТНО** | 180K/месец + 1K/ден | всички! |
-| | Копилот на GitHub | $10-19/месец | Месечно | Потребители на GitHub |
-| **🔑 КЛЮЧ ЗА API** | NVIDIA NIM | **БЕЗПЛАТНО**(dev forever) | ~40 RPM | 70+ отворени модела |
-| | Мозъци | **БЕЗПЛАТНО**(1M ток/ден) | 60K TPM / 30 RPM | Най-бързият в света |
-| | Groq | **БЕЗПЛАТНО**(30 RPM) | 14.4K RPD | Ултра-бърз Llama/Gemma |
-| | DeepSeek V3.2 | $0,27/$1,10 за 1M | Няма | Обосновка за най-добра цена/качество |
-| | xAI Grok-4 Бърз | **$0,20/$0,50 за 1M**🆕 | Няма | Най-бързо + извикване на инструмент, ултраниско |
-| | xAI Grok-4 (стандартен) | $0,20/$1,50 за 1M 🆕 | Няма | Разсъждаващ флагман от xAI |
-| | Мистрал | Безплатен пробен период + платен | Ограничена скорост | Европейски AI |
-| | OpenRouter | Плащане при използване | Няма | 100+ модела агр. |
-| **💰 ЕВТИНО** | GLM-5 (чрез Z.AI) 🆕 | $0,5/1 милион | Ежедневно 10 сутринта | 128K изход, най-новият флагман |
-| | GLM-4.7 | $0,6/1 милион | Ежедневно 10 сутринта | Резервно копие на бюджета |
-| | MiniMax M2.5 🆕 | $0,3/1M вход | 5-часово търкаляне | Разсъждение + агентски задачи |
-| | MiniMax M2.1 | $0,2/1 милион | 5-часово търкаляне | Най-евтиният вариант |
-| | Kimi K2.5 (Moonshot API) 🆕 | Плащане при използване | Няма | Директен достъп до API на Moonshot |
-| | Кими К2 | $9/месец апартамент | 10 милиона токена/месец | Предвидими разходи |
-| **🆓 БЕЗПЛАТНО** | Qoder | **$0** | Неограничен | 5 модела неограничено |
-| | Куен | **$0** | Неограничен | 4 модела неограничено |
-| | Киро | **$0** | Неограничен | Клод Сонет/Хайку (AWS Builder) |
-| | LongCat Flash-Lite 🆕 | **$0**(50M ток/ден 🔥) | 1 RPS | Най-голямата безплатна квота на Земята |
-| | Опрашвания AI 🆕 | **$0**(не е необходим ключ) | 1 изискване/15s | GPT-5, Claude, DeepSeek, Llama 4 |
-| | Cloudflare Workers AI 🆕 | **$0**(10K неврони/ден) | ~150 повторения/ден | 50+ модела, глобално предимство |
-| | Scaleway AI 🆕 | **$0**(общо 1 милион токена) | Ограничена скорост | ЕС/GDPR, Qwen3 235B, Llama 70B | > 🆕**Добавени нови модели (март 2026 г.):**Grok-4 Fast семейство на $0,20/$0,50/M (бенчмарк на 1143ms — 30% по-бързо от Gemini 2.5 Flash), GLM-5 чрез Z.AI с 128K изход, MiniMax M2.5 разсъждения, DeepSeek V3.2 актуализирани цени, Kimi K2.5 чрез Moonshot direct API. |
+| Tier | Provider | Cost | Quota Reset | Best For |
+| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- |
+| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed |
+| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users |
+| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! |
+| | GitHub Copilot | $10-19/mo | Monthly | GitHub users |
+| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models |
+| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest |
+| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma |
+| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning |
+| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow |
+| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI |
+| | Mistral | Free trial + paid | Rate limited | European AI |
+| | OpenRouter | Pay-per-use | None | 100+ models aggr. |
+| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship |
+| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup |
+| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks |
+| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option |
+| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access |
+| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost |
+| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited |
+| | Qwen | **$0** | Unlimited | 4 models unlimited |
+| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) |
+| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth |
+| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 |
+| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge |
+| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B |
-**💡 $0 Combo Stack — Пълната безплатна настройка:**```
+> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API.
+**💡 $0 Combo Stack — The Complete Free Setup:**
+
+```
# 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever
+Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
+Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
+Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
+Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
+NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+```
-Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
-Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
-LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
-Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
-Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
-Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
-Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
-Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
-Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
-NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
-Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever.
-````
-
-**Нулев разход. Никога не спира кодирането.**Конфигурирайте това като едно OmniRoute комбо и всички резервни варианти се случват автоматично – без ръчно превключване.---
+---
---
## 🆓 Free Models — What You Actually Get
-> Всички модели по-долу са**100% безплатни без изискване за кредитна карта**. OmniRoute автоматично пренасочва между тях, когато една квота изтече — комбинирайте ги всички за неразбиваема комбинация от $0.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
+> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo.
-| Модел | Префикс | Лимит | Ограничение на скоростта |
+### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------- | ------ | ------------- | --------------------- |
-| `claude-sonnet-4.5` | `kr/` |**Неограничен**| Няма отчетено дневно ограничение |
-| `claude-haiku-4.5` | `kr/` |**Неограничен**| Няма отчетено дневно ограничение |
-| `claude-opus-4.6` | `kr/` |**Неограничен**| Най-новият Opus чрез Kiro |### 🟢 QODER MODELS (Free PAT via qodercli)
+| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro |
-| Модел | Префикс | Лимит | Ограничение на скоростта |
+### 🟢 QODER MODELS (Free PAT via qodercli)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------ | ------ | ------------- | --------------- |
-| `kimi-k2-мислене` | `ако/` |**Неограничен**| Няма отчетено ограничение |
-| `qwen3-coder-plus` | `ако/` |**Неограничен**| Няма отчетено ограничение |
-| `deepseek-r1` | `ако/` |**Неограничен**| Няма отчетено ограничение |
-| `минимакс-m2.1` | `ако/` |**Неограничен**| Няма отчетено ограничение |
-| `kimi-k2` | `ако/` |**Неограничен**| Няма отчетено ограничение |
+| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap |
+| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap |
+| `deepseek-r1` | `if/` | **Unlimited** | No reported cap |
+| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap |
+| `kimi-k2` | `if/` | **Unlimited** | No reported cap |
-> Препоръчителен метод за свързване:**Personal Access Token + `qodercli`**. OAuth на браузъра е
-> експериментален и деактивиран по подразбиране, освен ако не са конфигурирани променливи на средата `QODER_OAUTH_*`.### 🟡 QWEN MODELS (Device Code Auth)
+> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is
+> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured.
-| Модел | Префикс | Лимит | Ограничение на скоростта |
+### 🟡 QWEN MODELS (Device Code Auth)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------- | ------ | ------------- | ------------------- |
-| `qwen3-coder-plus` | `qw/` |**Неограничен**| Няма отчетено ограничение |
-| `qwen3-coder-flash` | `qw/` |**Неограничен**| Няма отчетено ограничение |
-| `qwen3-coder-next` | `qw/` |**Неограничен**| Няма отчетено ограничение |
-| `модел-визия` | `qw/` |**Неограничен**| Мултимодални (изображения) |### 🟣 GEMINI CLI (Google OAuth)
+| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap |
+| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) |
-| Модел | Префикс | Лимит | Ограничение на скоростта |
-| ------------------------ | ------ | ---------------------------- | ------------- |
-| `gemini-3-flash-preview` | `gc/` |**180K tok/месец**+ 1K/ден | Месечно нулиране |
-| `gemini-2.5-pro` | `gc/` | 180K/месец (споделен басейн) | Високо качество |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
+### 🟣 GEMINI CLI (Google OAuth)
-| Ниво | Дневен лимит | Ограничение на скоростта | Бележки |
-| ---------- | ------------ | ----------- | ----------------------------------------------------- |
-| Безплатно (Dev) | Без ограничение на токена |**~40 RPM**| 70+ модела; преминаване към чисти лимити на лихвите в средата на 2025 г. |
+| Model | Prefix | Limit | Rate Limit |
+| ------------------------ | ------ | --------------------------- | ------------- |
+| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset |
+| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality |
-Популярни безплатни модели: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
+### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
-| Ниво | Дневен лимит | Ограничение на скоростта | Бележки |
-| ---- | ----------------- | ---------------- | -------------------------------------------- |
-| Безплатно |**1 милион токена/ден**| 60K TPM / 30 RPM | Най-бързият LLM извод в света; нулира ежедневно |
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---------- | ------------ | ----------- | ------------------------------------------------------ |
+| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 |
-Предлага се безплатно: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com)
+Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`
-| Ниво | Дневен лимит | Ограничение на скоростта | Бележки |
-| ---- | ------------- | ---------------- | ---------------------------------------------- |
-| Безплатно |**14,4K RPD**| 30 RPM за модел | Без кредитна карта; 429 на лимит, не се таксува |
+### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
-Предлага се безплатно: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---- | ----------------- | ---------------- | ------------------------------------------- |
+| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily |
-| Модел | Префикс | Дневна безплатна квота | Бележки |
+Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`
+
+### 🔴 GROQ (Free API Key — console.groq.com)
+
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---- | ------------- | ---------------- | ----------------------------------------- |
+| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged |
+
+Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`
+
+### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+
+| Model | Prefix | Daily Free Quota | Notes |
| ----------------------------- | ------ | ----------------- | ----------------------- |
-| `LongCat-Flash-Lite` | `lc/` |**50 милиона токена**💥 | Най-голямата безплатна квота досега |
-| `LongCat-Flash-Chat` | `lc/` | 500K токена | Многооборотен чат |
-| `LongCat-Flash-Thinking` | `lc/` | 500K токена | Разсъждения / CoT |
-| `LongCat-Flash-Thinking-2601` | `lc/` | 500K токена | Версия от януари 2026 г. |
-| `LongCat-Flash-Omni-2603` | `lc/` | 500K токена | Мултимодален |
+| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever |
+| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat |
+| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT |
+| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version |
+| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal |
-> 100% безплатно, докато сте в публична бета версия. Регистрирайте се в [longcat.chat](https://longcat.chat) с имейл или телефон. Нулира всеки ден в 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC.
-| Модел | Префикс | Ограничение на скоростта | Доставчик зад |
+### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+
+| Model | Prefix | Rate Limit | Provider Behind |
| ---------- | ------ | ---------- | ------------------ |
-| `опенай` | `pol/` | 1 изискване/15s | GPT-5 |
-| `клод` | `pol/` | 1 изискване/15s | Антропичен Клод |
-| `близнаци` | `pol/` | 1 изискване/15s | Google Gemini |
-| `deepseek` | `pol/` | 1 изискване/15s | DeepSeek V3 |
-| `лама` | `pol/` | 1 изискване/15s | Мета Лама 4 Скаут |
-| `мистрал` | `pol/` | 1 изискване/15s | Мистрал AI |
+| `openai` | `pol/` | 1 req/15s | GPT-5 |
+| `claude` | `pol/` | 1 req/15s | Anthropic Claude |
+| `gemini` | `pol/` | 1 req/15s | Google Gemini |
+| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 |
+| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout |
+| `mistral` | `pol/` | 1 req/15s | Mistral AI |
-> ✨**Нулево триене:**Без регистрация, без API ключ. Добавете доставчика на Опрашвания с празно поле за ключ и той работи веднага.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately.
-| Ниво | Ежедневни неврони | Еквивалентно използване | Бележки |
-| ---- | ------------- | ----------------------------------------------- | ----------------------- |
-| Безплатно |**10 000**| ~150 LLM resp / 500s аудио / 15K вграждания | Global edge, 50+ модела |
+### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
-Популярни безплатни модели: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (безплатно аудио!), `@cf/qwen/qwen2.5-coder-15b-instruct`
+| Tier | Daily Neurons | Equivalent Usage | Notes |
+| ---- | ------------- | --------------------------------------- | ----------------------- |
+| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models |
-> Изисква API Token + ID на акаунт от [dash.cloudflare.com](https://dash.cloudflare.com). Съхранявайте ID на акаунта в настройките на доставчика.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct`
-| Ниво | Безплатна квота | Местоположение | Бележки |
-| ---- | ------------- | ------------ | ---------------------------------- |
-| Безплатно |**1M токени**| 🇫🇷 Париж, ЕС | Не е необходима кредитна карта в рамките на лимити |
+> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings.
-Предлага се безплатно: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
+### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
-> Съвместим с ЕС/GDPR. Вземете API ключ на [console.scaleway.com](https://console.scaleway.com).
+| Tier | Free Quota | Location | Notes |
+| ---- | ------------- | ------------ | ----------------------------------- |
+| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits |
->**💡 Най-добрият безплатен стек (11 доставчици, $0 завинаги):**
+Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
+
+> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com).
+
+> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):**
>
> ```
-> Киро (kr/) → Клод Сонет/Хайку НЕОГРАНИЧЕНО
-> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 НЕОГРАНИЧЕНО
-> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 милиона токена/ден 🔥
-> Опрашвания (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — не е необходим ключ
-> Qwen (qw/) → qwen3-кодер модели НЕОГРАНИЧЕНИ
-> Gemini (gemini/) → Gemini 2.5 Flash — 1500 req/ден безплатно
-> Cloudflare AI (cf/) → 50+ модела — 10K неврони/ден
-> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M безплатни токени (ЕС)
-> Groq (groq/) → Llama/Gemma — 14.4K req/ден ултра-бърз
-> NVIDIA NIM (nvidia/) → 70+ отворени модела — 40 RPM завинаги
-> Cerebras (cerebras/) → Llama/Qwen най-бързият в света — 1M ток/ден
-> ```## 🎙️ Free Transcription Combo
+> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+> Qwen (qw/) → qwen3-coder models UNLIMITED
+> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free
+> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day
+> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast
+> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+> ```
-> Транскрибирайте всяко аудио/видео за**$0**— Deepgram води с $200 безплатно, AssemblyAI $50 резервен вариант, Groq Whisper като неограничено аварийно архивиране.
+## 🎙️ Free Transcription Combo
-| Доставчик | Безплатни кредити | Най-добър модел | Ограничение на скоростта |
+> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup.
+
+| Provider | Free Credits | Best Model | Rate Limit |
| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- |
-| 🟢**Deepgram**|**$200 безплатно**(регистрация) | `nova-3` — най-добра точност, 30+ езика | Без ограничение на RPM за безплатни кредити |
-| 🔵**AssemblyAI**|**$50 безплатно**(регистрация) | `universal-3-pro` — глави, настроение, PII | Без ограничение на RPM за безплатни кредити |
-| 🔴**Groq**|**Безплатно завинаги**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (ограничена скорост) |
+| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits |
+| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits |
+| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) |
-**Предложена комбинация в `/dashboard/combos`:**```
+**Suggested combo in `/dashboard/combos`:**
+
+```
Name: free-transcription
Strategy: Priority
Nodes:
[1] deepgram/nova-3 → uses $200 free first
[2] assemblyai/universal-3-pro → fallback when Deepgram credits run out
[3] groq/whisper-large-v3 → free forever, emergency fallback
-````
+```
-След това в `/dashboard/media` → раздел**Транскрипция**: качете произволен аудио или видео файл → изберете вашата комбинирана крайна точка → получете транскрипция в поддържани формати.## 💡 Key Features
+Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats.
-OmniRoute v2.0 е създаден като операционна платформа, а не просто релейно прокси.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
+## 💡 Key Features
-| Характеристика | Какво прави |
-| ---------------------------------------------------- | --------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ |
-| ⚡**Grok-4 Fast Family** | xAI модели при $0,20/$0,50/M — сравнително време 1143ms (30% по-бързо от Gemini 2.5 Flash) |
-| 🧠**GLM-5 чрез Z.AI** | 128K изходен контекст, $0,5/1M — най-новият флагман от семейството GLM |
-| 🔮**MiniMax M2.5** | Разсъждение + агентски задачи при $0,30/1M — значително надграждане от M2.1 |
-| 🎯**флаг за извикване на инструмент за модел** | `toolCalling: true/false` за модел в системния регистър — AutoCombo пропуска модели без инструмент |
-| 🌍**Откриване на многоезични намерения** | PT/ZH/ES/AR ключови думи в точкуването на AutoCombo — по-добър избор на модел за неанглийско съдържание |
-| 📊**Резервни резултати, управлявани от бенчмаркове** | Реална p95 латентност от живи заявки емисии комбо точкуване — AutoCombo се учи от действителни данни |
-| 🔁**Искайте дедупликация** | Прозорец за дедупиране, базиран на хеш съдържание — безопасен за много агенти, предотвратява дублиране на такси |
-| 🔌**Pluggable RouterStrategy** | Разширяем интерфейс `RouterStrategy` — добавете персонализирана логика за маршрутизиране като добавки | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP |
+OmniRoute v3.5 is built as an operational platform, not just a relay proxy.
-| Характеристика | Какво прави |
-| ------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |
-| 🎮**Моделна площадка** | Страница на таблото за директно тестване на всеки модел — селектори на доставчик/модел/крайна точка, Monaco Editor, стрийминг, прекъсване, време |
-| 🔏**CLI съпоставяне на пръстови отпечатъци** | Подреждане на заглавка/тяло на доставчик, за да съответства на оригиналните CLI подписи — превключете за доставчик в Настройки > Сигурност.**Вашият прокси IP е запазен** |
-| 🤝**Поддръжка на ACP (клиентски протокол на агент)** | Откриване на агент на CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + още 9), генериращ процес, крайна точка `/api/acp/agents` |
-| 🤖**Табло за управление на ACP агенти** | Страница за отстраняване на грешки › Агенти — мрежа от 14 агента със статус на инсталиране, версия, персонализирана форма на агент за всеки CLI инструмент. Потребителите на**OpenCode**получават бутон „Изтегляне на opencode.json“, който автоматично генерира готова за използване конфигурация с всички налични модели. |
-| 🔧**Маршрутизиране на потребителски модел `apiFormat`** | Персонализираните модели с `apiFormat: "responses"` вече насочват правилно към преводача на API за отговори |
-| 🏢**Изолация на работното пространство на Codex** | Множество работни пространства на Codex на имейл — OAuth правилно разделя връзките по ID на работното пространство |
-| 🔄**Електронно автоматично актуализиране** | Настолното приложение проверява за актуализации + автоматично инсталиране при рестартиране | ### 🤖 Agent & Protocol Operations (v2.0) |
+### 🆕 New — v3.5.5 Highlights (Apr 2026)
-| Характеристика | Какво прави |
-| --------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- |
-| 🔧**MCP сървър (25 инструмента)** | Инструменти за IDE/агент чрез 3 транспорта: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 ядра + 3 памет + 4 инструмента за умения |
-| 🤝**A2A сървър (JSON-RPC + SSE)** | Изпълнение на задачи от агент към агент със синхронизиране и поточно предаване |
-| 🧭**Страница с консолидирани крайни точки** | Страница за управление с раздели с раздели Endpoint Proxy, MCP, A2A и API Endpoints |
-| 🎚️**Превключватели за активиране/деактивиране на услуги** | Превключватели за ВКЛ./ИЗКЛ. за MCP и A2A с постоянни настройки (по подразбиране: ИЗКЛ.) |
-| 🛰️**MCP Runtime Heartbeat** | Реално състояние на процеса (pid, време на работа, възраст на сърдечния ритъм, транспорт, режим на обхвата) |
-| 📋**MCP одитна пътека** | Филтрируеми журнали за одит с успех/неуспех и ключово приписване |
-| 🔐**Прилагане на обхват на MCP** | 10 подробни разрешения за обхват за контролиран достъп до инструменти |
-| 📡**A2A Управление на жизнения цикъл на задачите** | Списък/филтриране на задачи, проверка на събития/артефакти, отмяна на изпълнявани задачи |
-| 📋**Откриване на карта на агент** | `/.well-known/agent.json` за автоматично откриване на клиенти |
-| 🧪**Протокол E2E Тестова система** | Истински MCP SDK + A2A клиент протича в `test:protocols:e2e` |
-| ⚙️**Оперативни контроли** | Превключете комбо, приложете профили на устойчивост, нулирайте прекъсвачите от една контролна повърхност | ### 🧠 Routing & Intelligence |
+| Feature | What It Does |
+| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation |
+| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs |
+| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions |
+| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection |
+| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts |
-| Характеристика | Какво прави |
-| ---------------------------------------------------------- | ---------------------------------------------------------------------------------------------- | ----------------------- |
-| 🎯**Интелигентен 4-степенен резервен вариант** | Автоматичен маршрут: Абонамент → API ключ → Евтини → Безплатно |
-| 📊**Проследяване на квоти в реално време** | Брой токени на живо + нулиране на обратното броене на доставчик |
-| 🔄**Форматиране на превода** | OpenAI ↔ Claude ↔ Gemini ↔ Отговори с безопасни за схема преобразувания |
-| 👥**Поддръжка за множество акаунти** | Няколко акаунта на доставчик с интелигентен избор |
-| 🔄**Автоматично опресняване на токени** | OAuth токените се опресняват автоматично с повторен опит |
-| 🎨**Персонализирани комбинации** | 9 стратегии за балансиране + резервен контрол на веригата |
-| 🌐**Wildcard Router** | `провайдер/*` динамично маршрутизиране |
-| 🧠**Мислене за контрол на бюджета** | Лимити за преминаване, автоматични, персонализирани и адаптивни разсъждения |
-| 🔀**Псевдоними на модели** | Вграден + персонализиран псевдоним на модела и безопасност на миграцията |
-| ⚡**Влошаване на фона** | Насочване на фонови задачи с нисък приоритет към по-евтини модели |
-| 🧪**Интелигентно маршрутизиране, съобразено със задачите** | Автоматичен избор на модел по тип съдържание (кодиране/визия/анализ/обобщение) |
-| 🔄**A2A Agent Workflows** | Детерминиран оркестратор на FSM за изпълнения на многоетапни агенти със състояние |
-| 🔀**Адаптивно маршрутизиране** | Динамична отмяна на стратегия въз основа на обема на токена и сложността на подканата |
-| 🎲**Разнообразие от доставчици** | Оценка на ентропията на Шанън за балансиране на разпределението на трафика с автоматично комбо |
-| 💬**Системно бързо инжектиране** | Глобални контроли на поведението, прилагани последователно |
-| 📄**Съвместимост с API за отговори** | Пълна поддръжка на `/v1/responses` за Codex и разширени агентни работни потоци | ### 🎵 Multi-Modal APIs |
+### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
-| Характеристика | Какво прави |
-| ------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- |
-| 🖼️**Генериране на изображения** | `/v1/images/generations` с облачен и локален бекенд |
-| 📐**Вграждания** | `/v1/embeddings` за търсене и RAG тръбопроводи |
-| 🎤**Аудио транскрипция** | `/v1/audio/transcriptions` — 7 доставчика (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), автоматично откриване на език, поддръжка на MP4/MP3/WAV |
-| 🔊**Текст към говор** | `/v1/audio/speech` — 10 доставчика (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) с правилни съобщения за грешки |
-| 🎬**Видео генериране** | `/v1/videos/generations` (работни процеси ComfyUI + SD WebUI) |
-| 🎵**Музикално поколение** | `/v1/music/generations` (работни процеси на ComfyUI) |
-| 🛡️**Модерации** | `/v1/moderations` проверки за безопасност |
-| 🔀**Прекласиране** | `/v1/rerank` за оценка на уместността |
-| 🔍**Търсене в мрежата**🆕 | `/v1/търсене` — 5 доставчика (Serper, Brave, Perplexity, Exa, Tavily), 6 500+ безплатно/месец, автоматичен отказ, кеш | ### 🛡️ Resilience, Security & Governance |
+| Feature | What It Does |
+| ------------------------------------ | ------------------------------------------------------------------------------------------- |
+| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) |
+| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family |
+| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 |
+| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models |
+| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content |
+| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data |
+| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges |
+| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins |
-| Характеристика | Какво прави |
-| --------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------- | -------------------------------- |
-| 🔌**Прекъсвачи** | За всеки модел пътуване/възстановяване с прагови контроли |
-| 🎯**Модели, съобразени с крайни точки** | Персонализираните модели декларират поддържани крайни точки + API формат |
-| 🛡️**Anti-Thundering Herd** | Защита на Mutex + семафор при събития за повторен опит/скорост |
-| 🧠**Семантичен + кеш на подписа** | Намаляване на разходите/закъснението с два кеш слоя |
-| ⚡**Искане на идемпотентност** | Дублиран защитен прозорец |
-| 🔒**TLS Fingerprint Spoofing** | Подобен на браузър TLS отпечатък —**намалява откриването на ботове и маркирането на акаунта** |
-| 🔏**CLI съпоставяне на пръстови отпечатъци** | Съвпада със собствените подписи на CLI заявка —**намалява риска от забрана, като същевременно запазва IP на проксито** |
-| 🌐**IP филтриране** | Списък с разрешени/списъци с блокирани контроли за открити внедрявания |
-| 📊**Редактируеми ограничения на скоростта** | Конфигурируеми глобални/на ниво доставчик ограничения с постоянство |
-| 📉**Изящна деградация** | Резервни възможности за многослойни възможности, защитаващи основните операции на шлюза |
-| 📜**Пътека за одит на конфигурация** | Проследяване на промяна, базирано на разлика, предотвратяващо оперативно отклонение с прости връщания |
-| ⏳**Синхронизиране на здравето на доставчика** | Проактивен мониторинг на изтичането на токена, задействащ предупреждения преди неуспешно оторизиране |
-| 🚪**Автоматично деактивиране на забранени акаунти** | Оперативен прекъсвач автоматично запечатва трайно блокирани токен акаунти |
-| 🔑**API Key Management + Scoping** | Сигурно издаване/ротация на ключове и контроли на модел/доставчик |
-| 👁️**Разкриване на API ключ с обхват**🆕 | Възстановяване с включване на API ключове чрез `ALLOW_API_KEY_REVEAL` |
-| 🛡️**Защитени `/models`** | Опционално удостоверяване и скриване на доставчик за каталог на модели | ### 📊 Observability & Analytics |
+### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP
-| Характеристика | Какво прави |
-| --------------------------------------------------------- | --------------------------------------------------------------------------- | ---------------------------- |
-| 📝**Заявка + Регистриране на прокси сървър** | Пълно регистриране на заявка/отговор и прокси |
-| 📉**Поточно предавани подробни регистрационни файлове**🆕 | Реконструира SSE потоците от полезен товар чисто в потребителския интерфейс |
-| 📋**Табло за управление на Unified Logs** | Изгледи на заявка, прокси, одит и конзола на една страница |
-| 🔍**Заявка за телеметрия** | p50/p95/p99 латентност и проследяване на заявки |
-| 🏥**Здравно табло** | Време на работа, състояния на прекъсване, блокировки, статистика на кеша |
-| 💰**Проследяване на разходите** | Контрол на бюджета и видимост на ценообразуването за модел |
-| 📈**Аналитични визуализации** | Прозрения за използването на модел/доставчик и изгледи на тенденции |
-| 🧪**Рамка за оценка** | Тестване на златен набор с конфигурируеми стратегии за мач |
-| 📡**Диагностика на живо**🆕 | Семантичен байпас на кеша за точно комбинирано тестване на живо | ### ☁️ Deployment & Platform |
+| Feature | What It Does |
+| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing |
+| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** |
+| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint |
+| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. |
+| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator |
+| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID |
+| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart |
-| Характеристика | Какво прави |
-| ----------------------------------------- | ------------------------------------------------------------------------------ | --------------------- |
-| 🌐**Разполагане навсякъде** | Localhost, VPS, Docker, облачни среди |
-| 🚇**Cloudflare Tunnel**🆕 | Интеграция с бърз тунел с едно щракване от таблото за управление |
-| 🔑**Филтриране на ключови модели на API** | Роден /v1/models отговор, филтриран чрез присвоени контекстни роли на носителя |
-| ⚡**Smart Cache Bypass** | Конфигурируеми TTL евристики и контроли за принудително повторно извличане |
-| 🔄**Архивиране/Възстановяване** | Експорт/импорт и потоци за възстановяване след бедствие |
-| 🧙**Съветник за присъединяване** | Насочвана настройка при първо стартиране |
-| 🔧**CLI Tools Dashboard** | Настройка с едно щракване за популярни инструменти за кодиране |
-| 🎮**Моделна площадка** | Тествайте всеки доставчик/модел/крайна точка от таблото |
-| 🔏**CLI Fingerprint Toggle** | Съвпадение на пръстови отпечатъци за всеки доставчик в Настройки > Сигурност |
-| 🌐**i18n (30 езика)** | Пълно табло за управление + езикова поддръжка на документи с RTL покритие |
-| 🧹**Изчистване на всички модели** | Изчистване на списък с модели с едно щракване в подробности за доставчика |
-| 👁️**Контроли на страничната лента**🆕 | Скриване на компоненти и интеграции от Настройки на външния вид |
-| 📋**Шаблони за проблеми** | Стандартизирани GitHub шаблони за грешки и функции |
-| 📂**Директория с персонализирани данни** | Замяна на `DATA_DIR` за място за съхранение | ### Feature Deep Dive |
+### 🤖 Agent & Protocol Operations (v2.0)
+
+| Feature | What It Does |
+| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
+| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools |
+| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows |
+| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs |
+| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) |
+| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) |
+| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution |
+| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access |
+| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks |
+| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery |
+| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` |
+| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface |
+
+### 🧠 Routing & Intelligence
+
+| Feature | What It Does |
+| ---------------------------------- | ------------------------------------------------------------------------ |
+| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free |
+| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider |
+| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions |
+| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection |
+| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry |
+| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control |
+| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session |
+| 🌐 **Wildcard Router** | `provider/*` dynamic routing |
+| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits |
+| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety |
+| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models |
+| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) |
+| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions |
+| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity |
+| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution |
+| 💬 **System Prompt Injection** | Global behavior controls applied consistently |
+| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows |
+
+### 🎵 Multi-Modal APIs
+
+| Feature | What It Does |
+| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends |
+| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines |
+| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support |
+| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages |
+| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) |
+| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) |
+| 🛡️ **Moderations** | `/v1/moderations` safety checks |
+| 🔀 **Reranking** | `/v1/rerank` for relevance scoring |
+| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache |
+
+### 🛡️ Resilience, Security & Governance
+
+| Feature | What It Does |
+| ----------------------------------- | -------------------------------------------------------------------------------------- |
+| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls |
+| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format |
+| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events |
+| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers |
+| ⚡ **Request Idempotency** | Duplicate protection window |
+| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** |
+| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** |
+| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments |
+| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence |
+| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations |
+| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks |
+| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures |
+| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically |
+| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls |
+| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` |
+| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog |
+
+### 📊 Observability & Analytics
+
+| Feature | What It Does |
+| -------------------------------- | ----------------------------------------------------- |
+| 📝 **Request + Proxy Logging** | Full request/response and proxy logging |
+| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI |
+| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page |
+| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing |
+| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats |
+| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility |
+| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views |
+| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies |
+| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing |
+
+### ☁️ Deployment & Platform
+
+| Feature | What It Does |
+| ------------------------------ | --------------------------------------------------------------------- |
+| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments |
+| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard |
+| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles |
+| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls |
+| 🔄 **Backup/Restore** | Export/import and disaster recovery flows |
+| 🧙 **Onboarding Wizard** | First-run guided setup |
+| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools |
+| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard |
+| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security |
+| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage |
+| 🧹 **Clear All Models** | One-click model list clearing in provider details |
+| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings |
+| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features |
+| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location |
+
+### Feature Deep Dive
#### Smart fallback with practical cost control
@@ -1295,103 +1464,132 @@ Combo: "my-coding-stack"
4. if/kimi-k2-thinking
```
-Когато квотата, скоростта или здравето са неуспешни, OmniRoute автоматично преминава към следващия кандидат без ръчно превключване.#### Protocol management that is visible and operable
+When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching.
-- MCP + A2A са откриваеми в UI и документи (не са скрити)
-- API за състоянието на протокола разкриват оперативни данни на живо (`/api/mcp/*`, `/api/a2a/*`)
-- Таблата за управление включват действия за операции от ден 2 (комбо превключвания, нулиране на прекъсвача, анулиране на задача)#### Translator + validation workflow
+#### Protocol management that is visible and operable
-Зоната за преводач включва:
+- MCP + A2A are discoverable in UI and docs (not hidden)
+- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`)
+- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation)
--**Playground**: поискайте проверки за трансформация -**Chat Tester**: пълна заявка/отговор двупосочно -**Тестова стенда**: множество случаи в едно изпълнение -**Монитор на живо**: изглед на трафика в реално време
+#### Translator + validation workflow
-Плюс проверка на протокола с реални клиенти чрез `npm run test:protocols:e2e`.
+The Translator area includes:
-> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Справка за инструменти, IDE конфигурации и примери за клиенти
+- **Playground**: request transformation checks
+- **Chat Tester**: full request/response round-trip
+- **Test Bench**: multiple cases in one run
+- **Live Monitor**: real-time traffic view
+
+Plus protocol validation with real clients via `npm run test:protocols:e2e`.
+
+> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples
>
-> 📖**[A2A Server README](src/lib/a2a/README.md)**— Умения, JSON-RPC методи, стрийминг и жизнен цикъл на задачите## 🧪 Evaluations (Evals)
+> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle
-OmniRoute включва вградена рамка за оценка за тестване на качеството на отговора на LLM спрямо златен набор. Достъп до него чрез**Analytics → Evals**в таблото за управление.### Built-in Golden Set
+## 🧪 Evaluations (Evals)
-Предварително зареденият "OmniRoute Golden Set" съдържа тестови случаи за:
+OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard.
-- Поздрави, математика, география, генериране на код
-- Съответствие с JSON формат, превод, генериране на маркдаун
-- Отказ за безопасност (вредно съдържание), броене, булева логика### Evaluation Strategies
+### Built-in Golden Set
-| Стратегия | Описание | Пример |
-| ------------ | ----------------------------------------------------------------------- | ------------------------------- | --- |
-| `точно` | Изходът трябва да съвпада точно | `"4"` |
-| `съдържа` | Изходът трябва да съдържа подниз (без значение за малки и големи букви) | `"Париж"` |
-| `регекс` | Изходът трябва да съответства на модела на регулярен израз | `"1.*2.*3"` |
-| `по поръчка` | Персонализираната JS функция връща true/false | `(изход) => изход.дължина > 10` | --- |
+The pre-loaded "OmniRoute Golden Set" contains test cases for:
+
+- Greetings, math, geography, code generation
+- JSON format compliance, translation, markdown generation
+- Safety refusal (harmful content), counting, boolean logic
+
+### Evaluation Strategies
+
+| Strategy | Description | Example |
+| ---------- | ------------------------------------------------ | -------------------------------- |
+| `exact` | Output must match exactly | `"4"` |
+| `contains` | Output must contain substring (case-insensitive) | `"Paris"` |
+| `regex` | Output must match regex pattern | `"1.*2.*3"` |
+| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` |
+
+---
## 📖 Setup Guide
### Protocol Setup (MCP + A2A)
-<подробности>
-<резюме>🧩 Настройка на MCP (моделен контекстен протокол)резюме>
+
+🧩 MCP Setup (Model Context Protocol)
-Стартирайте MCP транспорт в режим stdio:```bash
+Start MCP transport in stdio mode:
+
+```bash
omniroute --mcp
+```
-````
+Recommended validation flow:
-Препоръчителен поток за валидиране:
+1. Connect your MCP client over stdio.
+2. Run `omniroute_get_health`.
+3. Run `omniroute_list_combos`.
+4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit.
-1. Свържете вашия MCP клиент през stdio.
-2. Стартирайте `omniroute_get_health`.
-3. Стартирайте `omniroute_list_combos`.
-4. Отворете `/dashboard/mcp`, за да потвърдите пулса, активността и проверката.
-
-Полезни API за автоматизация:
+Useful APIs for automation:
- `GET /api/mcp/status`
- `GET /api/mcp/tools`
- `GET /api/mcp/audit`
-- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats`
-<подробности>
-<резюме>🤝 Настройка на A2A (Agent2Agent)резюме>
+
-Открийте агента:```bash
+
+🤝 A2A Setup (Agent2Agent)
+
+Discover the agent:
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
-Изпратете задача:```bash
+Send a task:
+
+```bash
curl -X POST http://localhost:20128/a2a \
- -H 'content-type: application/json' \
- -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+ -H 'content-type: application/json' \
+ -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+```
-````
-
-Управление на жизнения цикъл:
+Manage lifecycle:
- `GET /api/a2a/status`
- `GET /api/a2a/tasks`
- `GET /api/a2a/tasks/:id`
- `POST /api/a2a/tasks/:id/cancel`
-Оперативен потребителски интерфейс:
+Operational UI:
-- `/dashboard/a2a` за видимост на задача/състояние/поток и димни действия
+- `/dashboard/a2a` for task/state/stream observability and smoke actions
-<подробности>
-<резюме>🧪 Проверка на протокола от край до крайрезюме>
+
-Валидирайте и двата протокола с реални клиенти:```bash
+
+🧪 End-to-end protocol validation
+
+Validate both protocols with real clients:
+
+```bash
npm run test:protocols:e2e
-````
+```
-Това потвърждава:
+This verifies:
-- MCP SDK клиент за свързване/списък/обаждане
-- A2A откриване/изпращане/поток/получаване/отказ
-- Кръстосана проверка на данни в MCP одит и API за управление на задачи A2A
+- MCP SDK client connect/list/call
+- A2A discovery/send/stream/get/cancel
+- Cross-check data in MCP audit and A2A task management APIs
-<подробности>
-<резюме>💳 Доставчици на абонаментрезюме>### Claude Code (Pro/Max)
+
+
+
+💳 Subscription Providers
+
+### Claude Code (Pro/Max)
```bash
Dashboard → Providers → Connect Claude Code
@@ -1404,7 +1602,9 @@ Models:
cc/claude-haiku-4-5-20251001
```
-**Професионален съвет:**Използвайте Opus за сложни задачи, Sonnet за скорост. OmniRoute проследява квота за модел!### OpenAI Codex (Plus/Pro)
+**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model!
+
+### OpenAI Codex (Plus/Pro)
```bash
Dashboard → Providers → Connect Codex
@@ -1418,20 +1618,22 @@ Models:
#### Codex Account Limit Management (5h + Weekly)
-Всеки акаунт в Codex вече има превключватели на правилата в `Табло за управление -> Доставчици`:
+Each Codex account now has policy toggles in `Dashboard -> Providers`:
-- `5h` (ВКЛ./ИЗКЛ.): прилага политиката за 5-часов праг на прозореца.
-- `Седмично` (ВКЛ./ИЗКЛ.): прилагане на политиката за седмичния праг на прозореца.
-- Прагово поведение: когато активиран прозорец достигне >=90% използване, този акаунт се пропуска.
-- Ротационно поведение: OmniRoute автоматично пренасочва към следващия отговарящ на условията акаунт в Codex.
-- Поведение при нулиране: когато изтече времето за `resetAt` на доставчика, акаунтът отново автоматично става допустим.
+- `5h` (ON/OFF): enforce the 5-hour window threshold policy.
+- `Weekly` (ON/OFF): enforce the weekly window threshold policy.
+- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped.
+- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically.
+- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically.
-Сценарии:
+Scenarios:
-- `5h ON` + `Weekly ON`: акаунтът се пропуска, когато някой прозорец достигне прага.
-- `5h OFF` + `Weekly ON`: само седмично използване може да блокира акаунта.
-- `5h ВКЛ.` + `Weekly OFF`: само 5-часово използване може да блокира акаунта.
-- `resetAt` премина: акаунтът влиза отново в ротация автоматично (без ръчно повторно активиране).### Gemini CLI (FREE 180K/month!)
+- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold.
+- `5h OFF` + `Weekly ON`: only weekly usage can block the account.
+- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account.
+- `resetAt` passed: account re-enters rotation automatically (no manual re-enable).
+
+### Gemini CLI (FREE 180K/month!)
```bash
Dashboard → Providers → Connect Gemini CLI
@@ -1443,7 +1645,9 @@ Models:
gc/gemini-2.5-pro
```
-**Най-добра стойност:**Огромно безплатно ниво! Използвайте това преди платените нива.### GitHub Copilot
+**Best Value:** Huge free tier! Use this before paid tiers.
+
+### GitHub Copilot
```bash
Dashboard → Providers → Connect GitHub
@@ -1458,71 +1662,91 @@ Models:
-<подробности>
-<резюме>🔑 Доставчици на ключове за APIрезюме>### NVIDIA NIM (FREE developer access — 70+ models)
+
+🔑 API Key Providers
-1. Регистрирайте се: [build.nvidia.com](https://build.nvidia.com)
-2. Вземете безплатен API ключ (включени 1000 кредита за изводи)
-3. Табло → Добавяне на доставчик → NVIDIA NIM:
- - API ключ: `nvapi-вашият-ключ`
+### NVIDIA NIM (FREE developer access — 70+ models)
-**Модели:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` и още 50+
+1. Sign up: [build.nvidia.com](https://build.nvidia.com)
+2. Get free API key (1000 inference credits included)
+3. Dashboard → Add Provider → NVIDIA NIM:
+ - API Key: `nvapi-your-key`
-**Професионален съвет:**OpenAI-съвместим API — работи безпроблемно с превода на формати на OmniRoute!### DeepSeek
+**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more
-1. Регистрирайте се: [platform.deepseek.com](https://platform.deepseek.com)
-2. Вземете API ключ
-3. Табло → Добавяне на доставчик → DeepSeek
+**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation!
-**Модели:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!)
+### DeepSeek
-1. Регистрирайте се: [console.groq.com](https://console.groq.com)
-2. Вземете API ключ (включено безплатно ниво)
-3. Табло → Добавяне на доставчик → Groq
+1. Sign up: [platform.deepseek.com](https://platform.deepseek.com)
+2. Get API key
+3. Dashboard → Add Provider → DeepSeek
-**Модели:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b`
+**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder`
-**Професионален съвет:**Изключително бърз извод — най-добър за кодиране в реално време!### OpenRouter (100+ Models)
+### Groq (Free Tier Available!)
-1. Регистрирайте се: [openrouter.ai](https://openrouter.ai)
-2. Вземете API ключ
-3. Табло → Добавяне на доставчик → OpenRouter
+1. Sign up: [console.groq.com](https://console.groq.com)
+2. Get API key (free tier included)
+3. Dashboard → Add Provider → Groq
-**Модели:**Достъп до 100+ модела от всички основни доставчици чрез един API ключ.
+**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b`
-**Поведение на таблото:**Моделите OpenRouter се управляват от**Налични модели**. Ръчното добавяне, импортиране и автоматично синхронизиране актуализира един и същ списък.
+**Pro Tip:** Ultra-fast inference — best for real-time coding!
-<подробности>
-<резюме>💰 Евтини доставчици (резервни)резюме>### GLM-4.7 (Daily reset, $0.6/1M)
+### OpenRouter (100+ Models)
-1. Регистрирайте се: [Zhipu AI](https://open.bigmodel.cn/)
-2. Вземете API ключ от Coding Plan
-3. Табло → Добавяне на API ключ:
- - Доставчик: `glm`
- - API ключ: `вашият-ключ`
+1. Sign up: [openrouter.ai](https://openrouter.ai)
+2. Get API key
+3. Dashboard → Add Provider → OpenRouter
-**Използвайте:**`glm/glm-4.7`
+**Models:** Access 100+ models from all major providers through a single API key.
-**Професионален съвет:**Планът за кодиране предлага 3× квота на цена 1/7! Нулирайте всеки ден в 10:00 ч.### MiniMax M2.1 (5h reset, $0.20/1M)
+**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list.
-1. Регистрирайте се: [MiniMax](https://www.minimax.io/)
-2. Вземете API ключ
-3. Табло → Добавяне на API ключ
+
-**Използвайте:**`minimax/MiniMax-M2.1`
+
+💰 Cheap Providers (Backup)
-**Професионален съвет:**Най-евтината опция за дълъг контекст (1M токени)!### Kimi K2 ($9/month flat)
+### GLM-4.7 (Daily reset, $0.6/1M)
-1. Абонирайте се: [Moonshot AI](https://platform.moonshot.ai/)
-2. Вземете API ключ
-3. Табло → Добавяне на API ключ
+1. Sign up: [Zhipu AI](https://open.bigmodel.cn/)
+2. Get API key from Coding Plan
+3. Dashboard → Add API Key:
+ - Provider: `glm`
+ - API Key: `your-key`
-**Използвайте:**`kimi/kimi-latest`
+**Use:** `glm/glm-4.7`
-**Професионален съвет:**Фиксирани $9/месец за 10 милиона токена = $0,90/1 милион ефективна цена!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM.
-<подробности>
-<резюме>🆓 БЕЗПЛАТНИ доставчици (Спешно архивиране)резюме>### Qoder (5 FREE models via OAuth)
+### MiniMax M2.1 (5h reset, $0.20/1M)
+
+1. Sign up: [MiniMax](https://www.minimax.io/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `minimax/MiniMax-M2.1`
+
+**Pro Tip:** Cheapest option for long context (1M tokens)!
+
+### Kimi K2 ($9/month flat)
+
+1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `kimi/kimi-latest`
+
+**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost!
+
+
+
+
+🆓 FREE Providers (Emergency Backup)
+
+### Qoder (5 FREE models via OAuth)
```bash
Dashboard → Connect Qoder
@@ -1563,9 +1787,10 @@ Models:
-<подробности>
+
+🎨 Create Combos
-🎨 Създаване на комбинации
### Example 1: Maximize Subscription → Cheap Backup
+### Example 1: Maximize Subscription → Cheap Backup
```
Dashboard → Combos → Create New
@@ -1593,8 +1818,10 @@ Cost: $0 forever!
-<подробности>
-<резюме>🔧 CLI интеграциярезюме>### Cursor IDE
+
+🔧 CLI Integration
+
+### Cursor IDE
```
Settings → Models → Advanced:
@@ -1605,7 +1832,9 @@ Settings → Models → Advanced:
### Claude Code
-Използвайте страницата**CLI Tools**в таблото за управление за конфигурация с едно кликване или редактирайте `~/.claude/settings.json` ръчно.### Codex CLI
+Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually.
+
+### Codex CLI
```bash
export OPENAI_BASE_URL="http://localhost:20128"
@@ -1616,12 +1845,15 @@ codex "your prompt"
### OpenClaw
-**Вариант 1 — Табло (препоръчително):**```
+**Option 1 — Dashboard (recommended):**
+
+```
Dashboard → CLI Tools → OpenClaw → Select Model → Apply
+```
-````
+**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`:
-**Опция 2 — Ръчно:**Редактиране на `~/.openclaw/openclaw.json`:```json
+```json
{
"models": {
"providers": {
@@ -1633,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply
}
}
}
-````
+```
-> **Забележка:**OpenClaw работи само с локален OmniRoute. Използвайте „127.0.0.1“ вместо „localhost“, за да избегнете проблеми с разрешаването на IPv6.### Cline / Continue / RooCode
+> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues.
+
+### Cline / Continue / RooCode
```
Settings → API Configuration:
@@ -1647,15 +1881,17 @@ Settings → API Configuration:
### OpenCode
-**Стъпка 1:**Добавете OmniRoute като персонализиран доставчик:```bash
+**Step 1:** Add OmniRoute as a custom provider:
+
+```bash
opencode
/connect
-
# Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key
+```
-````
+**Step 2:** Create/edit `opencode.json` in your project root:
-**Стъпка 2:**Създайте/редактирайте `opencode.json` в корена на вашия проект:```json
+```json
{
"$schema": "https://opencode.ai/config.json",
"provider": {
@@ -1673,117 +1909,130 @@ opencode
}
}
}
-````
+```
-**Стъпка 3:**Изберете модела в OpenCode:```bash
+**Step 3:** Select the model in OpenCode:
+
+```bash
/models
-
# Select any OmniRoute model from the list
+```
-````
+> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard.
->**Съвет:**Добавете всеки модел, наличен във вашата крайна точка OmniRoute `/v1/models` към раздела `models`. Използвайте формата „провайдер/идентификатор на модел“ от таблото за управление на OmniRoute.
+
---
## Отстраняване на проблеми
-<подробности>
-Щракнете, за да разширите ръководството за отстраняване на неизправности
+
+Click to expand troubleshooting guide
-**„Езиковият модел не предостави съобщения“**
+**"Language model did not provide messages"**
-- Квотата на доставчика е изчерпана → Проверете инструмента за проследяване на квотата на таблото за управление
-- Решение: Използвайте комбо резервен вариант или преминете към по-евтино ниво
+- Provider quota exhausted → Check dashboard quota tracker
+- Solution: Use combo fallback or switch to cheaper tier
-**Ограничаване на скоростта**
+**Rate limiting**
-- Изчерпване на квотата за абонамент → Резервно връщане към GLM/MiniMax
-- Добавете комбо: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Subscription quota out → Fallback to GLM/MiniMax
+- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-**OAuth токенът е изтекъл**
+**OAuth token expired**
-- Автоматично опресняване от OmniRoute
-- Ако проблемите продължават: Табло за управление → Доставчик → Свързване отново
+- Auto-refreshed by OmniRoute
+- If issues persist: Dashboard → Provider → Reconnect
-**Високи разходи**
+**High costs**
-- Проверете статистическите данни за използването в Табло → Разходи
-- Превключете основния модел към GLM/MiniMax
-- Използвайте безплатно ниво (Gemini CLI, Qoder) за некритични задачи
+- Check usage stats in Dashboard → Costs
+- Switch primary model to GLM/MiniMax
+- Use free tier (Gemini CLI, Qoder) for non-critical tasks
-**Портовете на таблото/API са грешни**
+**Dashboard/API ports are wrong**
-- `PORT` е каноничният базов порт (и API порт по подразбиране)
-- `API_PORT` заменя само OpenAI-съвместим API слушател
-- `DASHBOARD_PORT` заменя само слушателя на таблото за управление/Next.js
-- Задайте `NEXT_PUBLIC_BASE_URL` на вашето табло за управление/публичен URL (за OAuth обратни извиквания)
+- `PORT` is the canonical base port (and API port by default)
+- `API_PORT` overrides only OpenAI-compatible API listener
+- `DASHBOARD_PORT` overrides only dashboard/Next.js listener
+- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks)
-**Грешки при синхронизиране в облак**
+**Cloud sync errors**
-- Уверете се, че `BASE_URL` сочи към вашия работещ екземпляр
-- Уверете се, че `CLOUD_URL` сочи към вашата очаквана крайна точка в облака
-- Поддържайте стойностите на `NEXT_PUBLIC_*` в съответствие със стойностите от страна на сървъра
+- Verify `BASE_URL` points to your running instance
+- Verify `CLOUD_URL` points to your expected cloud endpoint
+- Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**Първото влизане не работи**
+**First login not working**
-- Проверете `INITIAL_PASSWORD` в `.env`
-- Ако не е зададена, резервната парола е „123456“.
+- Check `INITIAL_PASSWORD` in `.env`
+- If unset, fallback password is `123456`
-**Няма регистрационни файлове за заявки**
+**No request logs**
-- Артефактите на заявката се записват в `DATA_DIR/call_logs/` като един JSON файл на заявка
-- Активирайте улавянето на тръбопровода от таблото за управление → Регистри → Искане на регистрационни файлове, ако имате нужда от подробни полезни товари на етап
-- Задайте `APP_LOG_TO_FILE=true`, ако също искате регистрационни файлове на конзолата на приложението в `logs/application/app.log`
-- Коригирайте `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` и `CALL_LOG_MAX_ENTRIES` според нуждите
+- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request
+- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads
+- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log`
+- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed
-**Тестът за връзка показва „Невалидно“ за OpenAI-съвместими доставчици**
+**Connection test shows "Invalid" for OpenAI-compatible providers**
-- Много доставчици не излагат крайна точка `/models`
-- OmniRoute v1.0.6+ включва резервно валидиране чрез завършвания на чат
-- Уверете се, че основният URL адрес включва суфикс `/v1`### 🔐 OAuth on a Remote Server
+- Many providers don't expose a `/models` endpoint
+- OmniRoute v1.0.6+ includes fallback validation via chat completions
+- Ensure base URL includes `/v1` suffix
+
+### 🔐 OAuth on a Remote Server
->**⚠️ Важно за потребители, работещи с OmniRoute на VPS, Docker или друг отдалечен сървър**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
+> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server**
-Доставчиците на**Antigravity**и**Gemini CLI**използват**Google OAuth 2.0**. Google изисква „redirect_uri“ в OAuth потока да съвпада точно с един от предварително регистрираните URI адреси в Google Cloud Console на приложението.
+#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
-Идентификационните данни за OAuth, включени в OmniRoute, са регистрирани**само за `localhost`**. Когато получите достъп до OmniRoute на отдалечен сървър (напр. `https://omniroute.myserver.com`), Google отхвърля удостоверяването с:```
+The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console.
+
+The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solution: Configure your own OAuth credentials
-Трябва да създадете**OAuth 2.0 Client ID**в Google Cloud Console с URI на вашия сървър.#### Step-by-step
+You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI.
-**1. Отворете Google Cloud Console**
+#### Step-by-step
-Отидете на: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
+**1. Open Google Cloud Console**
-**2. Създайте нов OAuth 2.0 клиентски идентификатор**
+Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-- Щракнете върху**"+ Създаване на идентификационни данни"**→**"OAuth клиентски идентификатор"**
-- Тип приложение:**"Уеб приложение"**
-- Име: каквото искате (напр. „OmniRoute Remote“)
+**2. Create a new OAuth 2.0 Client ID**
-**3. Добавете оторизирани URI адреси за пренасочване**
+- Click **"+ Create Credentials"** → **"OAuth client ID"**
+- Application type: **"Web application"**
+- Name: anything you like (e.g. `OmniRoute Remote`)
-В полето**„Оторизирани URI адреси за пренасочване“**добавете:```
+**3. Add Authorized Redirect URIs**
+
+In the **"Authorized redirect URIs"** field, add:
+
+```
https://your-server.com/callback
+```
-````
+> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`).
-> Заменете `your-server.com` с домейна или IP на вашия сървър (включете порта, ако е необходимо, напр. `http://45.33.32.156:20128/callback`).
+**4. Save and copy the credentials**
-**4. Запазете и копирайте идентификационните данни**
+After creating, Google will show the **Client ID** and **Client Secret**.
-След създаването Google ще покаже**Клиентски идентификатор**и**Клиентска тайна**.
+**5. Set environment variables**
-**5. Задайте променливи на средата**
+In your `.env` (or Docker environment variables):
-Във вашия `.env` (или променливи на средата Docker):```bash
+```bash
# For Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
@@ -1792,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
-````
+```
-**6. Рестартирайте OmniRoute**```bash
+**6. Restart OmniRoute**
+```bash
# npm:
-
npm run dev
# Docker:
-
docker restart omniroute
+```
-````
+**7. Try connecting again**
-**7. Опитайте да се свържете отново**
+Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth
-Табло → Доставчици → Antigravity (или Gemini CLI) → OAuth
+Google will now redirect correctly to `https://your-server.com/callback`.
-Google вече ще пренасочва правилно към `https://your-server.com/callback`.---
+---
#### Temporary workaround (without custom credentials)
-Ако не искате да настроите свои собствени идентификационни данни точно сега, можете да използвате**ръчния URL поток**:
+If you don't want to set up your own credentials right now, you can still use the **manual URL flow**:
-1. OmniRoute отваря URL адреса за оторизация на Google
-2. След упълномощаване Google се опитва да пренасочи към `localhost` (което не успява на отдалечения сървър)
-3.**Копирайте пълния URL**от адресната лента на вашия браузър (дори страницата да не се зарежда)
-4. Поставете този URL адрес в полето, показано в модала за свързване на OmniRoute
-5. Щракнете върху**"Свързване"**
+1. OmniRoute opens the Google authorization URL
+2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server)
+3. **Copy the full URL** from your browser's address bar (even if the page doesn't load)
+4. Paste that URL into the field shown in the OmniRoute connection modal
+5. Click **"Connect"**
-> Това работи, защото кодът за оторизация в URL адреса е валиден независимо дали страницата за пренасочване е заредена.---
+> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded.
-<подробности>
-<резюме>🇧🇷 Versão em Portuguêsрезюме>#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+---
-Доставчиците на**Antigravity**и**Gemini CLI**използват**Google OAuth 2.0**за удостоверяване. Google изисква, че `redirect_uri` не използва fluxo OAuth като**exatamente**, за да може URI преди кадастрада да не се използва Google Cloud Console за приложение.
+
+🇧🇷 Versão em Português
-Като пълномощия за OAuth не се използва OmniRoute в кадастрада**apenas para `localhost`**. Ако имате достъп до OmniRoute в дистанционния сървър (напр.: `https://omniroute.meuservidor.com`), или Google rejeita a autenticação com:```
+#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+
+Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo.
+
+As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solução: Configure suas próprias credenciais OAuth
-Изпишете точно**OAuth 2.0 Client ID**без Google Cloud Console чрез URI на вашия сървър.#### Passo a passo
+Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor.
-**1. Достъп до Google Cloud Console**
+#### Passo a passo
+
+**1. Acesse o Google Cloud Console**
Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
**2. Crie um novo OAuth 2.0 Client ID**
-- Кликнете върху**"+ Създаване на идентификационни данни"**→**"OAuth клиентски идентификатор"**
-- Tipo de aplicativo:**"Уеб приложение"**
-- Име: escolha qualquer име (напр.: `OmniRoute Remote`)
+- Clique em **"+ Create Credentials"** → **"OAuth client ID"**
+- Tipo de aplicativo: **"Web application"**
+- Nome: escolha qualquer nome (ex: `OmniRoute Remote`)
-**3. Adicione като оторизирани URI адреси за пренасочване**
+**3. Adicione as Authorized Redirect URIs**
-Без поле**„Оторизирани URI адреси за пренасочване“**, добавете:```
+No campo **"Authorized redirect URIs"**, adicione:
+
+```
https://seu-servidor.com/callback
+```
-````
-
-> Заменете `seu-servidor.com` домейн или IP на вашия сървър (включително необходим порт, напр.: `http://45.33.32.156:20128/callback`).
+> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`).
**4. Salve e copie as credenciais**
-Например, Google показва**Клиентски идентификатор**и**Клиентска тайна**.
+Após criar, o Google mostrará o **Client ID** e o **Client Secret**.
-**5. Конфигуриране като variáveis de ambiente**
+**5. Configure as variáveis de ambiente**
-Не се използва `.env` (или нашите варианти на средата на Docker):```bash
+No seu `.env` (ou nas variáveis de ambiente do Docker):
+
+```bash
# Para Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
@@ -1871,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
-````
+```
-**6. Reinicie o OmniRoute**```bash
+**6. Reinicie o OmniRoute**
+```bash
# Se usando npm:
-
npm run dev
# Se usando Docker:
-
docker restart omniroute
-
-````
+```
**7. Tente conectar novamente**
-Табло → Доставчици → Антигравитация (или Gemini CLI) → OAuth
+Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth
-Agora или Google пренасочва корретаментно за „https://seu-servidor.com/callback“ и функционира автентичност.---
+Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará.
+
+---
#### Workaround temporário (sem configurar credenciais próprias)
-Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo**manual de URL**:
+Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**:
-1. O OmniRoute премахва URL адрес за авторизация от Google
-2. Ако не разрешите, пренасочването на Google към „localhost“ (не може да се използва отдалечен сървър)
-3.**Копирайте пълния URL адрес**от страницата, която искате да прехвърлите в своя браузър (mesmo que a página não carregue)
+1. O OmniRoute abrirá a URL de autorização do Google
+2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto)
+3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue)
4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute
-5. Щракнете върху**"Свързване"**
+5. Clique em **"Connect"**
-> Това заобиколно решение функционира, ако кодът на авторизацията на URL е валиден независимо от пренасочването към пренасочване или не.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+
+
---
@@ -1909,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux
## 🛠️ Tech Stack
-<подробности>
-Щракнете, за да разгънете подробностите за технически стек
+
+Click to expand tech stack details
--**Време на изпълнение**: Node.js 18–22 LTS (⚠️ Node.js 24+**не се поддържа**— собствените бинарни файлове на `better-sqlite3` са несъвместими)
--**Език**: TypeScript 5.9 —**100% TypeScript**в `src/` и `open-sse/` (нула `any` в основните модули от v2.0)
--**Framework**: Next.js 16 + React 19 + Tailwind CSS 4
--**База данни**: LowDB (JSON) + SQLite (състояние на домейна + регистрационни файлове на прокси + MCP одит + решения за маршрутизиране)
--**Схеми**: Zod (валидиране на I/O инструмент за MCP, API договори)
--**Протоколи**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
--**Поточно предаване**: Изпратени от сървъра събития (SSE)
--**Auth**: OAuth 2.0 (PKCE) + JWT + API ключове + MCP оторизация с обхват
--**Тестване**: Node.js тестов инструмент + Vitest (900+ теста, включително модул, интеграция, E2E)
--**CI/CD**: Действия на GitHub (автоматично публикуване на npm + Docker Hub при пускане)
--**Уебсайт**: [omniroute.online](https://omniroute.online)
--**Пакет**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
--**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
--**Устойчивост**: прекъсвач, експоненциално отдръпване, анти-гръмотевично стадо, TLS подправяне, автоматично комбинирано самолечение
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible)
+- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0)
+- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4
+- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions)
+- **Schemas**: Zod (MCP tool I/O validation, API contracts)
+- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
+- **Streaming**: Server-Sent Events (SSE)
+- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization
+- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E)
+- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release)
+- **Website**: [omniroute.online](https://omniroute.online)
+- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
+- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
+- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing
+
+
---
## Документация
-| Документ | Описание |
-| ---------------------------------------------- | -------------------------------------------------- |
-| [Ръководство на потребителя](docs/USER_GUIDE.md) | Доставчици, комбинации, CLI интеграция, внедряване |
-| [Справочник за API](docs/API_REFERENCE.md) | Всички крайни точки с примери |
-| [MCP сървър](open-sse/mcp-server/README.md) | 16 MCP инструмента, IDE конфигурации, Python/TS/Go клиенти |
-| [A2A сървър](src/lib/a2a/README.md) | JSON-RPC 2.0 протокол, умения, стрийминг, управление на задачи |
-| [Auto-Combo Engine](docs/auto-combo.md) | 6-факторно оценяване, пакети с режими, самолечение |
-| [Отстраняване на неизправности](docs/TROUBLESHOOTING.md) | Често срещани проблеми и решения |
-| [Архитектура](docs/ARCHITECTURE.md) | Системна архитектура и вътрешност |
-| [Принос](CONTRIBUTING.md) | Настройка и насоки за разработка |
-| [OpenAPI Spec](docs/openapi.yaml) | Спецификация на OpenAPI 3.0 |
-| [Правила за сигурност](SECURITY.md) | Отчитане на уязвимости и практики за сигурност |
-| [Внедряване на VM](docs/VM_DEPLOYMENT_GUIDE.md) | Пълно ръководство: Настройка на VM + nginx + Cloudflare |
-| [Галерия с функции](docs/FEATURES.md) | Визуална обиколка на таблото с екранни снимки |
-| [Списък за проверка на изданието](docs/RELEASE_CHECKLIST.md) | Стъпки за валидиране преди пускане |---
+| Document | Description |
+| ---------------------------------------------- | --------------------------------------------------- |
+| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment |
+| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples |
+| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients |
+| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt |
+| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing |
+| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation |
+| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions |
+| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals |
+| [Contributing](CONTRIBUTING.md) | Development setup and guidelines |
+| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification |
+| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices |
+| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup |
+| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots |
+| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps |
+
+---
## 🗺️ Roadmap
-OmniRoute има**планирани 210+ функции**в множество фази на разработка. Ето основните области:
+OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas:
-| Категория | Планирани функции | Акценти |
-| ----------------------------- | ---------------- | ----------------------------------------------------------------------------------------------- |
-| 🧠**Маршрутизиране и разузнаване**| 25+ | Маршрутизиране с най-ниска латентност, маршрутизиране на базата на етикети, предварителен полет на квота, избор на P2C акаунт |
-| 🔒**Сигурност и съответствие**| 20+ | SSRF укрепване, прикриване на идентификационни данни, ограничение на скоростта за крайна точка, обхват на ключ за управление |
-| 📊**Наблюдаемост**| 15+ | OpenTelemetry интеграция, мониторинг на квоти в реално време, проследяване на разходите за модел |
-| 🔄**Интеграции на доставчици**| 20+ | Регистър на динамичен модел, изчакване на доставчика, Codex за множество акаунти, анализ на квота на Copilot |
-| ⚡**Изпълнение**| 15+ | Слой с двоен кеш, кеш за подкани, кеш за отговор, поддържане на активността при поточно предаване, партиден API |
-| 🌐**Екосистема**| 10+ | WebSocket API, горещо презареждане на конфигурация, разпределено хранилище за конфигурация, търговски режим |### 🔜 Coming Soon
+| Category | Planned Features | Highlights |
+| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- |
+| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection |
+| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping |
+| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model |
+| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing |
+| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API |
+| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode |
-- 🔗**OpenCode Integration**— Поддръжка на родния доставчик за IDE за кодиране OpenCode AI
-- 🔗**TRAE Integration**— Пълна поддръжка за рамката за разработка на TRAE AI
-- 📦**Batch API**— Асинхронна групова обработка за групови заявки
-- 🎯**Маршрутизиране на базата на етикети**— Маршрутизирайте заявки въз основа на персонализирани тагове и метаданни
-- 💰**Стратегия с най-ниска цена**— Автоматично изберете най-евтиния наличен доставчик
+### 🔜 Coming Soon
-> 📝 Пълните спецификации на функциите са налични в [`docs/new-features/`](docs/new-features/) (217 подробни спецификации)---
+- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE
+- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework
+- 📦 **Batch API** — Asynchronous batch processing for bulk requests
+- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata
+- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider
+
+> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs)
+
+---
## 👥 Contributors
@@ -1974,18 +2245,20 @@ OmniRoute има**планирани 210+ функции**в множество
### How to Contribute
-1. Разклонете хранилището
-2. Създайте свой клон на функции (`git checkout -b feature/amazing-feature`)
-3. Задайте вашите промени (`git commit -m 'Добавяне на невероятна функция'`)
-4. Пуш към клона (`git push origin feature/amazing-feature`)
-5. Отворете заявка за изтегляне
+1. Fork the repository
+2. Create your feature branch (`git checkout -b feature/amazing-feature`)
+3. Commit your changes (`git commit -m 'Add amazing feature'`)
+4. Push to the branch (`git push origin feature/amazing-feature`)
+5. Open a Pull Request
-Вижте [CONTRIBUTING.md](CONTRIBUTING.md) за подробни насоки.### Releasing a New Version
+See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines.
+
+### Releasing a New Version
```bash
# Create a release — npm publish happens automatically
gh release create v2.0.0 --title "v2.0.0" --generate-notes
-````
+```
---
@@ -1997,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes
## 🙏 Acknowledgments
-Специални благодарности на**[9router](https://github.com/decolua/9router)**от**[decolua](https://github.com/decolua)**— оригиналният проект, който вдъхнови това разклонение. OmniRoute се основава на тази невероятна основа с допълнителни функции, мултимодални API и пълно пренаписване на TypeScript.
+Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite.
-Специални благодарности на**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— оригиналната реализация на Go, която вдъхнови този JavaScript порт.---
+Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port.
+
+---
## Лиценз
-Лиценз на MIT - вижте [ЛИЦЕНЗ](ЛИЦЕНЗ) за подробности.---
+MIT License - see [LICENSE](LICENSE) for details.
+
+---
Built with ❤️ for developers who code 24/7
diff --git a/docs/i18n/bg/docs/ARCHITECTURE.md b/docs/i18n/bg/docs/ARCHITECTURE.md
index 0c0398c41f..4bb4e676d5 100644
--- a/docs/i18n/bg/docs/ARCHITECTURE.md
+++ b/docs/i18n/bg/docs/ARCHITECTURE.md
@@ -4,80 +4,93 @@
---
-_Последна актуализация: 2026-03-28_## Executive Summary
-OmniRoute е локален AI маршрутизиращ шлюз и табло за управление, изградено на Next.js.
-Той осигурява единична OpenAI-съвместима крайна точка (`/v1/*`) и маршрутизира трафик през множество доставчици нагоре по веригата с превод, резервен вариант, опресняване на токени и проследяване на използването.
-Основни възможности:
+_Last updated: 2026-03-28_
-- OpenAI-съвместима API повърхност за CLI/инструменти (28 доставчици)
-- Превод на заявка/отговор във форматите на доставчика
-- Резервна комбинация от модели (последователност от няколко модела)
-- Резервен вариант на ниво акаунт (мулти акаунт на доставчик)
-- OAuth + API-ключ управление на връзката на доставчика
-- Генериране на вграждане чрез `/v1/embeddings` (6 доставчика, 9 модела)
-- Генериране на изображения чрез `/v1/images/generations` (4 доставчика, 9 модела)
-- Синтактичен анализ на таг за мислене (`
...`) за разсъждаващи модели
-- Дезинфекция на отговора за стриктна съвместимост с OpenAI SDK
-- Нормализиране на ролята (разработчик→система, система→потребител) за съвместимост между доставчици
-- Структурирано преобразуване на изход (json_schema → Gemini responseSchema)
-- Локална устойчивост за доставчици, ключове, псевдоними, комбинации, настройки, ценообразуване
-- Проследяване на използване/разходи и регистриране на заявки
-- Допълнителна облачна синхронизация за синхронизиране на множество устройства/състояние
-- Списък с разрешени/блокирани IP адреси за контрол на достъпа до API
-- Мислещо управление на бюджета (преминаване/автоматично/персонализирано/адаптивно)
-- Бързо инжектиране на глобалната система
-- Проследяване на сесии и пръстови отпечатъци
-- Подобрено ограничаване на скоростта за всеки акаунт със специфични за доставчика профили
-- Модел на прекъсвача за устойчивост на доставчика
-- Анти-гръмотевична стадна защита с mutex заключване
-- Кеш за дедупликация на заявки, базиран на подпис
-- Слой на домейна: наличност на модела, правила за разходите, резервна политика, политика за блокиране
-- Устойчивост на състоянието на домейна (кеш за запис на SQLite за резервни варианти, бюджети, блокировки, прекъсвачи на верига)
-- Механизъм за правила за централизирана оценка на заявката (заключване → бюджет → резервен)
-- Заявка за телеметрия с p50/p95/p99 агрегиране на латентност
-- ID на корелация (X-Request-Id) за проследяване от край до край
-- Регистриране на одит за съответствие с отказ за всеки API ключ
-- Eval framework за осигуряване на качеството на LLM
-- Resilience UI табло със статус на прекъсвача в реално време
-- Модулни OAuth доставчици (12 отделни модула под `src/lib/oauth/providers/`)
+## Executive Summary
-Основен модел на изпълнение:
+OmniRoute is a local AI routing gateway and dashboard built on Next.js.
+It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking.
-- Маршрутите на приложението Next.js под `src/app/api/*` внедряват както API на таблото, така и API за съвместимост
-- Споделено SSE/маршрутизиращо ядро в `src/sse/*` + `open-sse/*` обработва изпълнението на доставчика, превода, стрийминг, резервен вариант и използване## Scope and Boundaries
+Core capabilities:
+
+- OpenAI-compatible API surface for CLI/tools (28 providers)
+- Request/response translation across provider formats
+- Model combo fallback (multi-model sequence)
+- Account-level fallback (multi-account per provider)
+- OAuth + API-key provider connection management
+- Embedding generation via `/v1/embeddings` (6 providers, 9 models)
+- Image generation via `/v1/images/generations` (4 providers, 9 models)
+- Think tag parsing (`
...`) for reasoning models
+- Response sanitization for strict OpenAI SDK compatibility
+- Role normalization (developer→system, system→user) for cross-provider compatibility
+- Structured output conversion (json_schema → Gemini responseSchema)
+- Local persistence for providers, keys, aliases, combos, settings, pricing
+- Usage/cost tracking and request logging
+- Optional cloud sync for multi-device/state sync
+- IP allowlist/blocklist for API access control
+- Thinking budget management (passthrough/auto/custom/adaptive)
+- Global system prompt injection
+- Session tracking and fingerprinting
+- Per-account enhanced rate limiting with provider-specific profiles
+- Circuit breaker pattern for provider resilience
+- Anti-thundering herd protection with mutex locking
+- Signature-based request deduplication cache
+- Domain layer: model availability, cost rules, fallback policy, lockout policy
+- Context Relay: session handoff summaries for account rotation continuity
+- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers)
+- Policy engine for centralized request evaluation (lockout → budget → fallback)
+- Request telemetry with p50/p95/p99 latency aggregation
+- Correlation ID (X-Request-Id) for end-to-end tracing
+- Compliance audit logging with opt-out per API key
+- Eval framework for LLM quality assurance
+- Resilience UI dashboard with real-time circuit breaker status
+- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`)
+
+Primary runtime model:
+
+- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs
+- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage
+
+## Scope and Boundaries
### In Scope
-- Време за изпълнение на локален шлюз
-- API за управление на таблото
-- Удостоверяване на доставчика и опресняване на токена
-- Заявка за превод и SSE стрийминг
-- Локално състояние + постоянство на използване
-- Допълнителна синхронизация в облака### Out of Scope
+- Local gateway runtime
+- Dashboard management APIs
+- Provider authentication and token refresh
+- Request translation and SSE streaming
+- Local state + usage persistence
+- Optional cloud sync orchestration
-- Внедряване на облачна услуга зад `NEXT_PUBLIC_CLOUD_URL`
-- SLA/контролна равнина на доставчика извън локалния процес
-- Самите външни CLI двоични файлове (Claude CLI, Codex CLI и т.н.)## Dashboard Surface (Current)
+### Out of Scope
-Главни страници под `src/app/(dashboard)/dashboard/`:
+- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL`
+- Provider SLA/control plane outside local process
+- External CLI binaries themselves (Claude CLI, Codex CLI, etc.)
-- `/dashboard` — бърз старт + преглед на доставчика
-- `/dashboard/endpoint` — крайна точка прокси + MCP + A2A + раздели за крайна точка на API
-- `/dashboard/providers` — връзки и идентификационни данни на доставчика
-- `/dashboard/combos` — комбинирани стратегии, шаблони, правила за маршрутизиране на модели
-- `/dashboard/costs` — агрегиране на разходите и видимост на цените
-- `/dashboard/analytics` — анализи и оценки на използването
-- `/dashboard/limits` — контроли на квоти/ставки
-- `/dashboard/cli-tools` — CLI включване, откриване по време на изпълнение, генериране на конфигурация
-- `/dashboard/agents` — открити ACP агенти + потребителска регистрация на агент
-- `/dashboard/media` — игрище за изображения/видео/музика
-- `/dashboard/search-tools` — тестване и история на доставчика на търсене
-- `/dashboard/health` — време на работа, прекъсвачи, ограничения на скоростта
-- `/dashboard/logs` — регистрационни файлове на заявка/прокси/одит/конзола
-- `/dashboard/settings` — раздели за системни настройки (общи, маршрутизиране, комбинирани настройки по подразбиране и т.н.)
-- `/dashboard/api-manager` — жизнен цикъл на API ключ и разрешения за модел## High-Level System Context
+## Dashboard Surface (Current)
+
+Main pages under `src/app/(dashboard)/dashboard/`:
+
+- `/dashboard` — quick start + provider overview
+- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs
+- `/dashboard/providers` — provider connections and credentials
+- `/dashboard/combos` — combo strategies, templates, model routing rules
+- `/dashboard/costs` — cost aggregation and pricing visibility
+- `/dashboard/analytics` — usage analytics and evaluations
+- `/dashboard/limits` — quota/rate controls
+- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation
+- `/dashboard/agents` — detected ACP agents + custom agent registration
+- `/dashboard/media` — image/video/music playground
+- `/dashboard/search-tools` — search provider testing and history
+- `/dashboard/health` — uptime, circuit breakers, rate limits
+- `/dashboard/logs` — request/proxy/audit/console logs
+- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.)
+- `/dashboard/api-manager` — API key lifecycle and model permissions
+
+## High-Level System Context
```mermaid
flowchart LR
@@ -129,139 +142,151 @@ flowchart LR
## 1) API and Routing Layer (Next.js App Routes)
-Основни директории:
+Main directories:
-- `src/app/api/v1/*` и `src/app/api/v1beta/*` за API за съвместимост
-- `src/app/api/*` за API за управление/конфигуриране
-- Следващото пренаписване в `next.config.mjs` преобразува `/v1/*` в `/api/v1/*`
+- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs
+- `src/app/api/*` for management/configuration APIs
+- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*`
-Важни пътища за съвместимост:
+Important compatibility routes:
- `src/app/api/v1/chat/completions/route.ts`
- `src/app/api/v1/messages/route.ts`
- `src/app/api/v1/responses/route.ts`
-- `src/app/api/v1/models/route.ts` — включва потребителски модели с `custom: true`
-- `src/app/api/v1/embeddings/route.ts` — генериране на вграждане (6 доставчика)
-- `src/app/api/v1/images/generations/route.ts` — генериране на изображения (4+ доставчици, включително Antigravity/Nebius)
+- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true`
+- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers)
+- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius)
- `src/app/api/v1/messages/count_tokens/route.ts`
-- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — специален чат за всеки доставчик
-- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — специални вграждания за всеки доставчик
-- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — специални изображения за всеки доставчик
+- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat
+- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings
+- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images
- `src/app/api/v1beta/models/route.ts`
- `src/app/api/v1beta/models/[...path]/route.ts`
-Домейни за управление:
+Management domains:
-- Удостоверяване/настройки: `src/app/api/auth/*`, `src/app/api/settings/*`
-- Доставчици/връзки: `src/app/api/providers*`
-- Възли на доставчик: `src/app/api/provider-nodes*`
-- Персонализирани модели: `src/app/api/provider-models` (GET/POST/DELETE)
-- Каталог с модели: `src/app/api/models/route.ts` (GET)
-- Прокси конфигурация: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
+- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*`
+- Providers/connections: `src/app/api/providers*`
+- Provider nodes: `src/app/api/provider-nodes*`
+- Custom models: `src/app/api/provider-models` (GET/POST/DELETE)
+- Model catalog: `src/app/api/models/route.ts` (GET)
+- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
- OAuth: `src/app/api/oauth/*`
-- Ключове/псевдоними/комбота/цени: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing`
-- Използване: `src/app/api/usage/*`
-- Синхронизиране/облак: `src/app/api/sync/*`, `src/app/api/cloud/*`
-- Помощни инструменти за CLI: `src/app/api/cli-tools/*`
-- IP филтър: `src/app/api/settings/ip-filter` (GET/PUT)
-- Мислен бюджет: `src/app/api/settings/thinking-budget` (GET/PUT)
-- Системна подкана: `src/app/api/settings/system-prompt` (GET/PUT)
-- Сесии: `src/app/api/sessions` (GET)
-- Ограничения на скоростта: `src/app/api/rate-limits` (GET)
-- Устойчивост: `src/app/api/resilience` (GET/PATCH) — профили на доставчика, прекъсвач, състояние на ограничение на скоростта
-- Нулиране на устойчивостта: `src/app/api/resilience/reset` (POST) — прекъсвачи за нулиране + охлаждане
-- Кеш статистики: `src/app/api/cache/stats` (GET/DELETE)
-- Наличност на модела: `src/app/api/models/availability` (GET/POST)
-- Телеметрия: `src/app/api/telemetry/summary` (GET)
-- Бюджет: `src/app/api/usage/budget` (GET/POST)
-- Резервни вериги: `src/app/api/fallback/chains` (GET/POST/DELETE)
-- Одит на съответствието: `src/app/api/compliance/audit-log` (GET)
+- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing`
+- Usage: `src/app/api/usage/*`
+- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*`
+- CLI tooling helpers: `src/app/api/cli-tools/*`
+- IP filter: `src/app/api/settings/ip-filter` (GET/PUT)
+- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT)
+- System prompt: `src/app/api/settings/system-prompt` (GET/PUT)
+- Sessions: `src/app/api/sessions` (GET)
+- Rate limits: `src/app/api/rate-limits` (GET)
+- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state
+- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns
+- Cache stats: `src/app/api/cache/stats` (GET/DELETE)
+- Model availability: `src/app/api/models/availability` (GET/POST)
+- Telemetry: `src/app/api/telemetry/summary` (GET)
+- Budget: `src/app/api/usage/budget` (GET/POST)
+- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE)
+- Compliance audit: `src/app/api/compliance/audit-log` (GET)
- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET)
-- Правила: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core
+- Policies: `src/app/api/policies` (GET/POST)
-Основни модули на потока:
+## 2) SSE + Translation Core
-- Запис: `src/sse/handlers/chat.ts`
-- Оркестрация на ядрото: `open-sse/handlers/chatCore.ts`
-- Адаптери за изпълнение на доставчик: `open-sse/executors/*`
-- Откриване на формат/конфигурация на доставчик: `open-sse/services/provider.ts`
-- Разбор/разрешаване на модела: `src/sse/services/model.ts`, `open-sse/services/model.ts`
-- Логика за резервен акаунт: `open-sse/services/accountFallback.ts`
-- Регистър на преводите: `open-sse/translator/index.ts`
-- Трансформации на потока: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
-- Извличане/нормализиране на използването: `open-sse/utils/usageTracking.ts`
-- Анализатор на мислен етикет: `open-sse/utils/thinkTagParser.ts`
-- Манипулатор за вграждане: `open-sse/handlers/embeddings.ts`
-- Регистър на доставчика на вграждане: `open-sse/config/embeddingRegistry.ts`
-- Манипулатор за генериране на изображения: `open-sse/handlers/imageGeneration.ts`
-- Регистър на доставчика на изображения: `open-sse/config/imageRegistry.ts`
-- Дезинфекция на отговора: `open-sse/handlers/responseSanitizer.ts`
-- Нормализация на ролята: `open-sse/services/roleNormalizer.ts`
+Main flow modules:
-Услуги (бизнес логика):
+- Entry: `src/sse/handlers/chat.ts`
+- Core orchestration: `open-sse/handlers/chatCore.ts`
+- Provider execution adapters: `open-sse/executors/*`
+- Format detection/provider config: `open-sse/services/provider.ts`
+- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts`
+- Account fallback logic: `open-sse/services/accountFallback.ts`
+- Translation registry: `open-sse/translator/index.ts`
+- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
+- Usage extraction/normalization: `open-sse/utils/usageTracking.ts`
+- Think tag parser: `open-sse/utils/thinkTagParser.ts`
+- Embedding handler: `open-sse/handlers/embeddings.ts`
+- Embedding provider registry: `open-sse/config/embeddingRegistry.ts`
+- Image generation handler: `open-sse/handlers/imageGeneration.ts`
+- Image provider registry: `open-sse/config/imageRegistry.ts`
+- Response sanitization: `open-sse/handlers/responseSanitizer.ts`
+- Role normalization: `open-sse/services/roleNormalizer.ts`
-- Избор/точкуване на акаунт: `open-sse/services/accountSelector.ts`
-- Управление на жизнения цикъл на контекста: `open-sse/services/contextManager.ts`
-- Налагане на IP филтър: `open-sse/services/ipFilter.ts`
-- Проследяване на сесии: `open-sse/services/sessionManager.ts`
-- Дедупликация на заявка: `open-sse/services/signatureCache.ts`
-- Инжектиране на системна подкана: `open-sse/services/systemPrompt.ts`
-- Мислещо управление на бюджета: `open-sse/services/thinkingBudget.ts`
-- Маршрутизиране на модел с заместващи символи: `open-sse/services/wildcardRouter.ts`
-- Управление на лимита на скоростта: `open-sse/services/rateLimitManager.ts`
-- Прекъсвач: `open-sse/services/circuitBreaker.ts`
+Services (business logic):
-Модули на ниво домейн:
+- Account selection/scoring: `open-sse/services/accountSelector.ts`
+- Context lifecycle management: `open-sse/services/contextManager.ts`
+- IP filter enforcement: `open-sse/services/ipFilter.ts`
+- Session tracking: `open-sse/services/sessionManager.ts`
+- Request deduplication: `open-sse/services/signatureCache.ts`
+- System prompt injection: `open-sse/services/systemPrompt.ts`
+- Thinking budget management: `open-sse/services/thinkingBudget.ts`
+- Wildcard model routing: `open-sse/services/wildcardRouter.ts`
+- Rate limit management: `open-sse/services/rateLimitManager.ts`
+- Circuit breaker: `open-sse/services/circuitBreaker.ts`
+- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy
+- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions
-- Наличност на модела: `src/lib/domain/modelAvailability.ts`
-- Правила/бюджети за разходи: `src/lib/domain/costRules.ts`
-- Резервна политика: `src/lib/domain/fallbackPolicy.ts`
-- Комбо резолвер: `src/lib/domain/comboResolver.ts`
-- Политика за блокиране: `src/lib/domain/lockoutPolicy.ts`
-- Механизъм за правила: `src/domain/policyEngine.ts` — централизирано блокиране → бюджет → резервна оценка
-- Каталог с кодове за грешки: `src/lib/domain/errorCodes.ts`
-- ID на заявката: `src/lib/domain/requestId.ts`
-- Време за изчакване на извличане: `src/lib/domain/fetchTimeout.ts`
-- Заявка за телеметрия: `src/lib/domain/requestTelemetry.ts`
-- Съответствие/одит: `src/lib/domain/compliance/index.ts`
-- Изпълнител на оценка: `src/lib/domain/evalRunner.ts`
-- Устойчивост на състоянието на домейна: `src/lib/db/domainState.ts` — SQLite CRUD за резервни вериги, бюджети, история на разходите, състояние на блокиране, прекъсвачи
+Domain layer modules:
-Модули за доставчик на OAuth (12 отделни файла под `src/lib/oauth/providers/`):
+- Model availability: `src/lib/domain/modelAvailability.ts`
+- Cost rules/budgets: `src/lib/domain/costRules.ts`
+- Fallback policy: `src/lib/domain/fallbackPolicy.ts`
+- Combo resolver: `src/lib/domain/comboResolver.ts`
+- Lockout policy: `src/lib/domain/lockoutPolicy.ts`
+- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation
+- Error codes catalog: `src/lib/domain/errorCodes.ts`
+- Request ID: `src/lib/domain/requestId.ts`
+- Fetch timeout: `src/lib/domain/fetchTimeout.ts`
+- Request telemetry: `src/lib/domain/requestTelemetry.ts`
+- Compliance/audit: `src/lib/domain/compliance/index.ts`
+- Eval runner: `src/lib/domain/evalRunner.ts`
+- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers
-- Индекс на регистъра: `src/lib/oauth/providers/index.ts`
-- Индивидуални доставчици: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
-- Тънка обвивка: `src/lib/oauth/providers.ts` — повторно експортиране от отделни модули## 3) Persistence Layer
+OAuth provider modules (12 individual files under `src/lib/oauth/providers/`):
-Основно състояние DB (SQLite):
+- Registry index: `src/lib/oauth/providers/index.ts`
+- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
+- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules
-- Основна информация: `src/lib/db/core.ts` (better-sqlite3, миграции, WAL)
-- Повторно експортиране на фасада: `src/lib/localDb.ts` (тънък слой за съвместимост за повикващите)
-- файл: `${DATA_DIR}/storage.sqlite` (или `$XDG_CONFIG_HOME/omniroute/storage.sqlite`, когато е зададено, иначе `~/.omniroute/storage.sqlite`)
-- обекти (таблици + KV пространства от имена): providerConnections, providerNodes, modelAliases, комбинации, apiKeys, настройки, ценообразуване,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt**
+## 3) Persistence Layer
-Устойчивост на употреба:
+Primary state DB (SQLite):
-- фасада: `src/lib/usageDb.ts` (декомпозирани модули в `src/lib/usage/*`)
-- SQLite таблици в `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs`
-- незадължителните файлови артефакти остават за съвместимост/отстраняване на грешки (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `
/logs/...`)
-- наследените JSON файлове се мигрират към SQLite чрез миграции при стартиране, когато има такива
+- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL)
+- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers)
+- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`)
+- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt**
-DB на състоянието на домейна (SQLite):
+Usage persistence:
-- `src/lib/db/domainState.ts` — CRUD операции за състояние на домейн
-- Таблици (създадени в `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
-- Модел на кеша за запис: Картите в паметта са авторитетни по време на изпълнение; мутациите се записват синхронно в SQLite; състоянието се възстановява от DB при студен старт## 4) Auth + Security Surfaces
+- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`)
+- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs`
+- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`)
+- legacy JSON files are migrated to SQLite by startup migrations when present
-- Удостоверяване на бисквитките на таблото за управление: `src/proxy.ts`, `src/app/api/auth/login/route.ts`
-- Генериране/проверка на API ключ: `src/shared/utils/apiKey.ts`
-- Тайните на доставчика се запазват в записите `providerConnections`
-- Поддръжка на изходящ прокси чрез `open-sse/utils/proxyFetch.ts` (env vars) и `open-sse/utils/networkProxy.ts` (конфигурируем за всеки доставчик или глобален)## 5) Cloud Sync
+Domain State DB (SQLite):
-- Инициализация на Scheduler: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
-- Периодична задача: `src/shared/services/cloudSyncScheduler.ts`
-- Периодична задача: `src/shared/services/modelSyncScheduler.ts`
-- Контролен маршрут: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`)
+- `src/lib/db/domainState.ts` — CRUD operations for domain state
+- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
+- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start
+
+## 4) Auth + Security Surfaces
+
+- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts`
+- API key generation/verification: `src/shared/utils/apiKey.ts`
+- Provider secrets persisted in `providerConnections` entries
+- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global)
+
+## 5) Cloud Sync
+
+- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
+- Periodic task: `src/shared/services/cloudSyncScheduler.ts`
+- Periodic task: `src/shared/services/modelSyncScheduler.ts`
+- Control route: `src/app/api/sync/cloud/route.ts`
+
+## Request Lifecycle (`/v1/chat/completions`)
```mermaid
sequenceDiagram
@@ -338,7 +363,9 @@ flowchart TD
Q -- No --> R[Return all unavailable]
```
-Резервните решения се управляват от `open-sse/services/accountFallback.ts`, като се използват кодове за състояние и евристика за съобщения за грешка. Комбинираното маршрутизиране добавя един допълнителен предпазител: 400-те с обхват на доставчика, като неизправности при блокиране на съдържание нагоре и проверка на роли, се третират като неизправности в локален модел, така че по-късните комбинирани цели все още могат да се изпълняват.## OAuth Onboarding and Token Refresh Lifecycle
+Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run.
+
+## OAuth Onboarding and Token Refresh Lifecycle
```mermaid
sequenceDiagram
@@ -368,7 +395,9 @@ sequenceDiagram
Test-->>UI: validation result
```
-Опресняването по време на трафик на живо се изпълнява вътре в `open-sse/handlers/chatCore.ts` чрез изпълнител `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable)
+Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`.
+
+## Cloud Sync Lifecycle (Enable / Sync / Disable)
```mermaid
sequenceDiagram
@@ -400,7 +429,9 @@ sequenceDiagram
Sync-->>UI: disabled
```
-Периодичното синхронизиране се задейства от „CloudSyncScheduler“, когато облакът е активиран.## Data Model and Storage Map
+Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled.
+
+## Data Model and Storage Map
```mermaid
erDiagram
@@ -501,12 +532,14 @@ erDiagram
}
```
-Файлове за физическо съхранение:
+Physical storage files:
-- основна база данни за изпълнение: `${DATA_DIR}/storage.sqlite`
-- Редове за заявка: `${DATA_DIR}/log.txt` (компат/дебъг артефакт)
-- структурирани архиви на полезния товар на повикванията: `${DATA_DIR}/call_logs/`
-- незадължителни сесии за преводач/заявка за отстраняване на грешки: `/logs/...`## Deployment Topology
+- primary runtime DB: `${DATA_DIR}/storage.sqlite`
+- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact)
+- structured call payload archives: `${DATA_DIR}/call_logs/`
+- optional translator/request debug sessions: `/logs/...`
+
+## Deployment Topology
```mermaid
flowchart LR
@@ -541,205 +574,249 @@ flowchart LR
### Route and API Modules
-- `src/app/api/v1/*`, `src/app/api/v1beta/*`: API за съвместимост
-- `src/app/api/v1/providers/[provider]/*`: специални маршрути за всеки доставчик (чат, вграждания, изображения)
-- `src/app/api/providers*`: доставчик CRUD, валидиране, тестване
-- `src/app/api/provider-nodes*`: персонализирано съвместимо управление на възли
-- `src/app/api/provider-models`: персонализирано управление на модела (CRUD)
-- `src/app/api/models/route.ts`: API за каталог на модели (псевдоними + потребителски модели)
-- `src/app/api/oauth/*`: потоци OAuth/код на устройство
-- `src/app/api/keys*`: жизнен цикъл на местен API ключ
-- `src/app/api/models/alias`: управление на псевдоними
-- `src/app/api/combos*`: резервно управление на комбо
-- `src/app/api/pricing`: отменя ценообразуването за изчисляване на разходите
-- `src/app/api/settings/proxy`: конфигурация на прокси (GET/PUT/DELETE)
-- `src/app/api/settings/proxy/test`: тест за свързване на изходящ прокси (POST)
-- `src/app/api/usage/*`: API за използване и регистрационни файлове
-- `src/app/api/sync/*` + `src/app/api/cloud/*`: облачно синхронизиране и помощници в облака
-- `src/app/api/cli-tools/*`: локални CLI конфигурационни писатели/проверки
-- `src/app/api/settings/ip-filter`: списък с разрешени/блокирани IP адреси (GET/PUT)
-- `src/app/api/settings/thinking-budget`: конфигурация на бюджета на мислещия токен (GET/PUT)
-- `src/app/api/settings/system-prompt`: глобална системна подкана (GET/PUT)
-- `src/app/api/sessions`: списък на активни сесии (GET)
-- `src/app/api/rate-limits`: състояние на ограничение на скоростта за всеки акаунт (GET)### Routing and Execution Core
+- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs
+- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images)
+- `src/app/api/providers*`: provider CRUD, validation, testing
+- `src/app/api/provider-nodes*`: custom compatible node management
+- `src/app/api/provider-models`: custom model management (CRUD)
+- `src/app/api/models/route.ts`: model catalog API (aliases + custom models)
+- `src/app/api/oauth/*`: OAuth/device-code flows
+- `src/app/api/keys*`: local API key lifecycle
+- `src/app/api/models/alias`: alias management
+- `src/app/api/combos*`: fallback combo management
+- `src/app/api/pricing`: pricing overrides for cost calculation
+- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE)
+- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST)
+- `src/app/api/usage/*`: usage and logs APIs
+- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers
+- `src/app/api/cli-tools/*`: local CLI config writers/checkers
+- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT)
+- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT)
+- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT)
+- `src/app/api/sessions`: active session listing (GET)
+- `src/app/api/rate-limits`: per-account rate limit status (GET)
-- `src/sse/handlers/chat.ts`: анализ на заявка, комбо обработка, цикъл за избор на акаунт
-- `open-sse/handlers/chatCore.ts`: превод, изпращане на изпълнителя, повторен опит/опресняване, настройка на потока
-- `open-sse/executors/*`: специфично за доставчика поведение на мрежа и формат### Translation Registry and Format Converters
+### Routing and Execution Core
-- `open-sse/translator/index.ts`: регистър на преводачите и оркестрация
-- Заявка за преводачи: `open-sse/translator/request/*`
-- Преводачи на отговор: `open-sse/translator/response/*`
-- Форматни константи: `open-sse/translator/formats.ts`### Persistence
+- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop
+- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup
+- `open-sse/executors/*`: provider-specific network and format behavior
-- `src/lib/db/*`: постоянна конфигурация/състояние и устойчивост на домейн на SQLite
-- `src/lib/localDb.ts`: повторно експортиране на съвместимост за DB модули
-- `src/lib/usageDb.ts`: хронология на използването/фасада на регистрационните файлове на повикванията върху SQLite таблици## Provider Executor Coverage (Strategy Pattern)
+### Translation Registry and Format Converters
-Всеки доставчик има специализиран изпълнител, разширяващ `BaseExecutor` (в `open-sse/executors/base.ts`), който осигурява изграждане на URL адрес, изграждане на заглавка, повторен опит с експоненциално забавяне, кукички за опресняване на идентификационни данни и метода за оркестрация `execute()`.
+- `open-sse/translator/index.ts`: translator registry and orchestration
+- Request translators: `open-sse/translator/request/*`
+- Response translators: `open-sse/translator/response/*`
+- Format constants: `open-sse/translator/formats.ts`
-| Изпълнител | Доставчик(и) | Специална обработка |
-| ---------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------------------------------------------------- |
-| `Изпълнител по подразбиране` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Конфигурация на динамичен URL/заглавие за доставчик |
-| `AntigravityExecutor` | Google Антигравитация | Идентификационни номера на персонализирани проекти/сесии, повторен опит след анализ |
-| `CodexExecutor` | OpenAI Codex | Вкарва системни инструкции, принуждава усилие за разсъждение |
-| `CursorExecutor` | Курсор IDE | ConnectRPC протокол, Protobuf кодиране, подписване на заявка чрез контролна сума |
-| `GithubExecutor` | Копилот на GitHub | Опресняване на Copilot token, заглавки, имитиращи VSCode |
-| `KiroExecutor` | AWS CodeWhisperer/Киро | AWS EventStream двоичен формат → SSE конвертиране |
-| `GeminiCLIExecutor` | Gemini CLI | Цикъл на опресняване на Google OAuth токен |
+### Persistence
-Всички други доставчици (включително персонализирани съвместими възли) използват `DefaultExecutor`.## Provider Compatibility Matrix
+- `src/lib/db/*`: persistent config/state and domain persistence on SQLite
+- `src/lib/localDb.ts`: compatibility re-export for DB modules
+- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables
-| Доставчик | Формат | Удостоверяване | Поток | Непоточно | Опресняване на токена | API за използване |
-| ----------------- | --------------- | ------------------------------ | ---------------- | --------- | --------------------- | ---------------------------- | ------------------------------ |
-| Клод | Клод | API ключ / OAuth | ✅ | ✅ | ✅ | ⚠️ Само администратор |
-| Близнаци | близнаци | API ключ / OAuth | ✅ | ✅ | ✅ | ⚠️ Облачна конзола |
-| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Облачна конзола |
-| Антигравитация | антигравитация | OAuth | ✅ | ✅ | ✅ | ✅ API с пълна квота |
-| OpenAI | openai | API ключ | ✅ | ✅ | ❌ | ❌ |
-| Кодекс | openai-отговори | OAuth | ✅ принуден | ❌ | ✅ | ✅ Ограничения на скоростта |
-| Копилот на GitHub | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Моментни снимки на квоти |
-| Курсор | курсор | Персонализирана контролна сума | ✅ | ✅ | ❌ | ❌ |
-| Киро | киро | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Ограничения за използване |
-| Куен | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ По заявка |
-| Qoder | openai | OAuth (основен) | ✅ | ✅ | ✅ | ⚠️ По заявка |
-| OpenRouter | openai | API ключ | ✅ | ✅ | ❌ | ❌ |
-| GLM/Кими/МиниМакс | Клод | API ключ | ✅ | ✅ | ❌ | ❌ |
-| DeepSeek | openai | API ключ | ✅ | ✅ | ❌ | ❌ |
-| Groq | openai | API ключ | ✅ | ✅ | ❌ | ❌ |
-| xAI (Grok) | openai | API ключ | ✅ | ✅ | ❌ | ❌ |
-| Мистрал | openai | API ключ | ✅ | ✅ | ❌ | ❌ |
-| Недоумение | openai | API ключ | ✅ | ✅ | ❌ | ❌ |
-| Заедно AI | openai | API ключ | ✅ | ✅ | ❌ | ❌ |
-| Фойерверки AI | openai | API ключ | ✅ | ✅ | ❌ | ❌ |
-| Мозъци | openai | API ключ | ✅ | ✅ | ❌ | ❌ |
-| Cohere | openai | API ключ | ✅ | ✅ | ❌ | ❌ |
-| NVIDIA NIM | openai | API ключ | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage |
+## Provider Executor Coverage (Strategy Pattern)
-Откритите изходни формати включват:
+Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method.
-- `опенай`
-- `openaj-отговори`
-- „Клод“.
-- "близнаци".
+| Executor | Provider(s) | Special Handling |
+| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- |
+| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider |
+| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing |
+| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort |
+| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum |
+| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers |
+| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion |
+| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle |
-Целевите формати включват:
+All other providers (including custom compatible nodes) use the `DefaultExecutor`.
-- OpenAI чат/Отговори
-- Клод
-- Gemini/Gemini-CLI/Антигравитационен плик
-- Киро
-- Курсор
+## Provider Compatibility Matrix
-Преводите използват**OpenAI като хъб формат**— всички реализации преминават през OpenAI като междинен:```
+| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API |
+| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ |
+| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only |
+| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API |
+| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits |
+| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots |
+| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ |
+| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits |
+| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request |
+| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request |
+| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ |
+| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+
+## Format Translation Coverage
+
+Detected source formats include:
+
+- `openai`
+- `openai-responses`
+- `claude`
+- `gemini`
+
+Target formats include:
+
+- OpenAI chat/Responses
+- Claude
+- Gemini/Gemini-CLI/Antigravity envelope
+- Kiro
+- Cursor
+
+Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate:
+
+```
Source Format → OpenAI (hub) → Target Format
+```
-````
+Translations are selected dynamically based on source payload shape and provider target format.
-Преводите се избират динамично въз основа на формата на изходния полезен товар и целевия формат на доставчика.
+Additional processing layers in the translation pipeline:
-Допълнителни слоеве за обработка в тръбопровода за превод:
+- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance
+- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE)
+- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field
+- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema`
--**Дефектиране на отговора**— Премахва нестандартните полета от отговорите във формат OpenAI (както стрийминг, така и без стрийминг), за да се гарантира стриктно съответствие с SDK
--**Нормализиране на ролята**— Преобразува `developer` → `system` за цели, които не са OpenAI; обединява `system` → `user` за модели, които отхвърлят системната роля (GLM, ERNIE)
--**Извличане на мислен етикет**— Анализира `...` блокове от съдържание в полето `reasoning_content`
--**Структуриран изход**— Преобразува OpenAI `response_format.json_schema` в `responseMimeType` + `responseSchema` на Gemini## Supported API Endpoints
+## Supported API Endpoints
-| Крайна точка | Формат | Манипулатор |
-| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------ |
-| `POST /v1/chat/completions` | OpenAI чат | `src/sse/handlers/chat.ts` |
-| `POST /v1/messages` | Съобщения на Клод | Същият манипулатор (автоматично разпознат) |
-| `POST /v1/responses` | OpenAI отговори | `open-sse/handlers/responsesHandler.ts` |
-| `POST /v1/вграждания` | OpenAI вграждания | `open-sse/handlers/embeddings.ts` |
-| `GET /v1/вграждания` | Списък на модели | API маршрут |
-| `POST /v1/images/generations` | OpenAI изображения | `open-sse/handlers/imageGeneration.ts` |
-| `GET /v1/images/generations` | Списък на модели | API маршрут |
-| `POST /v1/providers/{provider}/chat/completions` | OpenAI чат | Специализиран за всеки доставчик с валидиране на модел |
-| `POST /v1/providers/{provider}/embeddings` | OpenAI вграждания | Специализиран за всеки доставчик с валидиране на модел |
-| `POST /v1/providers/{provider}/images/generations` | OpenAI изображения | Специализиран за всеки доставчик с валидиране на модел |
-| `POST /v1/messages/count_tokens` | Клод Токен Брой | API маршрут |
-| `GET /v1/models` | Списък с модели на OpenAI | API маршрут (чат + вграждане + изображение + потребителски модели) |
-| `GET /api/models/catalog` | Каталог | Всички модели, групирани по доставчик + тип |
-| `POST /v1beta/models/*:streamGenerateContent` | Роден Близнаци | API маршрут |
-| `GET/PUT/DELETE /api/settings/proxy` | Прокси конфигурация | Конфигурация на мрежов прокси |
-| `POST /api/settings/proxy/test` | Прокси свързаност | Крайна точка на теста за изправност/свързване на прокси |
-| `GET/POST/DELETE /api/provider-models` | Модели на доставчици | Метаданни за модела на доставчика, поддържащи персонализирани и управлявани налични модели |## Bypass Handler
+| Endpoint | Format | Handler |
+| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- |
+| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` |
+| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) |
+| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` |
+| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` |
+| `GET /v1/embeddings` | Model listing | API route |
+| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` |
+| `GET /v1/images/generations` | Model listing | API route |
+| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation |
+| `POST /v1/messages/count_tokens` | Claude Token Count | API route |
+| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) |
+| `GET /api/models/catalog` | Catalog | All models grouped by provider + type |
+| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route |
+| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration |
+| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint |
+| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models |
-Обходният манипулатор (`open-sse/utils/bypassHandler.ts`) прихваща известни заявки за "изхвърляне" от Claude CLI - пингове за загряване, извличане на заглавия и броене на токени - и връща**фалшив отговор**, без да консумира токени на доставчика нагоре по веригата. Това се задейства само когато `User-Agent` съдържа `claude-cli`.## Request Logger Pipeline
+## Bypass Handler
-Регистраторът на заявки (`open-sse/utils/requestLogger.ts`) осигурява 7-етапен тръбопровод за регистриране на грешки, деактивиран по подразбиране, активиран чрез `ENABLE_REQUEST_LOGS=true`:```
+The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`.
+
+## Request Logger Pipeline
+
+The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`:
+
+```
1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json
→ 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt
-````
+```
-Файловете се записват в `/logs//` за всяка сесия на заявка.## Failure Modes and Resilience
+Files are written to `/logs//` for each request session.
+
+## Failure Modes and Resilience
## 1) Account/Provider Availability
-- изчакване на акаунта на доставчика при преходни/скоростни/удостоверителни грешки
-- резервен акаунт преди неуспешна заявка
-- резервен комбиниран модел, когато пътят на текущия модел/доставчик е изчерпан## 2) Token Expiry
+- provider account cooldown on transient/rate/auth errors
+- account fallback before failing request
+- combo model fallback when current model/provider path is exhausted
-- предварителна проверка и опресняване с повторен опит за опресняващи доставчици
-- 401/403 повторен опит след опит за опресняване в основния път## 3) Stream Safety
+## 2) Token Expiry
-- контролер на потоци с прекъсване на връзката
-- поток за превод с промиване в края на потока и обработка на `[DONE]`
-- резервна оценка на използването, когато липсват метаданни за използване на доставчика## 4) Cloud Sync Degradation
+- pre-check and refresh with retry for refreshable providers
+- 401/403 retry after refresh attempt in core path
-- появяват се грешки при синхронизиране, но локалното изпълнение продължава
-- планировчикът има логика с възможност за повторен опит, но периодичното изпълнение в момента извиква синхронизиране с един опит по подразбиране## 5) Data Integrity
+## 3) Stream Safety
-- Миграции на SQLite схема и кукички за автоматично надграждане при стартиране
-- наследен JSON → път за съвместимост на миграцията на SQLite## Observability and Operational Signals
+- disconnect-aware stream controller
+- translation stream with end-of-stream flush and `[DONE]` handling
+- usage estimation fallback when provider usage metadata is missing
-Източници на видимост по време на изпълнение:
+## 4) Cloud Sync Degradation
-- регистрационни файлове на конзолата от `src/sse/utils/logger.ts`
-- агрегати за използване на заявка в SQLite (`usage_history`, `call_logs`, `proxy_logs`)
-- четиристепенно улавяне на подробен полезен товар в SQLite (`request_detail_logs`), когато `settings.detailed_logs_enabled=true`
-- текстов регистър на състоянието на заявката в `log.txt` (по избор/compat)
-- незадължителни дълбоки регистрационни файлове за заявка/превод под `logs/`, когато `ENABLE_REQUEST_LOGS=true`
-- крайни точки за използване на таблото за управление (`/api/usage/*`) за използване на UI
+- sync errors are surfaced but local runtime continues
+- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default
-Подробно улавяне на полезен товар на заявка съхранява до четири етапа на полезен товар в JSON на маршрутизирано повикване:
+## 5) Data Integrity
-- необработена заявка, получена от клиента
-- преведена заявка, действително изпратена нагоре
-- отговор на доставчика, реконструиран като JSON; поточно предаваните отговори се уплътняват до крайното резюме плюс метаданни на потока
-- окончателен клиентски отговор, върнат от OmniRoute; поточно предаваните отговори се съхраняват в същата компактна обобщена форма## Security-Sensitive Boundaries
+- SQLite schema migrations and auto-upgrade hooks at startup
+- legacy JSON → SQLite migration compatibility path
-- JWT тайна (`JWT_SECRET`) защитава проверката/подписването на бисквитките на сесията на таблото за управление
-- Първоначалната парола за зареждане (`INITIAL_PASSWORD`) трябва да бъде изрично конфигурирана за осигуряване при първо стартиране
-- API ключ HMAC тайна (`API_KEY_SECRET`) защитава генерирания локален формат на API ключ
-- Тайните на доставчика (API ключове/токени) се съхраняват в локалната база данни и трябва да бъдат защитени на ниво файлова система
-- Крайните точки за синхронизиране в облак разчитат на удостоверяване на API ключ + семантика на идентификатор на машина## Environment and Runtime Matrix
+## Observability and Operational Signals
-Променливите на средата, използвани активно от кода:
+Runtime visibility sources:
-- Приложение/удостоверяване: `JWT_SECRET`, `INITIAL_PASSWORD`
-- Съхранение: `DATA_DIR`
-- Съвместимо поведение на възел: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
-- Допълнителна отмяна на базата за съхранение (Linux/macOS, когато `DATA_DIR` не е зададено): `XDG_CONFIG_HOME`
-- Хеширане на сигурността: `API_KEY_SECRET`, `MACHINE_ID_SALT`
-- Регистриране: `ENABLE_REQUEST_LOGS`
-- Синхронизиране/облачно URL адресиране: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
-- Изходящ прокси: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` и варианти с малки букви
-- Флагове на функцията SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
-- Помощници за платформа/време на изпълнение (не специфична за приложението конфигурация): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes
+- console logs from `src/sse/utils/logger.ts`
+- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`)
+- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true`
+- textual request status log in `log.txt` (optional/compat)
+- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true`
+- dashboard usage endpoints (`/api/usage/*`) for UI consumption
-1. `usageDb` и `localDb` споделят една и съща основна политика за директория (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) с мигриране на наследени файлове.
-2. `/api/v1/route.ts` делегира на същия унифициран конструктор на каталог, използван от `/api/v1/models` (`src/app/api/v1/models/catalog.ts`), за да се избегне семантично отклонение.
-3. Request logger записва пълни заглавки/тяло, когато е разрешено; третира регистрационната директория като чувствителна.
-4. Поведението в облака зависи от правилния `NEXT_PUBLIC_BASE_URL` и достъпността на крайната точка на облака.
-5. Директорията `open-sse/` се публикува като `@omniroute/open-sse`**npm workspace package**. Изходният код го импортира чрез `@omniroute/open-sse/...` (разрешено от Next.js `transpilePackages`). Пътищата на файловете в този документ все още използват името на директорията `open-sse/` за последователност.
-6. Диаграмите в таблото за управление използват**Recharts**(базирани на SVG) за достъпни, интерактивни аналитични визуализации (стълбовидни диаграми на използването на модела, таблици с разбивка на доставчиците с проценти на успех).
-7. E2E тестовете използват**Playwright**(`tests/e2e/`), изпълняват се чрез `npm run test:e2e`. Единичните тестове използват**Node.js test runner**(`tests/unit/`), изпълняват се чрез `npm run test:unit`. Изходният код под `src/` е**TypeScript**(`.ts`/`.tsx`); работното пространство `open-sse/` остава JavaScript (`.js`).
-8. Страницата с настройки е организирана в 5 раздела: Сигурност, Маршрутизация (6 глобални стратегии: първо запълване, кръгова система, p2c, произволна, най-малко използвана, оптимизирана по отношение на разходите), Устойчивост (ограничения на скоростта за редактиране, прекъсвач, политики), AI (мислещ бюджет, системна подкана, кеш за подкана), Разширени (прокси).## Operational Verification Checklist
+Detailed request payload capture stores up to four JSON payload stages per routed call:
-- Изграждане от източник: `npm run build`
-- Изграждане на Docker изображение: `docker build -t omniroute .`
-- Стартирайте услугата и проверете:
+- raw request received from the client
+- translated request actually sent upstream
+- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata
+- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form
+
+## Security-Sensitive Boundaries
+
+- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing
+- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning
+- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format
+- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level
+- Cloud sync endpoints rely on API key auth + machine id semantics
+
+## Environment and Runtime Matrix
+
+Environment variables actively used by code:
+
+- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD`
+- Storage: `DATA_DIR`
+- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
+- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME`
+- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT`
+- Logging: `ENABLE_REQUEST_LOGS`
+- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
+- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants
+- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
+- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`
+
+## Known Architectural Notes
+
+1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration.
+2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift.
+3. Request logger writes full headers/body when enabled; treat log directory as sensitive.
+4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability.
+5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency.
+6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates).
+7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`).
+8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy).
+9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed.
+10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22.
+11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible.
+
+## Operational Verification Checklist
+
+- Build from source: `npm run build`
+- Build Docker image: `docker build -t omniroute .`
+- Start service and verify:
- `GET /api/settings`
- `GET /api/v1/models`
-- CLI целевият базов URL трябва да бъде `http://:20128/v1`, когато `PORT=20128`
+- CLI target base URL should be `http://:20128/v1` when `PORT=20128`
diff --git a/docs/i18n/bg/docs/FEATURES.md b/docs/i18n/bg/docs/FEATURES.md
index f337763114..a1f923db97 100644
--- a/docs/i18n/bg/docs/FEATURES.md
+++ b/docs/i18n/bg/docs/FEATURES.md
@@ -4,102 +4,168 @@
---
-Визуално ръководство за всеки раздел на таблото за управление OmniRoute.---
+
+
+Visual guide to every section of the OmniRoute dashboard.
+
+---
## 🔌 Providers
-Управлявайте връзките на доставчици на AI: OAuth доставчици (Claude Code, Codex, Gemini CLI), доставчици на API ключове (Groq, DeepSeek, OpenRouter) и безплатни доставчици (Qoder, Qwen, Kiro). Сметките в Kiro включват проследяване на кредитния баланс — оставащи кредити, обща надбавка и дата на подновяване, видими в Табло за управление → Използване.
+Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage.
+
+
---
## 🎨 Combos
-Създавайте комбинации за маршрутизиране на модели с 6 стратегии: приоритетни, претеглени, кръгови, произволни, най-малко използвани и оптимизирани по отношение на разходите. Всяка комбинация свързва няколко модела с автоматичен резервен вариант и включва бързи шаблони и проверки за готовност.
+Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks.
+
+
---
## 📊 Analytics
-Изчерпателни анализи на използването с потребление на токени, оценки на разходите, топлинни карти на активността, седмични диаграми на разпределение и разбивки по доставчик.
+Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns.
+
+
---
## 🏥 System Health
-Мониторинг в реално време: време на работа, памет, версия, процентили на латентност (p50/p95/p99), статистика на кеша и състояния на прекъсвача на доставчика.
+Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states.
+
+
---
## 🔧 Translator Playground
-Четири режима за отстраняване на грешки в API преводи:**Playground**(конвертор на формати),**Chat Tester**(заявки на живо),**Test Bench**(пакетни тестове) и**Live Monitor**(поток в реално време).
+Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream).
+
+
---
## 🎮 Model Playground _(v2.0.9+)_
-Тествайте всеки модел директно от таблото. Изберете доставчик, модел и крайна точка, пишете подкани с Monaco Editor, предавайте отговори в реално време, прекъсвайте по средата на потока и преглеждайте показатели за времето.---
+Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics.
+
+---
## 🎨 Themes _(v2.0.5+)_
-Цветови теми с възможност за персонализиране за цялото табло. Изберете от 7 предварително зададени цвята (корал, син, червен, зелен, виолетов, оранжев, циан) или създайте персонализирана тема, като изберете всеки шестнадесетичен цвят. Поддържа светъл, тъмен и системен режим.---
+Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode.
+
+---
## ⚙️ Settings
-Изчерпателен панел с настройки с раздели:
+Comprehensive settings panel with tabs:
--**Общи**— Системно съхранение, управление на архивиране (база данни за експорт/импорт) -**Външен вид**— Селектор на тема (тъмно/светло/система), предварително зададени цветови теми и персонализирани цветове, видимост на журнала за здраве, контроли за видимост на елементи от страничната лента -**Сигурност**— API защита на крайната точка, персонализирано блокиране на доставчика, IP филтриране, информация за сесията -**Маршрутизиране**— Псевдоними на модела, влошаване на фоновата задача -**Устойчивост**— Устойчивост на лимита на скоростта, настройка на прекъсвача, автоматично деактивиране на забранени акаунти, наблюдение на изтичане на доставчика -**Разширени**— Замени на конфигурацията, одитна пътека на конфигурацията, резервен режим на влошаване
+- **General** — System storage, backup management (export/import database)
+- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls
+- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info
+- **Routing** — Model aliases, background task degradation
+- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration
+- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode
+
+
---
## 🔧 CLI Tools
-Конфигурация с едно кликване за инструменти за кодиране на AI: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor и Factory Droid. Включва автоматизирано прилагане/нулиране на конфигурация, профили на свързване и картографиране на модела.
+One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping.
+
+
---
## 🤖 CLI Agents _(v2.0.11+)_
-Табло за откриване и управление на CLI агенти. Показва мрежа от 14 вградени агента (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) с:
+Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with:
--**Състояние на инсталацията**— Инсталирано / Не е намерено с откриване на версия -**Протоколни значки**— stdio, HTTP и др. -**Персонализирани агенти**— Регистрирайте всеки CLI инструмент чрез формуляр (име, двоичен файл, команда за версия, аргументи за генериране) -**CLI съпоставяне на пръстови отпечатъци**— Превключване за всеки доставчик, за да съответства на собствените подписи на CLI заявка, намалявайки риска от забрана, като същевременно запазва прокси IP---
+- **Installation status** — Installed / Not Found with version detection
+- **Protocol badges** — stdio, HTTP, etc.
+- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args)
+- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP
+
+---
+
+## 🔗 Context Relay _(v3.5.5+)_
+
+A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context.
+
+Configurable via combo-level or global settings:
+- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%)
+- **Max Messages For Summary** — How much recent history to condense
+- **Summary Model** — Optional override model for generating the handoff summary
+
+Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md).
+
+---
+
+## 🛡️ Proxy Hardening _(v3.5.5+)_
+
+Comprehensive proxy configuration enforcement across the entire request pipeline:
+
+- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments
+- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings
+- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22
+- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS
+
+---
## 🖼️ Media _(v2.0.3+)_
-Генерирайте изображения, видеоклипове и музика от таблото за управление. Поддържа OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open и MusicGen.---
+Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen.
+
+---
## 📝 Request Logs
-Регистриране на заявки в реално време с филтриране по доставчик, модел, акаунт и API ключ. Показва кодове за състояние, използване на токени, латентност и подробности за отговора.
+Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details.
+
+
---
## 🌐 API Endpoint
-Вашата унифицирана крайна точка на API с разбивка на възможностите: завършвания на чатове, API за отговори, вграждания, генериране на изображения, прекласиране, аудио транскрипция, текст към говор, модериране и регистрирани ключове за API. Интегриране на Cloudflare Quick Tunnel и поддръжка на облачен прокси за отдалечен достъп.
+Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access.
+
+
---
## 🔑 API Key Management
-Създаване, обхват и отмяна на API ключове. Всеки ключ може да бъде ограничен до конкретни модели/доставчици с пълен достъп или разрешения само за четене. Визуално управление на ключове с проследяване на използването.---
+Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking.
+
+---
## 📋 Audit Log
-Проследяване на административни действия с филтриране по тип действие, актьор, цел, IP адрес и клеймо за време. Пълна история на събитията за сигурност.---
+Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history.
+
+---
## 🖥️ Desktop Application
-Настолно приложение Native Electron за Windows, macOS и Linux. Стартирайте OmniRoute като самостоятелно приложение с интеграция в системната област, офлайн поддръжка, автоматично актуализиране и инсталиране с едно щракване.
+Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install.
-Ключови характеристики:
+Key features:
-- Проучване на готовността на сървъра (без празен екран при студен старт)
-- Системна област с управление на портове
-- Политика за сигурност на съдържанието
-- Еднократно заключване
-- Автоматична актуализация при рестартиране
-- Платформено условен потребителски интерфейс (светофари на MacOS, заглавна лента по подразбиране на Windows/Linux)
-- Hardened Electron build packaging — символично свързаните `node_modules` в самостоятелния пакет се откриват и отхвърлят преди опаковането, предотвратявайки зависимостта по време на изпълнение от машината за изграждане (v2.5.5+)
+- Server readiness polling (no blank screen on cold start)
+- System tray with port management
+- Content Security Policy
+- Single-instance lock
+- Auto-update on restart
+- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar)
+- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+)
-📖 Вижте [`electron/README.md`](../electron/README.md) за пълна документация.
+📖 See [`electron/README.md`](../electron/README.md) for full documentation.
diff --git a/docs/i18n/bg/docs/TROUBLESHOOTING.md b/docs/i18n/bg/docs/TROUBLESHOOTING.md
index e339228624..45718fe147 100644
--- a/docs/i18n/bg/docs/TROUBLESHOOTING.md
+++ b/docs/i18n/bg/docs/TROUBLESHOOTING.md
@@ -4,65 +4,148 @@
---
-Често срещани проблеми и решения за OmniRoute.---## Quick Fixes
-| Проблем | Решение |
-| ----------------------------------------------- | ------------------------------------------------------------------------------ | --------------------- |
-| Първото влизане не работи | Задайте `INITIAL_PASSWORD` в `.env` (без твърдо кодирано подразбиране) |
-| Таблото се отваря на грешен порт | Задайте `PORT=20128` и `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
-| Няма регистрирани файлове за заявки под `logs/` | Задайте `ENABLE_REQUEST_LOGS=true` |
-| EACCES: разрешението е показано | Задайте `DATA_DIR=/path/to/writable/dir` да замените `~/.omniroute` |
-| Стратегията за маршрутизиране не се запазва | Актуализация до v1.4.11+ (корекция на Zod схема за постоянство на настройките) | ---## Provider Issues |
+
+Common problems and solutions for OmniRoute.
+
+---
+
+## Quick Fixes
+
+| Problem | Solution |
+| ----------------------------- | ------------------------------------------------------------------ |
+| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) |
+| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
+| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` |
+| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` |
+| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) |
+| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below |
+| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below |
+
+---
+
+## Node.js Compatibility
+
+
+
+### Login page crashes or shows "Module self-registration" error
+
+**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database.
+
+**Symptoms:**
+- Login page shows a blank screen or a server error
+- Console shows `Error: Module did not self-register` or similar native binding errors
+- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected
+
+**Fix:**
+
+1. Install Node.js 22 LTS (recommended):
+ ```bash
+ nvm install 22
+ nvm use 22
+ ```
+2. Verify your version: `node --version` should show `v22.x.x`
+3. Reinstall OmniRoute: `npm install -g omniroute`
+4. Restart: `omniroute`
+
+> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**.
+
+---
+
+## Proxy Issues
+
+
+
+### Provider validation shows "fetch failed"
+
+**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing.
+
+**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically.
+
+### Token health check fails with "fetch failed"
+
+**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection.
+
+**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+.
+
+### SOCKS5 proxy returns "invalid onRequestStart method"
+
+**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation.
+
+**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+.
+
+---
+
+## Provider Issues
### "Language model did not provide messages"
-**Причина:**Квотата на доставчика е изчерпана.
+**Cause:** Provider quota exhausted.
-**Коригиране:**
+**Fix:**
-1. Проверете инструмента за проследяване на квотите на таблото за управление
-2. Използвайте комбо с резервни нива
-3. Преминете към по-евтино/безплатно ниво### Rate Limiting
+1. Check dashboard quota tracker
+2. Use a combo with fallback tiers
+3. Switch to cheaper/free tier
-**Причина:**Абонаментната квота е изчерпана.
+### Rate Limiting
-**Коригиране:**
+**Cause:** Subscription quota exhausted.
-- Добавете резервен вариант: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-- Използвайте GLM/MiniMax като евтино резервно копие### OAuth Token Expired
+**Fix:**
-OmniRoute автоматично опреснява токените. Ако проблемите продължават:
+- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Use GLM/MiniMax as cheap backup
-1. Табло → Доставчик → Свързване отново
-2. Изтрийте и добавете отново връзката с доставчика---## Cloud Issues
+### OAuth Token Expired
+
+OmniRoute auto-refreshes tokens. If issues persist:
+
+1. Dashboard → Provider → Reconnect
+2. Delete and re-add the provider connection
+
+---
+
+## Cloud Issues
### Cloud Sync Errors
-1. Проверете дали `BASE_URL` е към вашия работен екземпляр (напр. `http://localhost:20128`)
-2. Проверете дали `CLOUD_URL` е към вашата крайна точка в облака (напр. `https://omniroute.dev`)
-3. Поддържайте стойността `NEXT_PUBLIC_*` в съответствие със стойността от страната на сървъра### Cloud `stream=false` Връща 500
+1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`)
+2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`)
+3. Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**Симптом:**`Неочакван токен 'd'...` в крайната точка на облака за повикане без точно предаване.
+### Cloud `stream=false` Returns 500
-**Причина:**Upstream връща SSE полезен продукт, докато клиентът очаква JSON.
+**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls.
-**Заобиколно решение:**Използвайте `stream=true` за директни повикания в облака. Локалното време за изпълнение включва резервен SSE→JSON.### Cloud Says Connected but "Invalid API key"
+**Cause:** Upstream returns SSE payload while client expects JSON.
-1. Създайте нов ключ от локалното табло за управление (`/api/keys`)
-2. Стартирайте облачна синхронизация: Активирайте облака → Синхронизирай сега
-3. Старите/несинхронизираните ключове все още могат да връщат „401“ в облака---## Docker Issues
+**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback.
+
+### Cloud Says Connected but "Invalid API key"
+
+1. Create a fresh key from local dashboard (`/api/keys`)
+2. Run cloud sync: Enable Cloud → Sync Now
+3. Old/non-synced keys can still return `401` on cloud
+
+---
+
+## Docker Issues
### CLI Tool Shows Not Installed
-1. Проверете полетата за изпълнение: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
-2. За преносим режим: използвайте целево изображение `runner-cli` (пакетни CLI)
-3. За режим на монтиране на хост: задайте `CLI_EXTRA_PATHS` и монтирайте директорията bin на хоста като само за четене
-4. Ако `installed=true` и `runnable=false`: двоичният файл е намерен, но проверката на състоянието е неуспешна### Quick Runtime Validation```bash
- curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
- curl -s http://localhost:20128/api/cli-tools/claude-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
- curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
+1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
+2. For portable mode: use image target `runner-cli` (bundled CLIs)
+3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only
+4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck
-````
+### Quick Runtime Validation
+
+```bash
+curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
+curl -s http://localhost:20128/api/cli-tools/claude-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
+curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
+```
---
@@ -70,108 +153,160 @@ OmniRoute автоматично опреснява токените. Ако п
### High Costs
-1. Проверете статистическите данни за употреба в Табло → Използване
-2. Превключете основния модел на GLM/MiniMax
-3. Използвайте безплатно ниво (Gemini CLI, Qoder) за некритични задачи
-4. Задайте бюджети за разходи за API ключ: Табло за управление → API ключове → Бюджет---## Debugging
+1. Check usage stats in Dashboard → Usage
+2. Switch primary model to GLM/MiniMax
+3. Use free tier (Gemini CLI, Qoder) for non-critical tasks
+4. Set cost budgets per API key: Dashboard → API Keys → Budget
+
+---
+
+## Debugging
### Enable Request Logs
-Задайте `ENABLE_REQUEST_LOGS=true` във вашия `.env` файл. Дневниците се появяват в директорията `logs/`.### Проверете здравето на доставчика```bash
+Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory.
+
+### Check Provider Health
+
+```bash
# Health dashboard
http://localhost:20128/dashboard/health
# API health check
curl http://localhost:20128/api/monitoring/health
-````
+```
### Runtime Storage
-- Основно състояние: `${DATA_DIR}/storage.sqlite` (доставчици, комбинации, псевдоними, ключове, настройки)
-- Използване: SQLite таблици в `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + незадължително `${DATA_DIR}/log.txt` и `${DATA_DIR}/call_logs/`
-- Заявки за регистрационни файлове: `/logs/...` (като `ENABLE_REQUEST_LOGS=true`)---## Circuit Breaker Issues
+- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings)
+- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/`
+- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`)
+
+---
+
+## Circuit Breaker Issues
### Provider stuck in OPEN state
-При прекъсване на веригата на доставчика е ОТВОРЕЕН, заявките се блокират, докато изтече времето за охлаждане.
+When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires.
-**Коригиране:**
+**Fix:**
-1. Отидете на**Табло → Настройки → Устойчивост**
-2. Проверете картата на прекъсвача на сървъра на доставчика
-3. Щракнете върху**Нулиране на всички**, за да изчистите всички прекъсвачи, или изчакайте времето за охлаждане да изтече
-4. Уверете се, че доставчикът действително е наличен, преди да нулира### Доставчикът продължава да изключва прекъсвача
+1. Go to **Dashboard → Settings → Resilience**
+2. Check the circuit breaker card for the affected provider
+3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire
+4. Verify the provider is actually available before resetting
-Ако доставчикът многократно влезе в ОТВОРЕНО състояние:
+### Provider keeps tripping the circuit breaker
-1. Проверете**Таблото → Здраве → Здраве на доставчика**за модел на повреда
-2. Отидете на**Настройки → Устойчивост → Профили на доставчици**и увеличите прага на отказ
-3. Проверете дали доставчикът е променил ограниченията на API или изисква повторно удостоверяване
-4. Прегледайте телеметрията за латентност — високата латентност може да причини грешки, базирани на изчакване---## Audio Transcription Issues
+If a provider repeatedly enters OPEN state:
+
+1. Check **Dashboard → Health → Provider Health** for the failure pattern
+2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold
+3. Check if the provider has changed API limits or requires re-authentication
+4. Review latency telemetry — high latency may cause timeout-based failures
+
+---
+
+## Audio Transcription Issues
### "Unsupported model" error
-- Уверете се, че използвате правилния префикс: `deepgram/nova-3` или `assemblyai/best`
-- Проверете дали доставчикът е свързан в**Табло → Доставчици**### Транскрипцията се връща празна или е неуспешна
+- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best`
+- Verify the provider is connected in **Dashboard → Providers**
-- Проверете поддържаните аудио формати: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm`
-- Уверете се, че размерът на файла е в границите на доставчика (обикновено < 25MB)
-- Проверете валидността на API ключа на доставчика в картата на доставчика---## Translator Debugging
+### Transcription returns empty or fails
-Използвайте**Таблица за управление → Преводач**за отстраняване на грешки при проблеми с превод на формат:
+- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm`
+- Verify file size is within provider limits (typically < 25MB)
+- Check provider API key validity in the provider card
-| Режим | Кога да използвате |
-| ------------------- | ------------------------------------------------------------------------------------------------------------- | ------------------------ |
-| **Детска площадка** | Сравнете входно/изходните формати един до друг — поставете неуспешна заявка, за да видите как се превежда |
-| **Чат тестер** | Изпращайте съобщения на живо и проверете допълнителен полезен продукт на заявка/отговор, включително заглавки |
-| **Тестова стенда** | Изпълнете групови тестове в комбинации от формати, за да откриете кои преводи са нарушени |
-| **Монитор на живо** | Гледайте потока на заявките в реално време, за да уловите периодични проблеми с превод | ### Common format issues |
+---
--**Тагове за мислене не се появяват**— Проверете дали целевият доставчик поддържа мисленето и настройката на бюджета за мислене -**Отпадане на извикванията на инструментите**— Някои преводи на формати могат да премахнат неподдържаните полета; потвърдете в режим Playground -**Липсва системна подкана**— Клод и Джемини обработват системните подкани по различен начин; проверка на резултата за превод -**SDK връща необработен низ вместо валиден обект**— Коригирано във v1.1.0: дезинфектантът за отговор вече премахва нестандартните полета (`x_groq`, `usage_breakdown` и т.н.), които предизвикват неуспешно пускане на OpenAI SDK Pydantic -**GLM/ERNIE отхвърля `системна` роля**— Коригирано във v1.1.0: нормализаторът на ролите автоматично обединява системни съобщения в потребителски съобщения за несъвместими модели
+## Translator Debugging
-- Ролята на**`разработчик` не е разпозната**— Коригирано във v1.1.0: автоматично се преобразува в `системата` за доставчици, не е с OpenAI -**`json_schema` не работи с Gemini**— Коригирано във v1.1.0: `response_format` вече се преобразува в `responseMimeType` + `responseSchema` на Gemini---## Resilience Settings
+Use **Dashboard → Translator** to debug format translation issues:
+
+| Mode | When to Use |
+| ---------------- | -------------------------------------------------------------------------------------------- |
+| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates |
+| **Chat Tester** | Send live messages and inspect the full request/response payload including headers |
+| **Test Bench** | Run batch tests across format combinations to find which translations are broken |
+| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues |
+
+### Common format issues
+
+- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting
+- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode
+- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output
+- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures
+- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models
+- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers
+- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema`
+
+---
+
+## Resilience Settings
### Auto rate-limit not triggering
-- Автоматичното ограничение на скоростта се прилага само за доставчици на API ключове (без OAuth/абонамент)
-- Уверете се, че**Настройки → Устойчивост → Профили на доставчици**има активиран автоматичен лимит на скоростта
-- Проверете дали доставчикът връща кодове за състояние `429` или заглавки `Retry-After`### Tuning exponential backoff
+- Auto rate-limit only applies to API key providers (not OAuth/subscription)
+- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled
+- Check if the provider returns `429` status codes or `Retry-After` headers
-Профилът на доставчика поддържа тези настройки:
+### Tuning exponential backoff
--**Базово плащане**— Първоначално време на изчакване след първото увреждане (по подразбиране: 1s) -**Максимално заплащане**— Максимално ограничение на времето за изчакване (по подразбиране: 30 секунди) -**Множител**— Колко да се увеличи закъснението за последователен отказ (по подразбиране: 2x)### Anti-thundering herd
+Provider profiles support these settings:
-Когато много едновременни заявки се появяват на доставчика с ограничена скорост, OmniRoute използва mutex + автоматично регулиране на скоростта, за да сериализира заявките и да предотврати каскадни грешки. Това е автоматично за доставчиците на API ключове.---## Optional RAG / LLM failure taxonomy (16 problems)
+- **Base delay** — Initial wait time after first failure (default: 1s)
+- **Max delay** — Maximum wait time cap (default: 30s)
+- **Multiplier** — How much to increase delay per consecutive failure (default: 2x)
-Някои потребители на OmniRoute поставят шлюза пред RAG или агент стекове. В тези настройки е обичайно да се вижда отстранен модел: OmniRoute изглежда здрав (доставчиците работят, профилите за маршрутизиране са добри, няма предупреждения за ограничение на скоростта), но крайният отговор е още по-грешен.
+### Anti-thundering herd
-На практика тези инциденти идват от тръбопровода RAG надолу по веригата, а не от вашия шлюз.
+When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers.
-Ако търсите отделен речник, който да напише тези повреди, можете да използвате WFGY ProblemMap, външен текстов ресурс за лиценз на MIT, който дефинира шестнадесет повтарящи се модели при отказ на RAG / LLM. На високо ниво: обхваща
+---
-- отклонение при извличане и нарушени контекстни граници
-- празни или остарели индекси и векторни хранилища
-- вграждане срещу семантично несъответствие
-- бързо сглобяване и проблеми с контекстния прозорец
-- логически колапс и изключително самоуверени отговори
-- дълга верига и неуспехи в координацията на агента
-- мултиагентна памет и дрейф на ролите
-- проблеми с внедряването и подреждането на избраното зареждане
+## Optional RAG / LLM failure taxonomy (16 problems)
-Идеята е проста:
+Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong.
-1. Когато проучвате лош отговор, заснемете:
- - потребителска задача и заявка
- - комбо маршрут или доставчик в OmniRoute
- - всеки RAG контекст, използван надолу по веригата (извлечени документи, извиквания на инструменти и т.н.)
-2. Съставете инцидента с едно или две номера на картата на проблемите на WFGY („No.1“ … „No.16“).
-3. Съхранявайте номера във вашето собствено табло, runbook или инструмент за проследяване на инциденти до регистриране на файлове в OmniRoute.
-4. Съществувате WFGY страница, за да решите дали трябва да промените своя RAG стек, ретривър или стратегия за маршрутизиране.
+In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself.
-Пълният текст и конкретните рецепти се намират тук (лиценз на MIT, само текстът):
+If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers:
+
+- retrieval drift and broken context boundaries
+- empty or stale indexes and vector stores
+- embedding versus semantic mismatch
+- prompt assembly and context window issues
+- logic collapse and overconfident answers
+- long chain and agent coordination failures
+- multi agent memory and role drift
+- deployment and bootstrap ordering problems
+
+The idea is simple:
+
+1. When you investigate a bad response, capture:
+ - user task and request
+ - route or provider combo in OmniRoute
+ - any RAG context used downstream (retrieved documents, tool calls, etc)
+2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`).
+3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs.
+4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy.
+
+Full text and concrete recipes live here (MIT license, text only):
[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md)
-Можете да пренебрегнете този раздел, ако не изпълните RAG или конвейери на агенти зад OmniRoute.---## Still Stuck?
+You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute.
--**Проблеми с GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Архитектура**: Вижте [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) за вътрешни подробности -**API Reference**: Вижте [`docs/API_REFERENCE.md`](API_REFERENCE.md) за всички крайни точки -**Таблица за управление на здравето**: Проверете**Таблица за управление → Здраве**за състоянието на системата в реално време -**Преводач**: Използвайте**Табло за управление → Преводач**за отстраняване на грешки във форматирането
+---
+
+## Still Stuck?
+
+- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details
+- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints
+- **Health Dashboard**: Check **Dashboard → Health** for real-time system status
+- **Translator**: Use **Dashboard → Translator** to debug format issues
diff --git a/docs/i18n/bg/llm.txt b/docs/i18n/bg/llm.txt
new file mode 100644
index 0000000000..fe9a151253
--- /dev/null
+++ b/docs/i18n/bg/llm.txt
@@ -0,0 +1,476 @@
+# OmniRoute (Български)
+
+🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt)
+
+---
+
+
+> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app.
+
+## Преглед
+
+OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free.
+
+**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost.
+
+**Current version:** 3.5.5
+
+## Tech Stack
+
+- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`)
+- **Framework:** Next.js 16 (App Router) with TypeScript 5.9
+- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations)
+- **State management:** Zustand (client), SQLite (server persistence)
+- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons
+- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth
+- **Schemas:** Zod v4 for all API / MCP input validation
+- **Background jobs:** Custom token health check scheduler, 24h model auto-sync
+- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses
+- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine
+- **i18n:** next-intl with 30 languages
+- **Desktop:** Electron (cross-platform: Windows, macOS, Linux)
+- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`)
+
+## Project Structure
+
+```
+/
+├── src/ # Main application source
+│ ├── app/ # Next.js App Router pages and API routes
+│ │ ├── (dashboard)/ # Dashboard UI pages
+│ │ │ └── dashboard/
+│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents)
+│ │ │ ├── analytics/ # Usage analytics and charts
+│ │ │ ├── api-manager/ # API key management
+│ │ │ ├── audit/ # Audit logs
+│ │ │ ├── auto-combo/ # Auto-combo engine dashboard
+│ │ │ ├── cache/ # Cache dashboard (semantic cache stats)
+│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.)
+│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates)
+│ │ │ ├── costs/ # Cost tracking per provider/model
+│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs
+│ │ │ ├── health/ # System health (uptime, circuit breakers, latency)
+│ │ │ ├── limits/ # Rate limits dashboard
+│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed)
+│ │ │ ├── media/ # Image/video/music generation + transcription
+│ │ │ ├── memory/ # Memory system dashboard
+│ │ │ ├── onboarding/ # Onboarding wizard
+│ │ │ ├── playground/ # Model playground (Monaco editor, streaming)
+│ │ │ ├── providers/ # Provider management (OAuth + API key + free)
+│ │ │ ├── search-tools/ # Search tools configuration
+│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced)
+│ │ │ ├── skills/ # Skills system dashboard
+│ │ │ ├── translator/ # Format translator + debug tools
+│ │ │ └── usage/ # Usage history
+│ │ ├── api/ # REST API endpoints (51 route directories)
+│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings,
+│ │ │ │ # images, audio, videos, music, moderations, rerank, search,
+│ │ │ │ # responses, messages, registered-keys, quotas, accounts)
+│ │ │ ├── v1beta/ # Gemini-compatible API
+│ │ │ ├── a2a/ # A2A agent management API
+│ │ │ ├── acp/ # ACP agent management API
+│ │ │ ├── oauth/ # OAuth flows per provider
+│ │ │ ├── providers/ # Provider CRUD and batch testing
+│ │ │ ├── models/ # Dashboard model listing and aliases
+│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains)
+│ │ │ ├── memory/ # Memory system API
+│ │ │ ├── skills/ # Skills system API
+│ │ │ ├── evals/ # Eval runner API
+│ │ │ ├── mcp/ # MCP HTTP transport API
+│ │ │ ├── search/ # Search provider API
+│ │ │ ├── webhooks/ # Webhook management
+│ │ │ ├── tunnels/ # Cloudflare tunnel management
+│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.)
+│ │ ├── landing/ # Landing page
+│ │ ├── login/ # Login page
+│ │ ├── forgot-password/ # Password recovery
+│ │ ├── status/ # Status page
+│ │ └── docs/ # In-app documentation
+│ ├── domain/ # Domain types and policy engine
+│ │ ├── policyEngine.ts # Central policy engine
+│ │ ├── comboResolver.ts # Combo resolution logic
+│ │ ├── costRules.ts # Cost calculation rules
+│ │ ├── degradation.ts # Graceful degradation
+│ │ ├── fallbackPolicy.ts # Fallback behavior
+│ │ ├── lockoutPolicy.ts # Account lockout logic
+│ │ ├── modelAvailability.ts # Model availability checks
+│ │ ├── providerExpiration.ts # Provider credential expiration
+│ │ ├── quotaCache.ts # Quota caching layer
+│ │ ├── configAudit.ts # Configuration auditing
+│ │ └── responses.ts # Domain response types
+│ ├── i18n/ # Internationalization
+│ │ └── messages/ # 30 language JSON files
+│ ├── lib/ # Core libraries
+│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server
+│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting)
+│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup
+│ │ │ └── streaming.ts # SSE streaming for A2A
+│ │ ├── acp/ # Agent Communication Protocol registry and manager
+│ │ ├── compliance/ # Compliance policy engine
+│ │ ├── db/ # SQLite database layer (21 modules + migrations)
+│ │ │ ├── core.ts # Database initialization, connection, schema
+│ │ │ ├── providers.ts # Provider connection CRUD
+│ │ │ ├── models.ts # Model catalog management
+│ │ │ ├── combos.ts # Combo configuration
+│ │ │ ├── apiKeys.ts # API key management
+│ │ │ ├── settings.ts # Settings persistence
+│ │ │ ├── backup.ts # Database backup/restore
+│ │ │ ├── proxies.ts # Proxy registry
+│ │ │ ├── prompts.ts # Prompt templates
+│ │ │ ├── webhooks.ts # Webhook subscriptions
+│ │ │ ├── detailedLogs.ts # Detailed request logging
+│ │ │ ├── domainState.ts # Domain state persistence
+│ │ │ ├── registeredKeys.ts # Registered API keys with quotas
+│ │ │ ├── quotaSnapshots.ts # Quota snapshot history
+│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings
+│ │ │ ├── cliToolState.ts # CLI tool state tracking
+│ │ │ ├── encryption.ts # Data encryption
+│ │ │ ├── readCache.ts # Read-through cache layer
+│ │ │ ├── secrets.ts # Secrets management
+│ │ │ ├── stateReset.ts # State reset utilities
+│ │ │ ├── migrationRunner.ts # Schema migration runner
+│ │ │ └── migrations/ # 16 SQL migration files
+│ │ ├── evals/ # Eval runner and scheduler
+│ │ ├── memory/ # Persistent conversational memory
+│ │ │ ├── extraction.ts # Memory extraction from conversations
+│ │ │ ├── injection.ts # Memory injection into context
+│ │ │ ├── retrieval.ts # Memory retrieval/search
+│ │ │ ├── store.ts # Memory persistence layer
+│ │ │ └── summarization.ts # Memory summarization
+│ │ ├── oauth/ # OAuth providers, services, and utilities
+│ │ │ ├── constants/ # Default OAuth credentials (overridable via env)
+│ │ │ ├── providers/ # Provider-specific OAuth configs
+│ │ │ ├── services/ # Provider-specific token exchange logic
+│ │ │ └── utils/ # PKCE, callback server, token helpers
+│ │ ├── plugins/ # Plugin system
+│ │ ├── skills/ # Extensible skill framework
+│ │ │ ├── registry.ts # Skill registration
+│ │ │ ├── executor.ts # Skill execution engine
+│ │ │ ├── sandbox.ts # Skill sandbox environment
+│ │ │ ├── builtin/ # Built-in skills
+│ │ │ ├── interception.ts # Skill request interception
+│ │ │ └── injection.ts # Skill context injection
+│ │ ├── usage/ # Usage tracking system
+│ │ │ ├── callLogs.ts # Call log persistence
+│ │ │ ├── costCalculator.ts # Cost calculation engine
+│ │ │ └── usageHistory.ts # Usage history queries
+│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers
+│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management
+│ │ ├── pricingSync.ts # LiteLLM pricing data sync
+│ │ ├── semanticCache.ts # Semantic caching layer
+│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler
+│ │ ├── webhookDispatcher.ts # Webhook event dispatcher
+│ │ └── localDb.ts # Unified re-export layer for all DB modules
+│ ├── middleware/ # Request middleware
+│ │ └── promptInjectionGuard.ts # Prompt injection detection
+│ ├── mitm/ # MITM proxy capability
+│ │ ├── cert/ # Certificate management
+│ │ ├── dns/ # DNS handling
+│ │ ├── targets/ # Target routing
+│ │ └── manager.ts # MITM proxy manager
+│ ├── shared/ # Shared utilities, components, and constants
+│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.)
+│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes
+│ │ ├── contracts/ # Shared API contracts
+│ │ ├── hooks/ # React hooks
+│ │ ├── middleware/ # Shared middleware utilities
+│ │ ├── schemas/ # Shared Zod schemas
+│ │ ├── services/ # Shared services
+│ │ ├── types/ # Shared TypeScript types
+│ │ ├── validation/ # Zod schemas (settings, providers, routes)
+│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID)
+│ ├── sse/ # SSE proxy pipeline
+│ │ ├── services/ # Auth resolution, format translation, response handling
+│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency
+│ ├── store/ # Zustand client-side stores (theme, providers, etc.)
+│ └── types/ # TypeScript type definitions
+├── open-sse/ # Standalone SSE server (npm workspace)
+│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video,
+│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models)
+│ ├── executors/ # Provider-specific request executors (14 executors)
+│ │ ├── base.ts # Base executor with shared logic
+│ │ ├── default.ts # Default OpenAI-compatible executor
+│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum)
+│ │ ├── codex.ts # OpenAI Codex CLI
+│ │ ├── antigravity.ts # Antigravity IDE
+│ │ ├── github.ts # GitHub Copilot
+│ │ ├── gemini-cli.ts # Gemini CLI
+│ │ ├── kiro.ts # Kiro AI
+│ │ ├── qoder.ts # Qoder AI
+│ │ ├── vertex.ts # Vertex AI (Service Account JSON)
+│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI
+│ │ ├── opencode.ts # OpenCode Zen/Go
+│ │ ├── pollinations.ts # Pollinations AI
+│ │ └── puter.ts # Puter AI
+│ ├── handlers/ # Request handlers per API type (11 handlers)
+│ │ ├── chatCore.ts # Main chat completions handler
+│ │ ├── responsesHandler.ts # OpenAI Responses API handler
+│ │ ├── embeddings.ts # Embedding generation
+│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.)
+│ │ ├── videoGeneration.ts # Video generation
+│ │ ├── musicGeneration.ts # Music generation
+│ │ ├── audioSpeech.ts # Text-to-speech
+│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI)
+│ │ ├── moderations.ts # Content moderation
+│ │ ├── rerank.ts # Reranking API
+│ │ └── search.ts # Web search API
+│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP)
+│ │ ├── server.ts # MCP server core (tool registration, scope enforcement)
+│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools)
+│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a)
+│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes)
+│ │ ├── audit.ts # Tool call audit logging
+│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat
+│ │ └── httpTransport.ts # HTTP transport handler
+│ ├── services/ # 36+ service modules
+│ │ ├── combo.ts # Core routing engine
+│ │ ├── usage.ts # Usage tracking
+│ │ ├── tokenRefresh.ts # OAuth token refresh
+│ │ ├── rateLimitManager.ts # Rate limit management
+│ │ ├── accountFallback.ts # Multi-account fallback
+│ │ ├── sessionManager.ts # Session management
+│ │ ├── wildcardRouter.ts # Wildcard model routing
+│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration)
+│ │ ├── intentClassifier.ts # Request intent classification
+│ │ ├── taskAwareRouter.ts # Task-aware routing
+│ │ ├── thinkingBudget.ts # Thinking budget management
+│ │ ├── contextManager.ts # Context window management
+│ │ ├── modelDeprecation.ts # Model deprecation handling
+│ │ ├── modelFamilyFallback.ts # Intra-family model fallback
+│ │ ├── emergencyFallback.ts # Emergency fallback
+│ │ ├── workflowFSM.ts # Workflow state machine
+│ │ ├── backgroundTaskDetector.ts # Background task detection
+│ │ ├── ipFilter.ts # IP-based access control
+│ │ ├── signatureCache.ts # CLI signature caching
+│ │ ├── volumeDetector.ts # Request volume detection
+│ │ ├── contextHandoff.ts # Context relay handoff generation and injection
+│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay
+│ │ └── ... # Additional services (14 more modules)
+│ ├── transformer/ # Responses API transformer
+│ │ └── responsesTransformer.ts
+│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek)
+│ │ ├── request/ # Request translators per provider
+│ │ ├── response/ # Response translators per provider
+│ │ ├── helpers/ # Translation helpers
+│ │ └── image/ # Image format translation
+│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.)
+├── electron/ # Electron desktop app (cross-platform)
+│ ├── main.js # Electron main process
+│ ├── preload.js # Preload script (IPC bridge)
+│ └── assets/ # App icons and assets
+├── tests/ # Test suites
+│ ├── unit/ # 122 unit test files
+│ ├── integration/ # Integration tests
+│ ├── e2e/ # Playwright E2E tests
+│ ├── security/ # Security tests
+│ ├── translator/ # Translator-specific tests
+│ └── load/ # Load tests
+├── docs/ # Documentation
+│ ├── i18n/ # 30-language translated docs
+│ ├── ARCHITECTURE.md # Full architecture documentation
+│ ├── API_REFERENCE.md # API reference
+│ ├── USER_GUIDE.md # User guide
+│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview
+│ ├── CLI-TOOLS.md # CLI tools integration guide
+│ ├── A2A-SERVER.md # A2A agent protocol documentation
+│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring)
+│ ├── MCP-SERVER.md # MCP server (25 tools)
+│ ├── TROUBLESHOOTING.md # Troubleshooting guide
+│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide
+│ ├── openapi.yaml # OpenAPI specification
+│ └── screenshots/ # Dashboard screenshots
+├── bin/ # CLI entry points (omniroute, reset-password)
+├── scripts/ # Build and utility scripts
+└── .env.example # Environment variable template
+```
+
+## Key Features (v3.5.5)
+
+### Core Proxy
+- **60+ AI providers** with automatic format translation
+- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible)
+- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay
+- **4-tier fallback**: Subscription → API Key → Cheap → Free
+- **Context Relay strategy**: Session handoff summaries on account rotation for continuity
+- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown
+- **Semantic caching** with cache hit/miss headers
+- **Idempotency** with configurable dedup window
+- **Circuit breaker** per provider with configurable thresholds
+- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback
+- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers
+- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement
+- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization
+- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills
+- **Prompt Injection Guard**: Middleware-level prompt injection detection
+- **MITM Proxy**: Certificate management, DNS handling, and target routing
+- **Cloudflare Tunnels**: Managed tunnel creation for remote access
+- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches)
+
+### Сигурност
+- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs)
+- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()`
+- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()`
+- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection
+- **CLI Fingerprint Matching** — Per-provider request signature matching
+- **Prompt injection guard** — Request middleware detection
+- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`)
+- **PII sanitizer** — Sensitive data scrubbing in logs
+
+### Dashboard Pages (23 sections)
+- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons
+- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies
+- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics
+- **Analytics** — Token consumption, cost, heatmaps, distributions
+- **Health** — Uptime, memory, latency percentiles, circuit breakers
+- **Logs** — Request, Proxy, Audit, Console (tabbed)
+- **Audit** — Audit trail and compliance logging
+- **Costs** — Cost tracking per provider/model
+- **Limits** — Rate limit monitoring
+- **Cache** — Semantic cache statistics and management
+- **CLI Tools** — One-click configuration for 10+ AI CLI tools
+- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration
+- **Playground** — Test any model with Monaco editor, streaming responses
+- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files)
+- **Search Tools** — Search provider configuration and testing
+- **Memory** — Memory system management and visualization
+- **Skills** — Skills framework management and execution
+- **Translator** — Format debugging: playground, chat tester, test bench, live monitor
+- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced
+- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed)
+- **Onboarding** — Setup wizard for new users
+- **Usage** — Usage history and analytics
+- **API Manager** — API key management with scoped permissions
+
+### Protocol Support
+- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations`
+- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens`
+- **OpenAI Responses** — `/v1/responses`
+- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}`
+- **Ollama** — `/v1/api/chat`, `/api/tags`
+- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily)
+- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP)
+- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills)
+- **ACP** — Agent Communication Protocol registry and manager
+
+### MCP Server (25 Tools)
+| Category | Tools |
+|-----------|-------|
+| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` |
+| Memory (3) | `memory_search`, `memory_add`, `memory_clear` |
+| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` |
+
+**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience`
+
+### Provider Categories
+
+**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI
+
+**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline
+
+**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan
+
+**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs
+
+### Internationalization
+- 30 languages for UI (all dashboard pages)
+- 30 translated documentation sets in docs/i18n/
+- Language switcher in documentation
+
+## Key Architectural Decisions
+
+1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints.
+
+2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently.
+
+3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation.
+
+4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity.
+
+5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client.
+
+6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes.
+
+7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`).
+
+8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages.
+
+9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations.
+
+10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`.
+
+11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation.
+
+12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline.
+
+13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast.
+
+## Main Flows
+
+### Proxy Request Flow
+1. Client sends OpenAI-format request to `/v1/chat/completions`
+2. API key validation
+3. Model resolution: direct model or combo lookup
+4. For combos: iterate through models with selected strategy
+5. Auth resolution: get credentials for the target provider
+6. Format translation: OpenAI → provider native format
+7. CLI fingerprint matching (if enabled for provider)
+8. Upstream request with circuit breaker and rate limiting
+9. Response translation: provider → OpenAI format
+10. omniModel tag sanitization (strip internal tags)
+11. SSE streaming back to client
+12. Memory extraction (if memory system enabled)
+13. Usage logging and cost calculation
+
+### OAuth Flow
+1. Dashboard initiates `/api/oauth/[provider]/authorize`
+2. User completes OAuth login in browser
+3. Callback hits `/api/oauth/[provider]/exchange`
+4. Tokens stored as a provider connection in SQLite
+5. Background job refreshes tokens before expiry
+
+## Important Notes for LLMs
+
+1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only).
+
+2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`).
+
+3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services.
+
+4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`.
+
+5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module.
+
+6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`).
+
+7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes.
+
+8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB.
+
+9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown.
+
+10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130).
+
+11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux).
+
+12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint.
+
+13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API.
+
+14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API.
+
+15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time.
+
+16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`.
+
+17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22.
+
+18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false.
+
+## Links
+
+- Repository: https://github.com/diegosouzapw/OmniRoute
+- Website: https://omniroute.online
+- npm: https://www.npmjs.com/package/omniroute
+- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute
+- Documentation: See `/docs/` directory
diff --git a/docs/i18n/cs/README.md b/docs/i18n/cs/README.md
index 7da2e6d4a3..6271fb5271 100644
--- a/docs/i18n/cs/README.md
+++ b/docs/i18n/cs/README.md
@@ -4,11 +4,14 @@
---
+
### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback.
-_Váš univerzální API proxy – jeden koncový bod, 60+ poskytovatelů, nulové prostoje. Nyní s**MCP Server (25 nástrojů)**,**Protokol A2A**,**Paměť/Skills Systems**a**Electron Desktop App**._
+_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._
-**Dokončení chatu • Vložení • Generování obrázků • Video • Hudba • Zvuk • Změna pořadí •**Vyhledávání na webu**• Server MCP • Protokol A2A • 100% TypeScript**---
+**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript**
+
+---
@@ -39,9 +42,13 @@ _Váš univerzální API proxy – jeden koncový bod, 60+ poskytovatelů, nulov
[](https://omniroute.online)
[](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-[🌐 Web](https://omniroute.online) • [🚀 Rychlý start](#-rychlý start) • [💡 Funkce](#-klíčových-funkcí) • [📖 Dokumenty](#-dokumentace) • [💰 Cena](#-cena-na první pohled) • [🬒 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-🌐**Dostupné v:**🇺🇸 [anglicky](README.md) | 🇧🇷 [Português (Brazílie)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dánsk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Maďarština](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugalsko)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)---
+
+
+🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)
+
+---
## 🖼️ Main Dashboard
@@ -53,63 +60,65 @@ _Váš univerzální API proxy – jeden koncový bod, 60+ poskytovatelů, nulov
## 📸 Dashboard Preview
-
-Kliknutím zobrazíte snímky obrazovky řídicího panelu
+
+Click to see dashboard screenshots
-| Strana | Snímek obrazovky |
-| --------------------- | --------------------------------------------------- | ---------- |
-| **Poskytovatelé** |  |
-| **Komba** |  |
-| **Analytika** |  |
-| **Zdraví** |  |
-| **Překladatel** |  |
-| **Nastavení** |  |
-| **Nástroje CLI** |  |
-| **Protokoly použití** |  |
-| **Koncové body** |  | |
+| Page | Screenshot |
+| -------------- | ------------------------------------------------- |
+| **Providers** |  |
+| **Combos** |  |
+| **Analytics** |  |
+| **Health** |  |
+| **Translator** |  |
+| **Settings** |  |
+| **CLI Tools** |  |
+| **Usage Logs** |  |
+| **Endpoints** |  |
+
+
---
### 🤖 Free AI Provider for your favorite coding agents
-_Připojte jakýkoli nástroj IDE nebo CLI s umělou inteligencí prostřednictvím OmniRoute – bezplatné brány API pro neomezené kódování._
+_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._
-
+

OpenClaw
- ⭐ 205 000
+ ⭐ 205K
|

NanoBot
- ⭐ 20,9 000
+ ⭐ 20.9K
|

PicoClaw
- ⭐ 14,6 000
+ ⭐ 14.6K
|

ZeroClaw
- ⭐ 9,9 000
+ ⭐ 9.9K
|
- 
+ 
IronClaw
- ⭐ 2,1 000
+ ⭐ 2.1K
|
@@ -118,489 +127,562 @@ _Připojte jakýkoli nástroj IDE nebo CLI s umělou inteligencí prostřednictv

OpenCode
- ⭐ 106 000
+ ⭐ 106K

Codex CLI
- ⭐ 60,8 000
+ ⭐ 60.8K
|
- 
+ 
Claude Code
- ⭐ 67,3 000
+ ⭐ 67.3K
|

Gemini CLI
- ⭐ 94,7 000
+ ⭐ 94.7K
|
- 
- Kilový kód
+ 
+ Kilo Code
- ⭐ 15,5 000
+ ⭐ 15.5K
|
-📡 Všichni agenti se připojují přes http://localhost:20128/v1 nebo http://cloud.omniroute.online/v1 — jedna konfigurace, neomezené modely a kvóta---
+📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota
+
+---
## 🤔 Why OmniRoute?
-**Přestaňte plýtvat penězi a narážet na limity:**
+**Stop wasting money and hitting limits:**
-–
Kvóta předplatného vyprší nevyužita každý měsíc
-–
Omezení sazby vám brání uprostřed kódování
-–
Drahá rozhraní API (20–50 $ měsíčně na poskytovatele)
+-
Subscription quota expires unused every month
+-
Rate limits stop you mid-coding
+-
Expensive APIs ($20-50/month per provider)
+-
Manual switching between providers
--
Ruční přepínání mezi poskytovateli
+**OmniRoute solves this:**
-**OmniRoute to řeší:**
+- ✅ **Maximize subscriptions** - Track quota, use every bit before reset
+- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime
+- ✅ **Multi-account** - Round-robin between accounts per provider
+- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool
-- ✅**Maximalizujte odběry**- Sledujte kvótu, před resetováním použijte každý bit
-- ✅**Automatická záloha**- Předplatné → Klíč API → Levné → Zdarma, nulové prostoje
-- ✅**Více účtů**- Round-robin mezi účty na poskytovatele
-- ✅**Universal**- Funguje s Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, jakýmkoliv nástrojem CLI---
+---
## 📧 Support
-> 💬**Připojte se k naší komunitě!**[Skupina WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Získejte nápovědu, sdílejte tipy a buďte v obraze.
+> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated.
--**Web**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problémy**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Skupina komunity](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Přispívání**: Podívejte se na [CONTRIBUTING.md](CONTRIBUTING.md), otevřete PR nebo si vyberte „dobré první číslo“ -**Původní projekt**: [9router by decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug?
+- **Website**: [omniroute.online](https://omniroute.online)
+- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute)
+- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue`
+- **Original Project**: [9router by decolua](https://github.com/decolua/9router)
-Při otevírání problému spusťte příkaz system-info a připojte vygenerovaný soubor:```bash
+### 🐛 Reporting a Bug?
+
+When opening an issue, please run the system-info command and attach the generated file:
+
+```bash
npm run system-info
-
```
-Tím se vygeneruje soubor `system-info.txt` s vaší verzí Node.js, verzí OmniRoute, podrobnostmi OS, nainstalovanými nástroji CLI (qoder, gemini, claude, codex, antigravity, droid atd.), stavem Docker/PM2 a systémovými balíčky – vše, co potřebujeme k rychlé reprodukci vašeho problému. Připojte soubor přímo k vašemu problému na GitHubu.---
+This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue.
+
+---
## 🔄 How It Works
```
-
┌─────────────┐
-│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
-│ Tool │
+│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
+│ Tool │
└──────┬──────┘
-│ http://localhost:20128/v1
-↓
+ │ http://localhost:20128/v1
+ ↓
┌─────────────────────────────────────────┐
-│ OmniRoute (Smart Router) │
-│ • Format translation (OpenAI ↔ Claude) │
-│ • Quota tracking + Embeddings + Images │
-│ • Auto token refresh │
+│ OmniRoute (Smart Router) │
+│ • Format translation (OpenAI ↔ Claude) │
+│ • Quota tracking + Embeddings + Images │
+│ • Auto token refresh │
└──────┬──────────────────────────────────┘
-│
-├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
-│ ↓ quota exhausted
-├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
-│ ↓ budget limit
-├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
-│ ↓ budget limit
-└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
+ │
+ ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
+ │ ↓ quota exhausted
+ ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
+ │ ↓ budget limit
+ ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
+ │ ↓ budget limit
+ └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
Result: Never stop coding, minimal cost
-
-````
+```
---
## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases
->**Každý vývojář používající nástroje AI čelí těmto problémům denně.**OmniRoute byl vytvořen tak, aby je vyřešil všechny – od překročení nákladů po regionální bloky, od přerušených toků OAuth po operace protokolů a podniková pozorovatelnost.
+> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability.
-
-💸 1. „Platím za drahé předplatné, ale stále mě vyrušují limity“
+
+💸 1. "I pay for an expensive subscription but still get interrupted by limits"
-Vývojáři platí 20–200 $ měsíčně za Claude Pro, Codex Pro nebo GitHub Copilot. I při placení má kvóta strop – 5 hodin používání, týdenní limity nebo limity sazby za minutu. Uprostřed relace kódování poskytovatel přestane reagovat a vývojář ztrácí tok a produktivitu.
+Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity.
-**Jak to řeší OmniRoute:**
+**How OmniRoute solves it:**
--**Chytrý 4-úrovňový záložní zdroj**– Pokud dojde k vyčerpání kvóty předplatného, automaticky se přesměruje na klíč API → Levné → Zdarma s nulovým ručním zásahem
--**Sledování limitů poskytovatele**– Snímky kvót v mezipaměti se obnovují podle plánu na straně serveru (výchozí `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) s možností ručního obnovení v uživatelském rozhraní
-–**Podpora více účtů**– Více účtů na poskytovatele s automatickým opakováním – když jeden dojde, přepne se na další
--**Vlastní komba**– Přizpůsobitelné záložní řetězce s 9 strategiemi vyvažování (prioritní, vážená, na prvním místě, s cyklem, P2C, náhodná, nejméně používaná, nákladově optimalizovaná, striktně náhodná)
--**Codex Business Quotas**— Sledování kvót Business/Tým pracovního prostoru přímo na řídicím panelu
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention
+- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI
+- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next
+- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**)
+- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard
-
-🔌 2. „Potřebuji používat více poskytovatelů, ale každý má jiné API“
+
-OpenAI používá jeden formát, Claude (Anthropic) jiný a Gemini ještě jiný. Pokud chce vývojář testovat modely od různých poskytovatelů nebo mezi nimi couvnout, musí překonfigurovat sady SDK, změnit koncové body, vypořádat se s nekompatibilními formáty. Vlastní poskytovatelé (FriendLI, NIM) mají nestandardní koncové body modelu.
+
+🔌 2. "I need to use multiple providers but each has a different API"
-**Jak to řeší OmniRoute:**
+OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints.
--**Unified Endpoint**– Jediný `http://localhost:20128/v1` slouží jako proxy pro všech 60+ poskytovatelů
--**Formátový překlad**— Automatický a transparentní: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
--**Response Sanitization**– Odstraňuje nestandardní pole (`x_groq`, `usage_breakdown`, `service_tier`), která porušují OpenAI SDK v1.83+
--**Normalizace rolí**— Převádí `vývojář` → `systém` pro poskytovatele, kteří nejsou OpenAI; `systém` → `uživatel` pro GLM/ERNIE
-–**Think Tag Extraction**– Extrahuje bloky „“ z modelů jako DeepSeek R1 do standardizovaného „reasoning_content“
--**Strukturovaný výstup pro Gemini**— `json_schema` → automatický převod `responseMimeType`/`responseSchema`
--**`stream` má výchozí hodnotu `false`**— Vyhovuje specifikaci OpenAI a zabraňuje neočekávanému SSE v sadách Python/Rust/Go SDK
+**How OmniRoute solves it:**
-
-🌐 3. „Můj poskytovatel umělé inteligence blokuje můj region/země“
+- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers
+- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
+- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+
+- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE
+- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content`
+- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion
+- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs
-Poskytovatelé jako OpenAI/Codex blokují přístup z určitých geografických oblastí. Během připojení OAuth a rozhraní API se uživatelům zobrazují chyby jako „unsupported_country_region_territory“. To je frustrující zejména pro vývojáře z rozvojových zemí.
+
-**Jak to řeší OmniRoute:**
+
+🌐 3. "My AI provider blocks my region/country"
-–**3úrovňová konfigurace proxy**– konfigurovatelný proxy na 3 úrovních: globální (veškerý provoz), podle poskytovatele (pouze jeden poskytovatel) a podle připojení/klíče
--**Barevně kódované odznaky proxy**— Vizuální indikátory: 🢢 globální proxy, 🟡 proxy poskytovatele, 🔵 proxy připojení, vždy zobrazující IP
--**Výměna tokenů OAuth přes proxy**– tok OAuth prochází také přes proxy a řeší se `unsupported_country_region_territory`
--**Testy připojení přes proxy**— Testy připojení používají nakonfigurovaný proxy (už žádné přímé obcházení)
--**Podpora SOCKS5**— Plná podpora proxy SOCKS5 pro odchozí směrování
--**TLS Fingerprint Spoofing**– TLS otisk prstu podobný prohlížeči přes `wreq-js` k obejití detekce botů
--**🔏 CLI Fingerprint Matching**– Změní pořadí hlaviček a polí těla tak, aby odpovídaly nativním binárním podpisům CLI, čímž se výrazně sníží riziko označení účtu. IP proxy serveru je zachována – získáte současně maskování IP maskování**a**utajení
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries.
-
-🆓 4. „Chci používat AI pro kódování, ale nemám peníze“
+**How OmniRoute solves it:**
-Ne každý může platit 20–200 $ měsíčně za předplatné AI. Studenti, vývojáři z rozvíjejících se zemí, fandové a nezávislí pracovníci potřebují přístup ke kvalitním modelům za nulové náklady.
+- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key
+- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP
+- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory`
+- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass)
+- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection
+- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously
-**Jak to řeší OmniRoute:**
+
--**Vestavění poskytovatelé bezplatných úrovní**— Nativní podpora pro 100% bezplatné poskytovatele: Qoder (5 neomezených modelů prostřednictvím OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 neomezené modely: qwen3-qwender-lash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID zdarma), Gemini CLI (180 000 tokenů/měsíc zdarma)
--**Ollama Cloud**– modely Ollama hostované v cloudu na `api.ollama.com` s bezplatnou úrovní „Light use“; použijte předponu `ollamacloud/`
-–**komba pouze zdarma**– řetězec `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 $/měsíc s nulovými prostoji
--**Volný přístup NVIDIA NIM**— ~40 RPM pro vývojáře - navždy bezplatný přístup k více než 70 modelům na build.nvidia.com (přechod z kreditů na limity čisté sazby)
--**Cost Optimized Strategy**– Strategie směrování, která automaticky vybírá nejlevnějšího dostupného poskytovatele
+
+🆓 4. "I want to use AI for coding but I have no money"
-
-🔒 5. „Potřebuji chránit svou bránu AI před neoprávněným přístupem“
+Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost.
-Při vystavení brány AI do sítě (LAN, VPS, Docker) může kdokoli s adresou spotřebovat tokeny/kvótu vývojáře. Bez ochrany jsou rozhraní API zranitelná vůči zneužití, rychlému vložení a zneužití.
+**How OmniRoute solves it:**
-**Jak to řeší OmniRoute:**
+- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free)
+- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix
+- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime
+- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits)
+- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider
--**Správa klíčů API**– Generování, rotace a rozsah podle poskytovatele pomocí vyhrazené stránky `/dashboard/api-manager`
--**Oprávnění na úrovni modelu**– Omezte klíče API na konkrétní modely (`openai/*`, vzory zástupných znaků) pomocí přepínače Povolit vše/Omezit
--**API Endpoint Protection**– Vyžadovat klíč pro `/v1/models` a blokovat konkrétní poskytovatele ze seznamu
--**Auth Guard + ochrana CSRF**– Všechny cesty řídicího panelu chráněny middlewarem „withAuth“ + tokeny CSRF
--**Rate Limiter**— omezení rychlosti na IP pomocí konfigurovatelných oken
--**IP Filtering**— Seznam povolených/blokovaných pro řízení přístupu
--**Prompt Injection Guard**– Dezinfekce proti škodlivým vzorům výzev
--**Šifrování AES-256-GCM**— Přihlašovací údaje jsou v klidu zašifrovány
+
-
-🛑 6. "Můj poskytovatel selhal a ztratil jsem tok kódování"
+
+🔒 5. "I need to protect my AI gateway from unauthorized access"
-Poskytovatelé umělé inteligence se mohou stát nestabilními, vracet chyby 5xx nebo narazit na dočasné limity sazeb. Pokud vývojář závisí na jediném poskytovateli, je přerušen. Bez jističů mohou opakované pokusy způsobit selhání aplikace.
+When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse.
-**Jak to řeší OmniRoute:**
+**How OmniRoute solves it:**
--**Jistič pro každý model**— Automatické otevírání/zavírání s konfigurovatelnými prahovými hodnotami a ochlazením (zavřeno/otevřeno/polootevřeno), s rozsahem pro každý model, aby se zabránilo kaskádovým blokům
--**Exponential Backoff**— Progresivní zpoždění opakování
--**Anti-Thundering Herd**— Mutex + semaforová ochrana proti souběžným opakovaným bouřím
--**Combo Fallback Chains**— Pokud primární poskytovatel selže, automaticky projde řetězcem bez zásahu
--**Combo Circuit Breaker**– Automaticky deaktivuje selhávající poskytovatele v rámci kombinovaného řetězce
-–**Health Dashboard**– Monitorování provozuschopnosti, stavy jističů, uzamčení, statistiky mezipaměti, latence p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page
+- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle
+- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing
+- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens
+- **Rate Limiter** — Per-IP rate limiting with configurable windows
+- **IP Filtering** — Allowlist/blocklist for access control
+- **Prompt Injection Guard** — Sanitization against malicious prompt patterns
+- **AES-256-GCM Encryption** — Credentials encrypted at rest
-
-🔧 7. „Konfigurace každého nástroje umělé inteligence je únavná a opakující se“
+
-Vývojáři používají Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Každý nástroj potřebuje jinou konfiguraci (API endpoint, klíč, model). Překonfigurování při změně poskytovatele nebo modelu je ztráta času.
+
+🛑 6. "My provider went down and I lost my coding flow"
-**Jak to řeší OmniRoute:**
+AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application.
--**CLI Tools Dashboard**– Vyhrazená stránka s nastavením jedním kliknutím pro Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
-–**GitHub Copilot Config Generator**– Generuje `chatLanguageModels.json` pro kód VS s hromadným výběrem modelu
--**Průvodce přihlášením**– Průvodce nastavením ve 4 krocích pro začínající uživatele
-–**Jeden koncový bod, všechny modely**– Jednou nakonfigurujte `http://localhost:20128/v1`, získáte přístup k více než 60 poskytovatelům
+**How OmniRoute solves it:**
-
-🔑 8. „Správa tokenů OAuth od více poskytovatelů je peklo“
+- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks
+- **Exponential Backoff** — Progressive retry delays
+- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms
+- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention
+- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain
+- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency
-Claude Code, Codex, Gemini CLI, Copilot – všechny používají OAuth 2.0 s končícími tokeny. Vývojáři se musí neustále znovu autentizovat, řešit `client_secret is missing`, `redirect_uri_mismatch` a selhání na vzdálených serverech. Zvláště problematické je OAuth na LAN/VPS.
+
-**Jak to řeší OmniRoute:**
+
+🔧 7. "Configuring each AI tool is tedious and repetitive"
--**Automatické obnovení tokenu**– Tokeny OAuth se před vypršením platnosti obnovují na pozadí
--**Vestavěný OAuth 2.0 (PKCE)**— Automatický tok pro Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
-–**Multi-Account OAuth**– Více účtů na poskytovatele prostřednictvím extrakce tokenů JWT/ID
--**OAuth LAN/Remote Fix**— Detekce privátní IP adresy pro `redirect_uri` + ruční režim URL pro vzdálené servery
--**OAuth Behind Nginx**– Používá `window.location.origin` pro zpětnou kompatibilitu proxy
-–**Průvodce vzdáleným OAuth**– Podrobný průvodce pro přihlašovací údaje Google Cloud na VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time.
-
-📊 9. "Nevím, kolik a kde utrácím"
+**How OmniRoute solves it:**
-Vývojáři využívají více placených poskytovatelů, ale nemají jednotný pohled na výdaje. Každý poskytovatel má svůj vlastní panel fakturace, ale neexistuje žádné konsolidované zobrazení. Neočekávané náklady se mohou nahromadit.
+- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
+- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection
+- **Onboarding Wizard** — Guided 4-step setup for first-time users
+- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers
-**Jak to řeší OmniRoute:**
+
--**Cost Analytics Dashboard**– Sledování nákladů na token a správa rozpočtu na poskytovatele
--**Rozpočtové limity na úroveň**– Strop útraty na úroveň, který spouští automatickou rezervu
--**Konfigurace cen za model**– Konfigurovatelné ceny za model
--**Statistika využití na klíč API**— Počet požadavků a naposledy použité časové razítko na klíč
-–**Panel Analytics**– Statistické karty, graf využití modelu, tabulka poskytovatelů s mírou úspěšnosti a latencí
+
+🔑 8. "Managing OAuth tokens from multiple providers is hell"
-
-🐛 10. „Nemohu diagnostikovat chyby a problémy ve voláních AI“
+Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic.
-Když se volání nezdaří, vývojář neví, zda to byl limit sazby, vypršela platnost tokenu, nesprávný formát nebo chyba poskytovatele. Fragmentované protokoly napříč různými terminály. Bez pozorovatelnosti je ladění metodou pokus-omyl.
+**How OmniRoute solves it:**
-**Jak to řeší OmniRoute:**
+- **Auto Token Refresh** — OAuth tokens refresh in background before expiration
+- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
+- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction
+- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers
+- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility
+- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker
--**Sjednocený panel protokolů**– 4 karty: Protokoly požadavků, Protokoly proxy, Protokoly auditu, Konzole
--**Console Log Viewer**— Prohlížeč ve stylu terminálu v reálném čase s barevně odlišenými úrovněmi, automatickým posouváním, vyhledáváním, filtrem
--**Protokoly SQLite Proxy**— Trvalé protokoly, které vydrží restartování serveru
--**Translator Playground**— 4 režimy ladění: Playground (překlad formátu), Tester chatu (zpáteční), Test Bench (dávka), Live Monitor (v reálném čase)
-–**Požadavek na telemetrii**– latence p50/p95/p99 + sledování X-Request-Id
--**Protokolování založené na souborech s rotací**– Protokoly aplikací se střídají podle velikosti, dnů uchování a počtu archivů; Artefakty protokolu hovorů rotují podle dnů uchování a počtu souborů
--**System Info Report**— `npm run system-info` vygeneruje `system-info.txt` s vaším úplným prostředím (verze uzlu, verze OmniRoute, OS, nástroje CLI, stav Docker/PM2). Připojte jej při hlášení problémů pro okamžité třídění.
+
-
-🏗️ 11. „Nasazení a údržba brány je složitá“
+
+📊 9. "I don't know how much I'm spending or where"
-Instalace, konfigurace a údržba AI proxy v různých prostředích (místní, VPS, Docker, cloud) je náročná na práci. Problémy jako pevně zakódované cesty, „EACCES“ v adresářích, konflikty portů a sestavení napříč platformami zvyšují tření.
+Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up.
-**Jak to řeší OmniRoute:**
+**How OmniRoute solves it:**
--**Globální instalace npm**– `npm install -g omniroute && omniroute` – hotovo
--**Docker Multi-Platform**– nativní AMD64 + ARM64 (Apple Silicon, AWS Graviton, Raspberry Pi)
--**Docker Compose Profiles**– `base` (žádné nástroje CLI) a `cli` (s Claude Code, Codex, OpenClaw)
--**Electron Desktop App**– nativní aplikace pro Windows/macOS/Linux se systémovou lištou, automatickým spuštěním, offline režimem
--**Split-Port Mode**– API a Dashboard na samostatných portech pro pokročilé scénáře (reverzní proxy, kontejnerová síť)
--**Cloud Sync**— Konfigurace synchronizace mezi zařízeními pomocí Cloudflare Workers
--**DB Backups**— Automatické zálohování, obnova, export a import všech nastavení s `DISABLE_SQLITE_AUTO_BACKUP` pro externě spravované zálohy
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider
+- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback
+- **Per-Model Pricing Configuration** — Configurable prices per model
+- **Usage Statistics Per API Key** — Request count and last-used timestamp per key
+- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency
-
-🌍 12. "Rozhraní je pouze v angličtině a můj tým nemluví anglicky"
+
-Týmy v neanglicky mluvících zemích, zejména v Latinské Americe, Asii a Evropě, se potýkají s rozhraním pouze v angličtině. Jazykové bariéry snižují přijetí a zvyšují chyby konfigurace.
+
+🐛 10. "I can't diagnose errors and problems in AI calls"
-**Jak to řeší OmniRoute:**
+When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error.
--**Dashboard i18n — 30 jazyků**— Všech 500+ kláves přeloženo včetně arabštiny, bulharštiny, dánštiny, němčiny, španělštiny, finštiny, francouzštiny, hebrejštiny, hindštiny, maďarštiny, indonéštiny, italštiny, japonštiny, korejštiny, malajštiny, holandštiny, norštiny, polštiny, portugalštiny (PT/BR), rumunštiny, ruštiny, slovenštiny, švédštiny, thajštiny, filipínštiny, vietnamštiny, angličtiny
--**Podpora RTL**— Podpora zprava doleva pro arabštinu a hebrejštinu
--**Vícejazyčné README**— 30 kompletních překladů dokumentace
--**Language Selector**— Ikona zeměkoule v záhlaví pro přepínání v reálném čase
+**How OmniRoute solves it:**
-
-🔄 13. „Potřebuji víc než jen chat – potřebuji vložení, obrázky, zvuk“
+- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console
+- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter
+- **SQLite Proxy Logs** — Persistent logs that survive server restarts
+- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time)
+- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing
+- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count
+- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage.
-AI není jen dokončení chatu. Vývojáři potřebují generovat obrázky, přepisovat zvuk, vytvářet vložení pro RAG, měnit hodnocení dokumentů a moderovat obsah. Každé API má jiný koncový bod a formát.
+
-**Jak to řeší OmniRoute:**
+
+🏗️ 11. "Deploying and maintaining the gateway is complex"
--**Vložení**— `/v1/embeddings` se 6 poskytovateli a 9+ modely
--**Generace obrázků**— `/v1/images/generations` s 10 poskytovateli a 20+ modely (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
--**Text-to-Video**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) a SD WebUI
--**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
--**Audio Transscription**— `/v1/audio/transscriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
--**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + stávající poskytovatelé
--**Moderations**— `/v1/moderations` — Kontroly bezpečnosti obsahu
--**Přehodnocení**— `/v1/rerank` — Změna pořadí podle relevance dokumentu
--**Responses API**— Plná podpora `/v1/responses` pro Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction.
-
-🧪 14. „Nemám způsob, jak testovat a porovnávat kvalitu napříč modely“
+**How OmniRoute solves it:**
-Vývojáři chtějí vědět, který model je pro jejich případ použití nejlepší – kód, překlad, uvažování – ale ruční porovnávání je pomalé. Neexistují žádné integrované nástroje eval.
+- **npm global install** — `npm install -g omniroute && omniroute` — done
+- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi)
+- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw)
+- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode
+- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking)
+- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers
+- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups
-**Jak to řeší OmniRoute:**
+
--**Hodnocení LLM**– Testování zlaté sady s 10 předem nahranými případy zahrnujícími pozdravy, matematiku, geografii, generování kódu, soulad s JSON, překlad, markdown, bezpečnostní odmítnutí
--**4 strategie shody**– `přesné`, `obsahuje`, `regulární výraz`, `vlastní` (funkce JS)
--**Testovací stolice pro překladatelské hřiště**– Dávkové testování s více vstupy a očekávanými výstupy, porovnání mezi poskytovateli
--**Chat Tester**– Kompletní zpáteční cesta s vykreslováním vizuální odezvy
--**Live Monitor**— Tok všech požadavků procházejících přes proxy v reálném čase
+
+🌍 12. "The interface is English-only and my team doesn't speak English"
-
-📈 15. „Potřebuji škálovat bez ztráty výkonu“
+Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors.
-Jak roste objem požadavků, bez ukládání stejných otázek do mezipaměti vznikají duplicitní náklady. Bez idempotence duplikát požaduje zpracování odpadu. Musí být dodrženy limity sazeb na poskytovatele.
+**How OmniRoute solves it:**
-**Jak to řeší OmniRoute:**
+- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English
+- **RTL Support** — Right-to-left support for Arabic and Hebrew
+- **Multi-Language READMEs** — 30 complete documentation translations
+- **Language Selector** — Globe icon in header for real-time switching
--**Sémantická mezipaměť**– Dvouvrstvá mezipaměť (podpis + sémantická) snižuje náklady a latenci
--**Idempotency požadavku**— 5s deduplikační okno pro identické požadavky
-–**Detekce limitu rychlosti**– RPM na poskytovatele, minimální mezera a maximální souběžné sledování
--**Upravitelné limity rychlosti**– Konfigurovatelné výchozí hodnoty v Nastavení → Odolnost s perzistencí
--**API Key Validation Cache**– 3vrstvá mezipaměť pro produkční výkon
-–**Health Dashboard s telemetrií**– latence p50/p95/p99, statistiky mezipaměti, doba provozu
+
-
-🤖 16. „Chci globálně ovládat chování modelu“
+
+🔄 13. "I need more than chat — I need embeddings, images, audio"
-Vývojáři, kteří chtějí všechny odpovědi v konkrétním jazyce, s konkrétním tónem nebo chtějí omezit tokeny uvažování. Konfigurace tohoto v každém nástroji/požadavku je nepraktická.
+AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format.
-**Jak to řeší OmniRoute:**
+**How OmniRoute solves it:**
--**System Prompt Injection**– Globální výzva aplikovaná na všechny požadavky
--**Thinking Budget Validation**— Řízení alokace tokenů na základě požadavku (průchozí, automatické, vlastní, adaptivní)
--**9 směrovacích strategií**— Globální strategie, které určují způsob distribuce požadavků
--**Wildcard Router**– vzory `poskytovatel/*` se dynamicky směrují k libovolnému poskytovateli
--**Povolit/zakázat přepínání komba**— Přepínejte komba přímo z řídicího panelu
--**Přepnutí poskytovatele**— Povolí/zakáže všechna připojení pro poskytovatele jedním kliknutím
-–**Blokovaní poskytovatelé**– vyloučení konkrétních poskytovatelů ze seznamu `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models
+- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
+- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI
+- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
+- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
+- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers
+- **Moderations** — `/v1/moderations` — Content safety checks
+- **Reranking** — `/v1/rerank` — Document relevance reranking
+- **Responses API** — Full `/v1/responses` support for Codex
-
-🧰 17. „Potřebuji nástroje MCP jako prvotřídní možnosti produktu“
+
-Mnoho bran AI odhaluje MCP pouze jako skrytý detail implementace. Týmy potřebují viditelnou a spravovatelnou provozní vrstvu.
+
+🧪 14. "I have no way to test and compare quality across models"
-**Jak to řeší OmniRoute:**
+Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist.
-- MCP se objeví na navigačním panelu a na kartě protokolu koncového bodu
-- Vyhrazená stránka správy MCP s procesem, nástroji, rozsahy a auditem
-- Vestavěný rychlý start pro `omniroute --mcp` a přihlášení klienta
+**How OmniRoute solves it:**
-
-🧠 18. „Potřebuji orchestraci A2A s cestami synchronizace + streamování“
+- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal
+- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function)
+- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison
+- **Chat Tester** — Full round-trip with visual response rendering
+- **Live Monitor** — Real-time stream of all requests flowing through the proxy
-Pracovní postupy agentů vyžadují jak přímé odpovědi, tak dlouhotrvající streamované spouštění s řízením životního cyklu.
+
-**Jak to řeší OmniRoute:**
+
+📈 15. "I need to scale without losing performance"
-– Koncový bod A2A JSON-RPC (`POST /a2a`) s `zprávou/odeslat` a `zprávou/streamem`
-- SSE streamování s šířením koncového stavu
-- Rozhraní API životního cyklu úloh pro `tasks/get` a `tasks/cancel`
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected.
-
-🛰️ 19. „Potřebuji skutečný stav procesu MCP, nikoli odhadovaný stav“
+**How OmniRoute solves it:**
-Operační týmy potřebují vědět, zda je MCP skutečně naživu, nejen zda je API dosažitelné.
+- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency
+- **Request Idempotency** — 5s deduplication window for identical requests
+- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking
+- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence
+- **API Key Validation Cache** — 3-tier cache for production performance
+- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime
-**Jak to řeší OmniRoute:**
+
-- Soubor srdečního tepu za běhu s PID, časovými razítky, transportem, počtem nástrojů a režimem rozsahu
-- Stavové API MCP kombinující srdeční tep + nedávnou aktivitu
-- Stavové karty uživatelského rozhraní pro aktuálnost procesu / provozuschopnosti / srdečního tepu
+
+🤖 16. "I want to control model behavior globally"
-
-📋 20. „Potřebuji provádění auditovatelného nástroje MCP“
+Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical.
-Když nástroje mutují konfiguraci nebo spouštějí operace operací, týmy potřebují forenzní sledovatelnost.
+**How OmniRoute solves it:**
-**Jak to řeší OmniRoute:**
+- **System Prompt Injection** — Global prompt applied to all requests
+- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive)
+- **9 Routing Strategies** — Global strategies that determine how requests are distributed
+- **Wildcard Router** — `provider/*` patterns route dynamically to any provider
+- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard
+- **Provider Toggle** — Enable/disable all connections for a provider with one click
+- **Blocked Providers** — Exclude specific providers from `/v1/models` listing
-- Protokolování auditu podporované SQLite pro volání nástrojů MCP
-- Filtry podle nástroje, úspěchu/neúspěchu, klíče API a stránkování
-- Tabulka auditu řídicího panelu + statistiky koncových bodů pro automatizaci
+
-
-🔐 21. „Potřebuji omezená oprávnění MCP na integraci“
+
+🧰 17. "I need MCP tools as first-class product capabilities"
-Různí klienti by měli mít nejméně privilegovaný přístup ke kategoriím nástrojů.
+Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer.
-**Jak to řeší OmniRoute:**
+**How OmniRoute solves it:**
-- 10 granulárních rozsahů MCP pro řízený přístup k nástrojům
-- Vynucení rozsahu a viditelnost v uživatelském rozhraní správy MCP
-- Bezpečná výchozí poloha pro provozní nástroje
+- MCP appears in the dashboard navigation and endpoint protocol tab
+- Dedicated MCP management page with process, tools, scopes, and audit
+- Built-in quick-start for `omniroute --mcp` and client onboarding
-
-⚙️ 22. „Potřebuji provozní kontroly bez přerozdělování“
+
-Týmy potřebují rychlé změny běhového prostředí během incidentů nebo nákladových událostí.
+
+🧠 18. "I need A2A orchestration with sync + stream task paths"
-**Jak to řeší OmniRoute:**
+Agent workflows need both direct replies and long-running streamed execution with lifecycle control.
-- Aktivace kombinace přepínačů přímo z řídicího panelu MCP
-- Použijte profily odolnosti z předdefinovaných balíčků zásad
-- Resetujte stav jističe ze stejného ovládacího panelu
+**How OmniRoute solves it:**
-
-🔄 23. „Potřebuji živou viditelnost a zrušení životního cyklu úkolu A2A“
+- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream`
+- SSE streaming with terminal state propagation
+- Task lifecycle APIs for `tasks/get` and `tasks/cancel`
-Bez viditelnosti životního cyklu je obtížné třídit incidenty úkolů.
+
-**Jak to řeší OmniRoute:**
+
+🛰️ 19. "I need real MCP process health, not guessed status"
-- Seznam úkolů / filtrování podle stavu / dovedností se stránkováním
-- Podrobnější informace o metadatech úkolů, událostech a artefaktech
-- Koncový bod zrušení úlohy a akce uživatelského rozhraní s potvrzením
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable.
-
-🌊 24. „Potřebuji aktivní metriky streamu pro zatížení A2A“
+**How OmniRoute solves it:**
-Streamovací pracovní postupy vyžadují provozní přehled o souběžných a živých připojeních.
+- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode
+- MCP status API combining heartbeat + recent activity
+- UI status cards for process/uptime/heartbeat freshness
-**Jak to řeší OmniRoute:**
+
-- Aktivní čítače toku integrované do stavu A2A
-- Časové razítko posledního úkolu a počty za stav
-- Karty palubní desky A2A pro monitorování operací v reálném čase
+
+📋 20. "I need auditable MCP tool execution"
-
-🪪 25. „Potřebuji pro klienty zjišťování standardních agentů“
+When tools mutate config or trigger ops actions, teams need forensic traceability.
-Externí klienti a orchestrátoři potřebují strojově čitelná metadata pro integraci.
+**How OmniRoute solves it:**
-**Jak to řeší OmniRoute:**
+- SQLite-backed audit logging for MCP tool calls
+- Filters by tool, success/failure, API key, and pagination
+- Dashboard audit table + stats endpoints for automation
-- Karta agenta vystavena na adrese `/.well-known/agent.json`
-- Schopnosti a dovednosti zobrazené v uživatelském rozhraní pro správu
-- A2A status API obsahuje metadata zjišťování pro automatizaci
+
-
-🧭 26. „Potřebuji zjistitelnost protokolu v uživatelském rozhraní produktu“
+
+🔐 21. "I need scoped MCP permissions per integration"
-Pokud uživatelé nemohou objevit protokolové povrchy, kvalita přijetí a podpory klesá.
+Different clients should have least-privilege access to tool categories.
-**Jak to řeší OmniRoute:**
+**How OmniRoute solves it:**
-– Konsolidovaná stránka**Koncové body**s kartami pro koncové body proxy, MCP, A2A a API
-- Přepínání stavu inline služby (Online/Offline) pro MCP a A2A
-- Odkazy z přehledu na vyhrazené karty správy
+- 10 granular MCP scopes for controlled tool access
+- Scope enforcement and visibility in MCP management UI
+- Safe default posture for operational tooling
-
-🧪 27. „Potřebuji komplexní ověření protokolu se skutečnými klienty“
+
-Falešné testy nestačí k ověření kompatibility protokolu před vydáním.
+
+⚙️ 22. "I need operational controls without redeploying"
-**Jak to řeší OmniRoute:**
+Teams need quick runtime changes during incidents or cost events.
-- Sada E2E, která spouští aplikaci a používá skutečný přenos klienta MCP SDK
-- Klient A2A testuje toky zjišťování, odesílání, streamování, získávání a rušení
-- Křížová kontrola tvrzení proti auditu MCP a API úloh A2A
+**How OmniRoute solves it:**
-
-📡 28. „Potřebuji jednotnou pozorovatelnost napříč všemi rozhraními“
+- Switch combo activation directly from MCP dashboard
+- Apply resilience profiles from pre-defined policy packs
+- Reset circuit breaker state from the same operations panel
-Rozdělení pozorovatelnosti protokolem vytváří slepá místa a delší MTTR.
+
-**Jak to řeší OmniRoute:**
+
+🔄 23. "I need live A2A task lifecycle visibility and cancellation"
-- Sjednocené dashboardy/logy/analýzy v jednom produktu
-- Zdraví + audit + telemetrie požadavků napříč vrstvami OpenAI, MCP a A2A
-- Provozní API pro stav a automatizaci
+Without lifecycle visibility, task incidents become hard to triage.
-
-💼 29. "Potřebuji jeden runtime pro proxy + nástroje + orchestraci agenta"
+**How OmniRoute solves it:**
-Provozování mnoha samostatných služeb zvyšuje provozní náklady a způsoby selhání.
+- Task listing/filtering by state/skill with pagination
+- Drill-down on task metadata, events, and artifacts
+- Task cancellation endpoint and UI action with confirmation
-**Jak to řeší OmniRoute:**
+
-- Proxy, MCP server a A2A server v jednom zásobníku kompatibilní s OpenAI
-- Sdílená autentizace, odolnost, úložiště dat a pozorovatelnost
-- Konzistentní model politiky na všech interakčních plochách
+
+🌊 24. "I need active stream metrics for A2A load"
-
-🚀 30. „Potřebuji odeslat agentské pracovní postupy bez rozšiřování kódu lepidla“
+Streaming workflows require operational insight into concurrency and live connections.
-Týmy ztrácejí rychlost při spojování více ad-hoc služeb a skriptů.
+**How OmniRoute solves it:**
-**Jak to řeší OmniRoute:**
+- Active stream counters integrated into A2A status
+- Last task timestamp and per-state counts
+- A2A dashboard cards for real-time ops monitoring
-- Jednotná strategie koncových bodů pro klienty a agenty
-- Vestavěná uživatelská rozhraní pro správu protokolů a cesty ověřování kouře
-- Základy připravené na výrobu (zabezpečení, protokolování, odolnost, zálohování)
+
+
+
+🪪 25. "I need standard agent discovery for clients"
+
+External clients and orchestrators need machine-readable metadata for onboarding.
+
+**How OmniRoute solves it:**
+
+- Agent Card exposed at `/.well-known/agent.json`
+- Capabilities and skills shown in management UI
+- A2A status API includes discovery metadata for automation
+
+
+
+
+🧭 26. "I need protocol discoverability in the product UX"
+
+If users cannot discover protocol surfaces, adoption and support quality drop.
+
+**How OmniRoute solves it:**
+
+- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints
+- Inline service status toggles (Online/Offline) for MCP and A2A
+- Links from overview to dedicated management tabs
+
+
+
+
+🧪 27. "I need end-to-end protocol validation with real clients"
+
+Mock tests are not enough to validate protocol compatibility before release.
+
+**How OmniRoute solves it:**
+
+- E2E suite that boots app and uses real MCP SDK client transport
+- A2A client tests for discovery, send, stream, get, and cancel flows
+- Cross-check assertions against MCP audit and A2A tasks APIs
+
+
+
+
+📡 28. "I need unified observability across all interfaces"
+
+Splitting observability by protocol creates blind spots and longer MTTR.
+
+**How OmniRoute solves it:**
+
+- Unified dashboards/logs/analytics in one product
+- Health + audit + request telemetry across OpenAI, MCP, and A2A layers
+- Operational APIs for status and automation
+
+
+
+
+💼 29. "I need one runtime for proxy + tools + agent orchestration"
+
+Running many separate services increases operational cost and failure modes.
+
+**How OmniRoute solves it:**
+
+- OpenAI-compatible proxy, MCP server, and A2A server in one stack
+- Shared auth, resilience, data store, and observability
+- Consistent policy model across all interaction surfaces
+
+
+
+
+🚀 30. "I need to ship agentic workflows without glue-code sprawl"
+
+Teams lose velocity when stitching multiple ad-hoc services and scripts.
+
+**How OmniRoute solves it:**
+
+- Unified endpoint strategy for clients and agents
+- Built-in protocol management UIs and smoke validation paths
+- Production-ready foundations (security, logging, resilience, backup)
+
+
### Example Playbooks (Integrated Use Cases)
-**Příručka A: Maximalizujte placené předplatné + levné zálohování**```txt
+**Playbook A: Maximize paid subscription + cheap backup**
+
+```txt
Combo: "maximize-claude"
1. cc/claude-opus-4-6
2. glm/glm-4.7
@@ -608,21 +690,23 @@ Combo: "maximize-claude"
Monthly cost: $20 + small backup spend
Outcome: higher quality, near-zero interruption
-````
+```
-**Příručka B: Sada kódování s nulovými náklady**```txt
+**Playbook B: Zero-cost coding stack**
+
+```txt
Combo: "free-forever"
-
-1. gc/gemini-3-flash
-2. if/kimi-k2-thinking
-3. qw/qwen3-coder-plus
+ 1. gc/gemini-3-flash
+ 2. if/kimi-k2-thinking
+ 3. qw/qwen3-coder-plus
Monthly cost: $0
Outcome: stable free coding workflow
+```
-````
+**Playbook C: 24/7 always-on fallback chain**
-**Příručka C: 24/7 vždy zapnutý záložní řetězec**```txt
+```txt
Combo: "always-on"
1. cc/claude-opus-4-6
2. cx/gpt-5.2-codex
@@ -631,122 +715,134 @@ Combo: "always-on"
5. if/kimi-k2-thinking
Outcome: deep fallback depth for deadline-critical workloads
-````
+```
-**Příručka D: Operace agenta s MCP + A2A**```txt
+**Playbook D: Agent ops with MCP + A2A**
-1. Start MCP transport (`omniroute --mcp`) for tool-driven operations
-2. Run A2A tasks via `message/send` and `message/stream`
-3. Observe via /dashboard/endpoint (MCP and A2A tabs)
-4. Toggle services via inline status controls
-
-````
+```txt
+1) Start MCP transport (`omniroute --mcp`) for tool-driven operations
+2) Run A2A tasks via `message/send` and `message/stream`
+3) Observe via /dashboard/endpoint (MCP and A2A tabs)
+4) Toggle services via inline status controls
+```
---
## 🆓 Start Free — Zero Configuration Cost
-> Nastavení kódování AI během několika minut za**$0/měsíc**. Propojte tyto bezplatné účty a použijte vestavěnou kombinaci**Free Stack**.
+> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo.
-| Krok | Akce | Poskytovatelé odemčeni |
-| ---- | --------------------------------------------------- | ------------------------------------------------------------------- |
-| 1 | Connect**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**neomezeno**|
-| 2 | Připojte**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**bez omezení**|
-| 3 | Připojte**Qwen**(kód zařízení) | qwen3-coder-plus, qwen3-coder-flash... —**bez omezení**|
-| 4 | Připojte**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180 000/měsíc zdarma**|
-| 5 | `/dashboard/combos` → Šablona**Free Stack (0 $)**| Round-robin všechny bezplatné poskytovatele automaticky |
+| Step | Action | Providers Unlocked |
+| ---- | -------------------------------------------------- | ------------------------------------------------------------------ |
+| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** |
+| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** |
+| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** |
+| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** |
+| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically |
-**Nasměrujte libovolné IDE/CLI na:**`http://localhost:20128/v1` · Klíč API: `any-string` · Hotovo.
+**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done.
->**Volitelné dodatečné pokrytí (také zdarma):**Klíč Groq API (30 RPM zdarma), NVIDIA NIM (40 RPM zdarma, 70+ modelů), Cerebras (1 M token/den), LongCat API klíč (50 M tokenů/den!), Cloudflare Workers AI (10 000 neuronů/den, 50+ modelů).## Rychlý start
+> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models).
+
+## Rychlý start
### 1) Install and run
```bash
npm install -g omniroute
omniroute
-````
+```
-> **Uživatelé pnpm:**Po instalaci spusťte příkaz `pnpm accept-builds -g`, abyste povolili nativní skripty sestavení vyžadované `better-sqlite3` a `@swc/core`:
+> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`:
>
> ```bash
> pnpm install -g omniroute
-> pnpm schválit-builds -g # Vybrat všechny balíčky → schválit
+> pnpm approve-builds -g # Select all packages → approve
> omniroute
> ```
-Dashboard se otevře na adrese `http://localhost:20128` a základní adresa URL API je `http://localhost:20128/v1`.
+Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`.
-| Příkaz | Popis |
-| ----------------------- | ------------------------------------------------------------------ |
-| "všestranná cesta" | Spustit server (`PORT=20128`, API a řídicí panel na stejném portu) |
-| `omniroute --port 3000` | Nastavte kanonický/API port na 3000 |
-| `omniroute --mcp` | Spustit MCP server (stdio transport) |
-| `omniroute --no-open` | Neotevírat automaticky prohlížeč |
-| `omniroute --help` | Zobrazit nápovědu |
+| Command | Description |
+| ----------------------- | ----------------------------------------------------------- |
+| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) |
+| `omniroute --port 3000` | Set canonical/API port to 3000 |
+| `omniroute --mcp` | Start MCP server (stdio transport) |
+| `omniroute --no-open` | Don't auto-open browser |
+| `omniroute --help` | Show help |
-Volitelný režim rozděleného portu:```bash
+Optional split-port mode:
+
+```bash
PORT=20128 DASHBOARD_PORT=20129 omniroute
-
-# API: http://localhost:20128/v1
-
+# API: http://localhost:20128/v1
# Dashboard: http://localhost:20129
-
-````
+```
### Long-Running Streaming Timeouts
-Pro většinu nasazení potřebujete pouze:
+For most deployments, you only need:
-| Proměnná | Výchozí | Účel |
-| ------------------------- | ------------------------------ | ------------------------------------------------------------- -------------------------------------------------------------- |
-| `REQUEST_TIMEOUT_MS` | "600 000" | Sdílená základní linie pro upstream načítání, skryté časové limity Undici, požadavky otisků prstů TLS a časové limity požadavků/proxy mostu API |
-| `STREAM_IDLE_TIMEOUT_MS` | zdědí `REQUEST_TIMEOUT_MS` | Maximální mezera mezi streamovanými bloky, než OmniRoute přeruší stream SSE |
+| Variable | Default | Purpose |
+| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- |
+| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts |
+| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream |
-Zpětná kompatibilita je zachována: stávající `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` a další proměnná časového limitu pro jednotlivé vrstvy stále fungují a přepisují sdílenou základní linii.
+Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline.
-Pokud potřebujete jemnější ovládání, jsou k dispozici pokročilé přepisy:| Proměnná | Výchozí | Účel |
-| ----------------------------------------- | ------------------------------------------- | --------------------------------------------------------------------- |
-| `FETCH_TIMEOUT_MS` | zdědí `REQUEST_TIMEOUT_MS` | Celkový časový limit upstream požadavku použitý signálem přerušení hlavního načítání |
-| `FETCH_HEADERS_TIMEOUT_MS` | zdědí `FETCH_TIMEOUT_MS` | Undici časový limit pro příjem upstream hlaviček odpovědí |
-| `FETCH_BODY_TIMEOUT_MS` | zdědí `FETCH_TIMEOUT_MS` | Undici časový limit mezi upstream body těla (`0` to zakáže) |
-| `FETCH_CONNECT_TIMEOUT_MS` | "30 000" | Vypršel časový limit připojení Undici TCP |
-| `FETCH_KEEPALIVE_TIMEOUT_MS` | "4000" | Undici nečinný keep-alive socket timeout |
-| `TLS_CLIENT_TIMEOUT_MS` | zdědí `FETCH_TIMEOUT_MS` | Vypršel časový limit pro požadavky otisku TLS provedené prostřednictvím `wreq-js` |
-| `API_BRIDGE_PROXY_TIMEOUT_MS` | zdědí `REQUEST_TIMEOUT_MS` nebo `30000` | Časový limit pro přesměrování proxy `/v1` z portu API na port řídicího panelu |
-| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Časový limit příchozího požadavku na serveru API mostu |
-| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | "60 000" | Časový limit příchozí hlavičky na serveru API mostu |
-| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | "5000" | Udržovací časový limit na serveru API mostu |
-| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | "0" | Časový limit nečinnosti soketu na serveru API mostu (`0` jej zakáže) |
+Advanced overrides are available if you need finer control:
-Pokud spouštíte OmniRoute za Nginx, Caddy, Cloudflare nebo jiným reverzním proxy, ujistěte se, že proxy
-časové limity jsou také vyšší než časové limity streamu/načtení OmniRoute.### 2) Connect providers and create your API key
+| Variable | Default | Purpose |
+| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- |
+| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal |
+| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers |
+| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) |
+| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout |
+| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout |
+| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` |
+| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port |
+| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server |
+| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server |
+| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server |
+| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) |
-1. Otevřete Dashboard → `Providers` a připojte alespoň jednoho poskytovatele (OAuth nebo API klíč).
-2. Otevřete Dashboard → `Koncové body` a vytvořte klíč API.
-3. (Volitelné) Otevřete Dashboard → `Komba` a nastavte svůj záložní řetězec.### 3) Point your coding tool to OmniRoute
+If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy
+timeouts are also higher than your OmniRoute stream/fetch timeouts.
+
+### 2) Connect providers and create your API key
+
+1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key).
+2. Open Dashboard → `Endpoints` and create an API key.
+3. (Optional) Open Dashboard → `Combos` and set your fallback chain.
+
+### 3) Point your coding tool to OmniRoute
```txt
Base URL: http://localhost:20128/v1
API Key: [copy from Endpoint page]
Model: if/kimi-k2-thinking (or any provider/model prefix)
-````
+```
-Pracuje s Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode a SDK kompatibilní s OpenAI.### 4) Enable and validate protocols (v2.0)
+Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs.
-**MCP (pro operace řízené nástrojem):**```bash
+### 4) Enable and validate protocols (v2.0)
+
+**MCP (for tool-driven operations):**
+
+```bash
omniroute --mcp
+```
-````
-
-Poté připojte svého MCP klienta přes `stdio` a otestujte nástroje jako:
+Then connect your MCP client over `stdio` and test tools like:
- `omniroute_get_health`
-- `combos_omniroute_list_combos`
+- `omniroute_list_combos`
-**A2A (pro pracovní postupy mezi agenty):**```bash
+**A2A (for agent-to-agent workflows):**
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
```bash
curl -X POST http://localhost:20128/a2a \
@@ -760,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \
npm run test:protocols:e2e
```
-Tato sada ověřuje skutečné klientské toky MCP a A2A proti běžící aplikaci.### Alternative: run from source
+This suite validates real MCP and A2A client flows against a running app.
+
+### Alternative: run from source
```bash
cp .env.example .env
@@ -768,13 +866,13 @@ npm install
PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev
```
-
-Void Linux (šablona `xbps-src`)
+
+Void Linux (`xbps-src` template)
-Pro uživatele Void Linuxu můžete vytvořit nativní balíček pomocí `xbps-src`. Uložte tento blok jako `srcpkgs/omniroute/template`:```bash
+For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`:
+```bash
# Template file for 'omniroute'
-
pkgname=omniroute
version=3.4.1
revision=1
@@ -786,7 +884,7 @@ license="MIT"
homepage="https://github.com/diegosouzapw/OmniRoute"
distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz"
checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b
-system_accounts="\_omniroute"
+system_accounts="_omniroute"
omniroute_homedir="/var/lib/omniroute"
export NODE_ENV=production
export npm_config_engine_strict=false
@@ -794,71 +892,70 @@ export npm_config_loglevel=error
export npm_config_fund=false
export npm_config_audit=false
-do_build() { # Determine target CPU arch for node-gyp
-local \_gyp_arch
-case "$XBPS_TARGET_MACHINE" in
-aarch64*) \_gyp_arch=arm64 ;;
-armv7*|armv6*) \_gyp_arch=arm ;;
-i686*) \_gyp_arch=ia32 ;;
-\*) \_gyp_arch=x64 ;;
-esac
+do_build() {
+ # Determine target CPU arch for node-gyp
+ local _gyp_arch
+ case "$XBPS_TARGET_MACHINE" in
+ aarch64*) _gyp_arch=arm64 ;;
+ armv7*|armv6*) _gyp_arch=arm ;;
+ i686*) _gyp_arch=ia32 ;;
+ *) _gyp_arch=x64 ;;
+ esac
- # 1) Install all deps – skip scripts (no network in do_build, native modules
- # compiled separately below; better-sqlite3 is serverExternalPackage so
- # Next.js does not execute it during next build)
- NODE_ENV=development npm ci --ignore-scripts
+ # 1) Install all deps – skip scripts (no network in do_build, native modules
+ # compiled separately below; better-sqlite3 is serverExternalPackage so
+ # Next.js does not execute it during next build)
+ NODE_ENV=development npm ci --ignore-scripts
- # 2) Build the Next.js standalone bundle
- npm run build
+ # 2) Build the Next.js standalone bundle
+ npm run build
- # 3) Copy static assets into standalone
- cp -r .next/static .next/standalone/.next/static
- [ -d public ] && cp -r public .next/standalone/public || true
+ # 3) Copy static assets into standalone
+ cp -r .next/static .next/standalone/.next/static
+ [ -d public ] && cp -r public .next/standalone/public || true
- # 4) Compile better-sqlite3 native binding for the target architecture.
- # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
- # without npm altering them.
- local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
- (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
+ # 4) Compile better-sqlite3 native binding for the target architecture.
+ # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
+ # without npm altering them.
+ local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
+ (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
- # 5) Place the compiled binding into the standalone bundle
- local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
- mkdir -p "$_bs3_release"
- cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
+ # 5) Place the compiled binding into the standalone bundle
+ local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
+ mkdir -p "$_bs3_release"
+ cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
- # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
- # so sharp is not used at runtime; x64 .so files would break aarch64 strip
- rm -rf .next/standalone/node_modules/@img
-
- # 7) Copy pino runtime deps omitted by Next.js static analysis:
- # pino-abstract-transport – required by pino's worker thread
- # split2 – dep of pino-abstract-transport
- # process-warning – dep of pino itself
- for _mod in pino-abstract-transport split2 process-warning; do
- cp -r "node_modules/$_mod" .next/standalone/node_modules/
- done
+ # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
+ # so sharp is not used at runtime; x64 .so files would break aarch64 strip
+ rm -rf .next/standalone/node_modules/@img
+ # 7) Copy pino runtime deps omitted by Next.js static analysis:
+ # pino-abstract-transport – required by pino's worker thread
+ # split2 – dep of pino-abstract-transport
+ # process-warning – dep of pino itself
+ for _mod in pino-abstract-transport split2 process-warning; do
+ cp -r "node_modules/$_mod" .next/standalone/node_modules/
+ done
}
do_check() {
-npm run test:unit
+ npm run test:unit
}
do_install() {
-vmkdir usr/lib/omniroute/.next
+ vmkdir usr/lib/omniroute/.next
- vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
+ vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
- # Prevent removal of empty Next.js app router dirs by the post-install hook
- for _d in \
- .next/standalone/.next/server/app/dashboard \
- .next/standalone/.next/server/app/dashboard/settings \
- .next/standalone/.next/server/app/dashboard/providers; do
- touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
- done
-
- cat > "${WRKDIR}/omniroute" <<'EOF'
+ # Prevent removal of empty Next.js app router dirs by the post-install hook
+ for _d in \
+ .next/standalone/.next/server/app/dashboard \
+ .next/standalone/.next/server/app/dashboard/settings \
+ .next/standalone/.next/server/app/dashboard/providers; do
+ touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
+ done
+ cat > "${WRKDIR}/omniroute" <<'EOF'
#!/bin/sh
export PORT="${PORT:-20128}"
export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}"
@@ -870,10 +967,9 @@ EOF
}
post_install() {
-vlicense LICENSE
+ vlicense LICENSE
}
-
-````
+```
@@ -881,9 +977,11 @@ vlicense LICENSE
## 🐳 Docker
-OmniRoute je k dispozici jako veřejný obrázek Dockeru na [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
+OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
-**Rychlý běh:**```bash
+**Quick run:**
+
+```bash
docker run -d \
--name omniroute \
--restart unless-stopped \
@@ -891,85 +989,96 @@ docker run -d \
-p 20128:20128 \
-v omniroute-data:/app/data \
diegosouzapw/omniroute:latest
-````
+```
-**Se souborem prostředí:**```bash
+**With environment file:**
+```bash
# Copy and edit .env first
-
cp .env.example .env
docker run -d \
- --name omniroute \
- --restart unless-stopped \
- --stop-timeout 40 \
- --env-file .env \
- -p 20128:20128 \
- -v omniroute-data:/app/data \
- diegosouzapw/omniroute:latest
+ --name omniroute \
+ --restart unless-stopped \
+ --stop-timeout 40 \
+ --env-file .env \
+ -p 20128:20128 \
+ -v omniroute-data:/app/data \
+ diegosouzapw/omniroute:latest
+```
-````
+**Using Docker Compose:**
-**Použití Docker Compose:**```bash
+```bash
# Base profile (no CLI tools)
docker compose --profile base up -d
# CLI profile (Claude Code, Codex, OpenClaw built-in)
docker compose --profile cli up -d
-````
+```
-Podpora řídicího panelu pro nasazení Dockeru nyní zahrnuje**Cloudflare Quick Tunnel**na jedno kliknutí na `Dashboard → Endpoints`. První povolí stahování `cloudflared` pouze v případě potřeby, spustí dočasný tunel k vašemu aktuálnímu koncovému bodu `/v1` a zobrazí vygenerovanou URL `https://*.trycloudflare.com/v1` přímo pod vaší normální veřejnou adresou URL.
+Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL.
-Poznámky:
+Notes:
-- URL Quick Tunnel jsou dočasné a mění se po každém restartu.
-- Rychlé tunely se po restartu OmniRoute nebo kontejneru automaticky neobnoví. V případě potřeby je znovu povolte z řídicího panelu.
-- Spravovaná instalace aktuálně podporuje Linux, macOS a Windows na `x64` / `arm64`.
-- Spravované rychlé tunely jsou výchozí pro přenos HTTP/2, aby se zabránilo hlučným varováním vyrovnávací paměti QUIC UDP v prostředí s omezenými kontejnery. Pokud chcete jiný přenos, nastavte `CLOUDFLARED_PROTOCOL=quic` nebo `auto`.
-- Obrazy Dockeru svazují kořeny systémové CA a předávají je spravovanému `cloudflared`, což zabraňuje selhání důvěryhodnosti TLS při zavádění tunelu uvnitř kontejneru.
-- SQLite běží v režimu WAL. `Docker stop` by mělo být povoleno dokončit, aby OmniRoute mohl zkontrolovat nejnovější změny zpět do `storage.sqlite`.
-- V přiložených souborech Compose je již nastavena doba odkladu 40 s. Pokud spouštíte obraz přímo, ponechte hodnotu `--stop-timeout 40` (nebo podobnou), aby ruční zastavení nepřerušilo čištění při vypnutí.
-- Nastavte `CLOUDFLARED_BIN=/absolutní/cesta/k/cloudflared`, pokud chcete, aby OmniRoute místo stahování používal existující binární soubor.
+- Quick Tunnel URLs are temporary and change after every restart.
+- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed.
+- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`.
+- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport.
+- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container.
+- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`.
+- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup.
+- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one.
-**Použití Docker Compose s Caddy (HTTPS Auto-TLS):**
+**Using Docker Compose with Caddy (HTTPS Auto-TLS):**
-OmniRoute lze bezpečně zpřístupnit pomocí automatického zřizování SSL Caddy. Ujistěte se, že záznam DNS A vaší domény ukazuje na IP adresu vašeho serveru.```yaml
+OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP.
+
+```yaml
services:
-omniroute:
-image: diegosouzapw/omniroute:latest
-container_name: omniroute
-restart: unless-stopped
-volumes: - omniroute-data:/app/data
-environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com
+ omniroute:
+ image: diegosouzapw/omniroute:latest
+ container_name: omniroute
+ restart: unless-stopped
+ volumes:
+ - omniroute-data:/app/data
+ environment:
+ - PORT=20128
+ - NEXT_PUBLIC_BASE_URL=https://your-domain.com
-caddy:
-image: caddy:latest
-container_name: caddy
-restart: unless-stopped
-ports: - "80:80" - "443:443"
-command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
+ caddy:
+ image: caddy:latest
+ container_name: caddy
+ restart: unless-stopped
+ ports:
+ - "80:80"
+ - "443:443"
+ command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
volumes:
-omniroute-data:
+ omniroute-data:
+```
-````
+| Image | Tag | Size | Description |
+| ------------------------ | -------- | ------ | --------------------- |
+| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release |
+| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version |
-| Obrázek | Štítek | Velikost | Popis |
-| ------------------------- | -------- | ------ | ---------------------- |
-| `diegosouzapw/omniroute` | "nejnovější" | ~250 MB | Poslední stabilní verze |
-| `diegosouzapw/omniroute` | "1.0.3" | ~250 MB | Aktuální verze |---
+---
## 🖥️ Desktop App — Offline & Always-On
-> 🆕**NOVINKA!**OmniRoute je nyní k dispozici jako**nativní desktopová aplikace**pro Windows, macOS a Linux.
+> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux.
-Spusťte OmniRoute jako samostatnou desktopovou aplikaci – pro místní modely není potřeba žádný terminál, žádný prohlížeč ani internet. Aplikace založená na Electronu zahrnuje:
+Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes:
-- 🖥️**Nativní okno**— Vyhrazené okno aplikace s integrací na systémové liště
-- 🔄**Auto-Start**– Spusťte OmniRoute při přihlášení do systému
-- 🔔**Nativní oznámení**– Získejte upozornění na vyčerpání kvóty nebo problémy s poskytovatelem
-- ⚡**Instalace jedním kliknutím**— NSIS (Windows), DMG (macOS), AppImage (Linux)
-- 🌐**Režim offline**– Funguje plně offline s přibaleným serverem### Rychlý start
+- 🖥️ **Native Window** — Dedicated app window with system tray integration
+- 🔄 **Auto-Start** — Launch OmniRoute on system login
+- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues
+- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux)
+- 🌐 **Offline Mode** — Works fully offline with bundled server
+
+### Rychlý start
```bash
# Development mode
@@ -980,308 +1089,370 @@ npm run electron:build # Current platform
npm run electron:build:win # Windows (.exe)
npm run electron:build:mac # macOS (.dmg) — x64 & arm64
npm run electron:build:linux # Linux (.AppImage)
-````
+```
### System Tray
-Když je minimalizován, OmniRoute žije v systémové liště s rychlými akcemi:
+When minimized, OmniRoute lives in your system tray with quick actions:
-- Otevřete palubní desku
-- Změňte port serveru
-- Ukončete aplikaci
+- Open dashboard
+- Change server port
+- Quit application
-📖 Úplná dokumentace: [`electron/README.md`](electron/README.md)---
+📖 Full documentation: [`electron/README.md`](electron/README.md)
+
+---
## 💰 Pricing at a Glance
-| Úroveň | Poskytovatel | Cena | Obnovení kvóty | Nejlepší pro |
-| ----------------- | --------------------------- | ------------------------------- | --------------------------- | ------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| **💳 PŘEDPLATNÉ** | Claude Code (Pro) | 20 $/měsíc | 5h + týdně | Již přihlášeno |
-| | Codex (Plus/Pro) | 20–200 USD/měsíc | 5h + týdně | Uživatelé OpenAI |
-| | Gemini CLI | **ZDARMA** | 180 tis./měsíc + 1 tis./den | Každý! |
-| | GitHub Copilot | 10–19 USD/měsíc | Měsíčně | Uživatelé GitHubu |
-| **🔑 API KEY** | NVIDIA NIM | **ZDARMA**(dev forever) | ~40 RPM | 70+ otevřených modelů |
-| | Cerebras | **ZDARMA**(1 milion toku/den) | 60 000 TPM / 30 RPM | Nejrychlejší na světě |
-| | Groq | **ZDARMA**(30 RPM) | 14,4K RPD | Ultra rychlá lama/gemma |
-| | DeepSeek V3.2 | 0,27 $ / 1,10 $ za 1 milion | Žádné | Nejlepší zdůvodnění cena/kvalita |
-| | xAI Grok-4 Fast | **0,20 $/0,50 $ za 1M**🆕 | Žádné | Nejrychlejší + volání nástroje, ultranízké |
-| | xAI Grok-4 (standardní) | 0,20 $/1,50 $ za 1M 🆕 | Žádné | Reasoning vlajková loď od xAI |
-| | Mistral | Vyzkoušení zdarma + placené | Omezená sazba | Evropská umělá inteligence |
-| | OpenRouter | Platba za použití | Žádné | 100+ modelů agr. |
-| **💰 LEVNĚ** | GLM-5 (přes Z.AI) 🆕 | 0,5 $/1 mil. | Denně 10:00 | 128K výstup, nejnovější vlajková loď |
-| | GLM-4.7 | 0,6 $/1 mil. | Denně 10:00 | Záloha rozpočtu |
-| | MiniMax M2,5 🆕 | Vstup 0,3 $/1 milion | 5hodinové válcování | Úvahy + agentské úkoly |
-| | MiniMax M2.1 | 0,2 $/1 milion | 5hodinové válcování | Nejlevnější varianta |
-| | Kimi K2.5 (Moonshot API) 🆕 | Platba za použití | Žádné | Přímý přístup Moonshot API |
-| | Kimi K2 | 9 $/měsíc byt | 10 milionů tokenů/měsíc | Předvídatelné náklady |
-| **🆓 ZDARMA** | Qoder | **$0** | Neomezené | 5 modelů neomezeně |
-| | Qwen | **$0** | Neomezené | 4 modely neomezeně |
-| | Kiro | **$0** | Neomezené | Claude Sonnet/Haiku (stavitel AWS) |
-| | LongCat Flash-Lite 🆕 | **$0**(50 milionů toku/den 🔥) | 1 RPS | Největší bezplatná kvóta na Zemi |
-| | Opylování AI 🆕 | **$0**(není potřeba žádný klíč) | 1 požadavek/15s | GPT-5, Claude, DeepSeek, Llama 4 |
-| | Cloudflare Workers AI 🆕 | **$0**(10 000 neuronů/den) | ~150 resp./den | 50+ modelů, globální náskok |
-| | Scaleway AI 🆕 | **$0**(celkem 1 milion tokenů) | Omezená sazba | EU/GDPR, Qwen3 235B, Lama 70B | > 🆕**Přidané nové modely (březen 2026):**Rodina Grok-4 Fast za 0,20 $/0,50 $/M (porovnávací rychlost 1143 ms – o 30 % rychlejší než Gemini 2.5 Flash), GLM-5 přes Z.AI s výstupem 128K, aktualizovaná cena MiniMax M2.5 V5, přímé zdůvodnění Kimi K2.2 Moon. |
+| Tier | Provider | Cost | Quota Reset | Best For |
+| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- |
+| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed |
+| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users |
+| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! |
+| | GitHub Copilot | $10-19/mo | Monthly | GitHub users |
+| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models |
+| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest |
+| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma |
+| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning |
+| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow |
+| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI |
+| | Mistral | Free trial + paid | Rate limited | European AI |
+| | OpenRouter | Pay-per-use | None | 100+ models aggr. |
+| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship |
+| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup |
+| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks |
+| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option |
+| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access |
+| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost |
+| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited |
+| | Qwen | **$0** | Unlimited | 4 models unlimited |
+| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) |
+| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth |
+| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 |
+| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge |
+| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B |
-**💡 Combo Stack 0 $ — Kompletní bezplatné nastavení:**```
+> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API.
+**💡 $0 Combo Stack — The Complete Free Setup:**
+
+```
# 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever
+Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
+Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
+Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
+Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
+NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+```
-Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
-Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
-LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
-Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
-Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
-Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
-Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
-Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
-Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
-NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
-Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever.
-````
-
-**Nulové náklady. Nikdy nepřestane kódovat.**Nakonfigurujte si to jako jednu kombinaci OmniRoute a všechna nouzová řešení se stanou automaticky – žádné ruční přepínání.---
+---
---
## 🆓 Free Models — What You Actually Get
-> Všechny níže uvedené modely jsou**100% zdarma bez nutnosti použití kreditní karty**. OmniRoute mezi nimi automaticky směruje, když dojde jedna kvóta – zkombinujte je všechny a získáte nerozbitnou kombinaci 0 $.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
+> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo.
-| Model | Předpona | Limit | Limit sazby |
-| -------------------- | ------ | ------------- | ---------------------- |
-| `claude-sonnet-4.5` | `kr/` |**Neomezeno**| Žádný hlášený denní limit |
-| `claude-haiku-4,5` | `kr/` |**Neomezeno**| Žádný hlášený denní limit |
-| `claude-opus-4.6` | `kr/` |**Neomezeno**| Nejnovější Opus přes Kiro |### 🟢 QODER MODELS (Free PAT via qodercli)
+### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
-| Model | Předpona | Limit | Limit sazby |
-| ------------------- | ------ | ------------- | ---------------- |
-| "kimi-k2-myšlení" | `jestli/` |**Neomezeno**| Žádný nahlášený strop |
-| `qwen3-coder-plus` | `jestli/` |**Neomezeno**| Žádný nahlášený strop |
-| `deepseek-r1` | `jestli/` |**Neomezeno**| Žádný nahlášený strop |
-| `minimax-m2.1` | `jestli/` |**Neomezeno**| Žádný nahlášený strop |
-| "kimi-k2" | `jestli/` |**Neomezeno**| Žádný nahlášený strop |
+| Model | Prefix | Limit | Rate Limit |
+| ------------------- | ------ | ------------- | --------------------- |
+| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro |
-> Doporučený způsob připojení:**Personal Access Token + `qodercli`**. OAuth prohlížeče je
-> experimentální a ve výchozím nastavení zakázáno, pokud nejsou nakonfigurovány proměnné prostředí `QODER_OAUTH_*`.### 🟡 QWEN MODELS (Device Code Auth)
+### 🟢 QODER MODELS (Free PAT via qodercli)
-| Model | Předpona | Limit | Limit sazby |
-| -------------------- | ------ | ------------- | -------------------- |
-| `qwen3-coder-plus` | `qw/` |**Neomezeno**| Žádný nahlášený strop |
-| `qwen3-coder-flash` | `qw/` |**Neomezeno**| Žádný nahlášený strop |
-| `qwen3-coder-next` | `qw/` |**Neomezeno**| Žádný nahlášený strop |
-| "model vidění" | `qw/` |**Neomezeno**| Multimodální (obrázky) |### 🟣 GEMINI CLI (Google OAuth)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------ | ------ | ------------- | --------------- |
+| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap |
+| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap |
+| `deepseek-r1` | `if/` | **Unlimited** | No reported cap |
+| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap |
+| `kimi-k2` | `if/` | **Unlimited** | No reported cap |
-| Model | Předpona | Limit | Limit sazby |
-| ------------------------- | ------ | ---------------------------- | ------------- |
-| `gemini-3-flash-preview` | `gc/` |**180 tis./měsíc**+ 1 tis./den | Měsíční reset |
-| `gemini-2.5-pro` | `gc/` | 180 tis./měsíc (sdílený bazén) | Vysoká kvalita |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
+> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is
+> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured.
-| Úroveň | Denní limit | Limit sazby | Poznámky |
-| ---------- | ------------ | ----------- | ------------------------------------------------------- |
-| Zdarma (Dev) | Žádný token cap |**~40 RPM**| 70+ modelů; přechod na limity čisté sazby v polovině roku 2025 |
+### 🟡 QWEN MODELS (Device Code Auth)
-Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------- | ------ | ------------- | ------------------- |
+| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap |
+| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) |
-| Úroveň | Denní limit | Limit sazby | Poznámky |
-| ---- | ------------------ | ----------------- | -------------------------------------------- |
-| Zdarma |**1 mil. tokenů/den**| 60 000 TPM / 30 RPM | Světově nejrychlejší odvození LLM; resetuje denně |
+### 🟣 GEMINI CLI (Google OAuth)
-Dostupné zdarma: `lama-3.3-70b`, `lama-3.1-8b`, `deepseek-r1-distill-lama-70b`### 🔴 GROQ (Free API Key — console.groq.com)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------------ | ------ | --------------------------- | ------------- |
+| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset |
+| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality |
-| Úroveň | Denní limit | Limit sazby | Poznámky |
-| ---- | ------------- | ----------------- | ------------------------------------------ |
-| Zdarma |**14,4K RPD**| 30 ot./min na model | Žádná kreditní karta; 429 na limit, neúčtuje se |
+### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
-Dostupné zdarma: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---------- | ------------ | ----------- | ------------------------------------------------------ |
+| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 |
-| Model | Předpona | Denní kvóta zdarma | Poznámky |
-| ------------------------------ | ------ | ------------------ | ------------------------ |
-| "LongCat-Flash-Lite" | `lc/` |**50 milionů tokenů**💥 | Největší bezplatná kvóta všech dob |
-| "LongCat-Flash-Chat" | `lc/` | 500 000 tokenů | Víceotáčkový chat |
-| "LongCat-Flash-Thinking" | `lc/` | 500 000 tokenů | Zdůvodnění / CoT |
-| "LongCat-Flash-Thinking-2601" | `lc/` | 500 000 tokenů | Verze z ledna 2026 |
-| "LongCat-Flash-Omni-2603" | `lc/` | 500 000 tokenů | Multimodální |
+Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`
-> 100 % zdarma ve veřejné beta verzi. Zaregistrujte se na [longcat.chat](https://longcat.chat) pomocí e-mailu nebo telefonu. Resetuje se denně v 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
-| Model | Předpona | Limit sazby | Poskytovatel za |
-| ---------- | ------ | ---------- | ------------------- |
-| "openai" | `pol/` | 1 požadavek/15s | GPT-5 |
-| "claude" | `pol/` | 1 požadavek/15s | Antropický Claude |
-| "blíženci" | `pol/` | 1 požadavek/15s | Google Gemini |
-| "hluboké vyhledávání" | `pol/` | 1 požadavek/15s | DeepSeek V3 |
-| "lama" | `pol/` | 1 požadavek/15s | Meta Llama 4 Scout |
-| "mistrál" | `pol/` | 1 požadavek/15s | Mistral AI |
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---- | ----------------- | ---------------- | ------------------------------------------- |
+| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily |
-> ✨**Nulové tření:**Žádná registrace, žádný klíč API. Přidejte poskytovatele Pollinations s prázdným polem klíče a funguje to okamžitě.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`
-| Úroveň | Denní neurony | Ekvivalentní použití | Poznámky |
-| ---- | ------------- | ---------------------------------------- | ------------------------ |
-| Zdarma |**10 000**| ~150 LLM resp / 500s audio / 15K vložení | Global edge, 50+ modelů |
+### 🔴 GROQ (Free API Key — console.groq.com)
-Oblíbené bezplatné modely: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (zvuk zdarma!), `@cf/qwen/qwen2.5-coder-`1
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---- | ------------- | ---------------- | ----------------------------------------- |
+| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged |
-> Vyžaduje API Token + ID účtu z [dash.cloudflare.com](https://dash.cloudflare.com). Uložte ID účtu v nastavení poskytovatele.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`
-| Úroveň | Kvóta zdarma | Umístění | Poznámky |
-| ---- | ------------- | ------------ | ------------------------------------ |
-| Zdarma |**1 milion tokenů**| 🇫🇷 Paříž, EU | V rámci limitů není potřeba žádná kreditní karta |
+### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
-Dostupné zdarma: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
+| Model | Prefix | Daily Free Quota | Notes |
+| ----------------------------- | ------ | ----------------- | ----------------------- |
+| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever |
+| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat |
+| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT |
+| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version |
+| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal |
-> V souladu s EU/GDPR. Získejte API klíč na [console.scaleway.com](https://console.scaleway.com).
+> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC.
->**💡 The Ultimate Free Stack (11 poskytovatelů, 0 $ navždy):**
+### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+
+| Model | Prefix | Rate Limit | Provider Behind |
+| ---------- | ------ | ---------- | ------------------ |
+| `openai` | `pol/` | 1 req/15s | GPT-5 |
+| `claude` | `pol/` | 1 req/15s | Anthropic Claude |
+| `gemini` | `pol/` | 1 req/15s | Google Gemini |
+| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 |
+| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout |
+| `mistral` | `pol/` | 1 req/15s | Mistral AI |
+
+> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately.
+
+### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+
+| Tier | Daily Neurons | Equivalent Usage | Notes |
+| ---- | ------------- | --------------------------------------- | ----------------------- |
+| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models |
+
+Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct`
+
+> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings.
+
+### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+
+| Tier | Free Quota | Location | Notes |
+| ---- | ------------- | ------------ | ----------------------------------- |
+| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits |
+
+Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
+
+> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com).
+
+> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):**
>
> ```
-> Kiro (kr/) → Claude Sonnet/Haiku NEOMEZENO
-> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
-> LongCat Lite (lc/) → LongCat-Flash-Lite – 50 milionů tokenů/den 🔥
-> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — není potřeba žádný klíč
-> Qwen (qw/) → modely qwen3-kodér NEOMEZENÉ
-> Gemini (gemini/) → Gemini 2.5 Flash — 1 500 req/den zdarma
-> Cloudflare AI (cf/) → 50+ modelů — 10 000 neuronů/den
-> Scaleway (scw/) → Qwen3 235B, Llama 70B – 1M bezplatných tokenů (EU)
-> Groq (groq/) → Llama/Gemma – ultrarychlé 14,4 000 požadavků/den
-> NVIDIA NIM (nvidia/) → 70+ otevřených modelů — 40 RPM navždy
-> Cerebras (cerebras/) → Nejrychlejší lama/Qwen na světě – 1 milion toku/den
-> ```## 🎙️ Free Transcription Combo
+> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+> Qwen (qw/) → qwen3-coder models UNLIMITED
+> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free
+> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day
+> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast
+> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+> ```
-> Přepis jakéhokoli zvuku/videa za**$0**— Deepgram vede s 200 $ zdarma, AssemblyAI 50 $ nouzové zálohy, Groq Whisper jako neomezené nouzové zálohování.
+## 🎙️ Free Transcription Combo
-| Poskytovatel | Kredity zdarma | Nejlepší modelka | Limit sazby |
-| ------------------ | ----------------------- | --------------------------------------------- | ----------------------------- |
-| 🢢**Deepgram**|**200 $ zdarma**(registrace) | `nova-3` — nejlepší přesnost, více než 30 jazyků | Žádný limit RPM na bezplatné kredity |
-| 🔵**SestaveníAI**|**50 $ zdarma**(registrace) | `universal-3-pro` — kapitoly, sentiment, PII | Žádný limit RPM na bezplatné kredity |
-| 🔴**Groq**|**Navždy zdarma**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (rychlost omezená) |
+> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup.
-**Doporučená kombinace v `/dashboard/combos`:**```
+| Provider | Free Credits | Best Model | Rate Limit |
+| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- |
+| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits |
+| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits |
+| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) |
+
+**Suggested combo in `/dashboard/combos`:**
+
+```
Name: free-transcription
Strategy: Priority
Nodes:
[1] deepgram/nova-3 → uses $200 free first
[2] assemblyai/universal-3-pro → fallback when Deepgram credits run out
[3] groq/whisper-large-v3 → free forever, emergency fallback
-````
+```
-Poté v `/dashboard/media` → karta**Přepis**: nahrajte jakýkoli zvukový nebo video soubor → vyberte svůj kombinovaný koncový bod → získejte přepis v podporovaných formátech.## 💡 Key Features
+Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats.
-OmniRoute v2.0 je postaven jako operační platforma, nikoli pouze jako přenosová proxy.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
+## 💡 Key Features
-| Funkce | Co to dělá |
-| ---------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ |
-| ⚡**Grok-4 Fast Family** | xAI modely za 0,20 $/0,50 $/M – srovnávací 1143 ms (o 30 % rychlejší než Gemini 2.5 Flash) |
-| 🧠**GLM-5 přes Z.AI** | 128 000 výstupní kontext, 0,5 $/1 milion – nejnovější vlajková loď z rodiny GLM |
-| 🔮**MiniMax M2.5** | Úvahy + agentní úkoly za 0,30 $/1 milion – významný upgrade z M2,1 |
-| 🎯**toolCalling Flag na model** | „ToolCalling: true/false“ pro model v registru – AutoCombo přeskočí modely, které nepodporují nástroje |
-| 🌍**Multilingual Intent Detection** | Klíčová slova PT/ZH/ES/AR v hodnocení AutoCombo – lepší výběr modelu pro neanglický obsah |
-| 📊**Zástupy založené na benchmarku** | Skutečná latence p95 z kombinovaného bodování zdrojů živých požadavků – AutoCombo se učí ze skutečných dat |
-| 🔁**Požádat o deduplikaci** | Okno pro odstranění duplicitního obsahu založené na hašování obsahu – bezpečné pro více agentů, zabraňuje duplicitním poplatkům |
-| 🔌**Strategie připojitelného směrovače** | Rozšiřitelné rozhraní `RouterStrategy` — přidejte vlastní logiku směrování jako zásuvné moduly | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP |
+OmniRoute v3.5 is built as an operational platform, not just a relay proxy.
-| Funkce | Co to dělá |
-| -------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------- |
-| 🎮**Modelové hřiště** | Stránka řídicího panelu pro přímé testování jakéhokoli modelu — voliče poskytovatele/modelu/koncového bodu, editor Monaco, streamování, přerušení, načasování |
-| 🔏**CLI Fingerprint Matching** | Uspořádání záhlaví/těla podle poskytovatele tak, aby odpovídalo nativním signaturám CLI – přepněte podle poskytovatele v Nastavení > Zabezpečení.**Vaše IP adresa proxy je zachována** |
-| 🤝**Podpora ACP (Protokol klienta agenta)** | Objevování agentů CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 dalších), proces spawner, koncový bod `/api/acp/agents` |
-| 🤖**Hlavní panel agentů AKT** | Debug › Stránka Agenti — mřížka 14 agentů se stavem instalace, verzí, uživatelským formulářem agenta pro libovolný nástroj CLI. Uživatelé**OpenCode**získají tlačítko „Stáhnout opencode.json“, které automaticky vygeneruje konfiguraci připravenou k použití se všemi dostupnými modely. |
-| 🔧**Směrování vlastního modelu `apiFormat`** | Vlastní modely s `apiFormat: "responses"` nyní správně směrují do překladače Responses API |
-| 🏢**Codex Workspace Isolation** | Více pracovních prostorů Codex na e-mail — OAuth správně odděluje připojení podle ID pracovního prostoru |
-| 🔄**Elektronová automatická aktualizace** | Desktopová aplikace kontroluje aktualizace + automatická instalace při restartu | ### 🤖 Agent & Protocol Operations (v2.0) |
+### 🆕 New — v3.5.5 Highlights (Apr 2026)
-| Funkce | Co to dělá |
-| -------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- |
-| 🔧**MCP Server (25 nástrojů)** | Nástroje IDE/agenta prostřednictvím 3 přenosů: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 jader + 3 paměti + 4 nástroje pro dovednosti |
-| 🤝**Server A2A (JSON-RPC + SSE)** | Provádění úlohy agent-agent se synchronizací a streamováním |
-| 🧭**Stránka konsolidovaných koncových bodů** | Stránka správy s kartami s kartami Endpoint Proxy, MCP, A2A a API Endpoints |
-| 🎚️**Přepínače aktivace/deaktivace služby** | Spínače ON/OFF pro MCP a A2A s trvalým nastavením (výchozí: OFF) |
-| 🛰️**MCP Runtime Heartbeat** | Skutečný stav procesu (pid, doba provozu, doba srdečního tepu, transport, režim rozsahu) |
-| 📋**MCP Audit Trail** | Filtrovatelné protokoly auditu s úspěchem/neúspěchem a přiřazením klíče |
-| 🔐**Vymáhání rozsahu MCP** | 10 podrobných oprávnění k rozsahu pro řízený přístup k nástrojům |
-| 📡**A2A Task Lifecycle Management** | Vypsat/filtrovat úlohy, zkontrolovat události/artefakty, zrušit běžící úlohy |
-| 📋**Zjištění karty agenta** | `/.well-known/agent.json` pro automatické zjišťování klienta |
-| 🧪**Protokol E2E Test Harness** | Skutečný MCP SDK + klient A2A toky v `test:protocols:e2e` |
-| ⚙️**Provozní ovládací prvky** | Kombinace přepínačů, použití profilů odolnosti, resetování jističů z jedné ovládací plochy | ### 🧠 Routing & Intelligence |
+| Feature | What It Does |
+| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation |
+| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs |
+| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions |
+| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection |
+| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts |
-| Funkce | Co to dělá |
-| ----------------------------------------------- | ----------------------------------------------------------------------------- | ----------------------- |
-| 🎯**Chytrý 4úrovňový záložní zdroj** | Automatická trasa: Předplatné → Klíč API → Levné → Zdarma |
-| 📊**Sledování kvót v reálném čase** | Živý počet tokenů + reset odpočítávání na poskytovatele |
-| 🔄**Formátový překlad** | OpenAI ↔ Claude ↔ Gemini ↔ Odpovědi s převody bezpečnými pro schéma |
-| 👥**Podpora více účtů** | Více účtů na poskytovatele s inteligentním výběrem |
-| 🔄**Automatické obnovení tokenu** | Tokeny OAuth se automaticky obnovují s opakováním |
-| 🎨**Vlastní kombinace** | 9 vyvažovacích strategií + řízení záložního řetězce |
-| 🌐**Wildcard Router** | `poskytovatel/*` dynamické směrování |
-| 🧠**Přemýšlení o kontrolách rozpočtu** | Limity průchozího, automatického, vlastního a adaptivního uvažování |
-| 🔀**Aliasy modelů** | Vestavěný + vlastní model aliasing a bezpečnost migrace |
-| ⚡**Degradace pozadí** | Směrujte úlohy s nízkou prioritou na pozadí na levnější modely |
-| 🧪**Inteligentní směrování s ohledem na úkoly** | Automatický výběr modelu podle typu obsahu (kódování/vize/analýza/souhrn) |
-| 🔄**Pracovní postupy agentů A2A** | Deterministický orchestrátor FSM pro stavové spouštění agentů ve více krocích |
-| 🔀**Adaptivní směrování** | Dynamické přepisování strategie založené na objemu tokenů a složitosti výzvy |
-| 🎲**Rozmanitost poskytovatelů** | Shannon entropie bodování vyvažování auto-kombo rozložení provozu |
-| 💬**System Prompt Injection** | Globální ovládací prvky chování používané konzistentně |
-| 📄**Kompatibilita rozhraní Responses API** | Plná podpora `/v1/responses` pro Codex a pokročilé agentní pracovní postupy | ### 🎵 Multi-Modal APIs |
+### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
-| Funkce | Co to dělá |
-| --------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- |
-| 🖼️**Generování obrázků** | `/v1/images/generations` s cloudem a místními backendy |
-| 📐**Vložení** | `/v1/embeddings` pro vyhledávání a potrubí RAG |
-| 🎤**Přepis zvuku** | `/v1/audio/transscriptions` — 7 poskytovatelů (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), automatická detekce jazyka, podpora MP4/MP3/WAV |
-| 🔊**Převod textu na řeč** | `/v1/audio/speech` — 10 poskytovatelů (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) se správnými chybovými zprávami |
-| 🎬**Generace videa** | `/v1/videos/generations` (pracovní postupy ComfyUI + SD WebUI) |
-| 🎵**Music Generation** | `/v1/music/generations` (pracovní postupy ComfyUI) |
-| 🛡️**Moderování** | `/v1/moderations` bezpečnostní kontroly |
-| 🔀**Reranking** | `/v1/rerank` pro hodnocení relevance |
-| 🔍**Vyhledávání na webu**🆕 | `/v1/search` — 5 poskytovatelů (Serper, Brave, Perplexity, Exa, Tavily), 6 500+ zdarma/měsíc, automatické přepnutí při selhání, mezipaměť | ### 🛡️ Resilience, Security & Governance |
+| Feature | What It Does |
+| ------------------------------------ | ------------------------------------------------------------------------------------------- |
+| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) |
+| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family |
+| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 |
+| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models |
+| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content |
+| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data |
+| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges |
+| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins |
-| Funkce | Co to dělá |
-| --------------------------------------- | ----------------------------------------------------------------------------------------------- | -------------------------------- |
-| 🔌**Jističe** | Vypnutí/obnovení pro každý model s ovládáním prahu |
-| 🎯**Koncové modely** | Vlastní modely deklarují podporované koncové body + formát API |
-| 🛡️**Stádo proti hromům** | Mutex + semaforové ochrany při opakování/rychlosti událostí |
-| 🧠**Sémantická + mezipaměť podpisů** | Snížení nákladů/latence se dvěma vrstvami mezipaměti |
-| ⚡**Žádost o idempotenci** | Duplicitní ochranné okno |
-| 🔒**TLS Fingerprint Spoofing** | Otisk TLS jako v prohlížeči —**snižuje detekci robotů a nahlašování účtu** |
-| 🔏**CLI Fingerprint Matching** | Odpovídá nativním podpisům požadavku CLI —**snižuje riziko zákazu při zachování proxy IP** |
-| 🌐**Filtrování IP** | Kontrola seznamu povolených/blokovaných pro vystavená nasazení |
-| 📊**Upravitelné limity sazeb** | Konfigurovatelné globální limity/limity na úrovni poskytovatele s perzistencí |
-| 📉**Půvabná degradace** | Záložní funkce vícevrstvé ochrany chránící operace hlavní brány |
-| 📜**Config Audit Trail** | Sledování změn založené na rozdílech zabraňující provoznímu posunu s jednoduchým vrácením zpět |
-| ⏳**Provider Health Sync** | Proaktivní monitorování vypršení platnosti tokenu spouštějící výstrahy před selháním autorizace |
-| 🚪**Automaticky zakázat zakázané účty** | Provozní jistič automaticky zaplombuje trvale zablokované tokenové účty |
-| 🔑**Správa klíčů API + rozsah** | Bezpečné vydávání/otočení klíčů a ovládání modelu/poskytovatele |
-| 👁️**Scoped API Key Reveal**🆕 | Přihlaste se k obnově klíčů API prostřednictvím `ALLOW_API_KEY_REVEAL` |
-| 🛡️**Chráněno `/modely`** | Volitelné ověřování a skrytí poskytovatele pro katalog modelů | ### 📊 Observability & Analytics |
+### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP
-| Funkce | Co to dělá |
-| -------------------------------------- | ---------------------------------------------------------------------- | ---------------------------- |
-| 📝**Požadavek + protokolování proxy** | Úplný požadavek/odpověď a protokolování proxy |
-| 📉**Streamované podrobné protokoly**🆕 | Čistě rekonstruuje datové proudy SSE do uživatelského rozhraní |
-| 📋**Sjednocený panel protokolů** | Požadavek, proxy, audit a zobrazení konzoly na jedné stránce |
-| 🔍**Požádejte o telemetrii** | p50/p95/p99 latence a sledování požadavků |
-| 🏥**Health Dashboard** | Doba provozuschopnosti, stavy jističe, uzamčení, statistiky mezipaměti |
-| 💰**Sledování nákladů** | Kontroly rozpočtu a viditelnost cen podle modelu |
-| 📈**Analytické vizualizace** | Statistiky využití modelu/poskytovatele a zobrazení trendů |
-| 🧪**Rámec hodnocení** | Testování zlaté sady s konfigurovatelnými strategiemi shody |
-| 📡**Live Diagnostics**🆕 | Sémantické vynechání mezipaměti pro přesné kombinované živé testování | ### ☁️ Deployment & Platform |
+| Feature | What It Does |
+| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing |
+| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** |
+| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint |
+| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. |
+| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator |
+| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID |
+| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart |
-| Funkce | Co to dělá |
-| ----------------------------------------- | ---------------------------------------------------------------------------- | --------------------- |
-| 🌐**Nasadit kdekoli** | Localhost, VPS, Docker, cloudová prostředí |
-| 🚇**Tunel Cloudflare**🆕 | Integrace rychlého tunelu jedním kliknutím z řídicího panelu |
-| 🔑**Filtrování modelu klíče API** | Nativní odpověď /v1/models filtrovaná přes přiřazené kontextové role nosiče |
-| ⚡**Smart Cache Bypass** | Konfigurovatelná heuristika TTL a ovládací prvky nuceného opětovného načtení |
-| 🔄**Zálohování/Obnova** | Export/import a toky obnovy po havárii |
-| 🧙**Průvodce onboardingem** | První spuštění průvodce nastavením |
-| 🔧**CLI Tools Dashboard** | Nastavení jedním kliknutím pro oblíbené kódovací nástroje |
-| 🎮**Modelové hřiště** | Otestujte libovolného poskytovatele/model/koncový bod z řídicího panelu |
-| 🔏**CLI Fingerprint Toggle** | Shoda otisků prstů jednotlivých poskytovatelů v Nastavení > Zabezpečení |
-| 🌐**i18n (30 jazyků)** | Plná podpora řídicího panelu + docs s pokrytím RTL |
-| 🧹**Vymazat všechny modely** | Vymazání seznamu modelů jedním kliknutím v detailech poskytovatele |
-| 👁️**Ovládací prvky postranního panelu**🆕 | Skrýt komponenty a integrace z Nastavení vzhledu |
-| 📋**Šablony vydání** | Standardizované šablony GitHub pro chyby a funkce |
-| 📂**Custom Data Directory** | Přepsání `DATA_DIR` pro umístění úložiště | ### Feature Deep Dive |
+### 🤖 Agent & Protocol Operations (v2.0)
+
+| Feature | What It Does |
+| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
+| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools |
+| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows |
+| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs |
+| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) |
+| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) |
+| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution |
+| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access |
+| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks |
+| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery |
+| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` |
+| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface |
+
+### 🧠 Routing & Intelligence
+
+| Feature | What It Does |
+| ---------------------------------- | ------------------------------------------------------------------------ |
+| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free |
+| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider |
+| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions |
+| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection |
+| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry |
+| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control |
+| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session |
+| 🌐 **Wildcard Router** | `provider/*` dynamic routing |
+| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits |
+| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety |
+| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models |
+| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) |
+| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions |
+| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity |
+| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution |
+| 💬 **System Prompt Injection** | Global behavior controls applied consistently |
+| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows |
+
+### 🎵 Multi-Modal APIs
+
+| Feature | What It Does |
+| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends |
+| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines |
+| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support |
+| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages |
+| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) |
+| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) |
+| 🛡️ **Moderations** | `/v1/moderations` safety checks |
+| 🔀 **Reranking** | `/v1/rerank` for relevance scoring |
+| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache |
+
+### 🛡️ Resilience, Security & Governance
+
+| Feature | What It Does |
+| ----------------------------------- | -------------------------------------------------------------------------------------- |
+| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls |
+| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format |
+| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events |
+| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers |
+| ⚡ **Request Idempotency** | Duplicate protection window |
+| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** |
+| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** |
+| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments |
+| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence |
+| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations |
+| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks |
+| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures |
+| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically |
+| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls |
+| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` |
+| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog |
+
+### 📊 Observability & Analytics
+
+| Feature | What It Does |
+| -------------------------------- | ----------------------------------------------------- |
+| 📝 **Request + Proxy Logging** | Full request/response and proxy logging |
+| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI |
+| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page |
+| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing |
+| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats |
+| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility |
+| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views |
+| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies |
+| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing |
+
+### ☁️ Deployment & Platform
+
+| Feature | What It Does |
+| ------------------------------ | --------------------------------------------------------------------- |
+| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments |
+| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard |
+| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles |
+| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls |
+| 🔄 **Backup/Restore** | Export/import and disaster recovery flows |
+| 🧙 **Onboarding Wizard** | First-run guided setup |
+| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools |
+| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard |
+| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security |
+| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage |
+| 🧹 **Clear All Models** | One-click model list clearing in provider details |
+| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings |
+| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features |
+| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location |
+
+### Feature Deep Dive
#### Smart fallback with practical cost control
@@ -1293,103 +1464,132 @@ Combo: "my-coding-stack"
4. if/kimi-k2-thinking
```
-Když kvóta, rychlost nebo stav selžou, OmniRoute automaticky přejde na dalšího kandidáta bez ručního přepínání.#### Protocol management that is visible and operable
+When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching.
-- MCP + A2A jsou zjistitelné v uživatelském rozhraní a dokumentech (nejsou skryté)
-- Rozhraní API stavu protokolu zpřístupňují živá provozní data (`/api/mcp/*`, `/api/a2a/*`)
-- Panely obsahují akce pro operace 2. dne (přepínání kombinací, resetování jističe, zrušení úkolu)#### Translator + validation workflow
+#### Protocol management that is visible and operable
-Oblast překladatele zahrnuje:
+- MCP + A2A are discoverable in UI and docs (not hidden)
+- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`)
+- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation)
--**Hřiště**: Vyžádejte si kontroly transformace -**Chat Tester**: kompletní zpáteční cesta na žádost/odpověď -**Testovací stolice**: více případů v jednom běhu -**Live Monitor**: zobrazení dopravy v reálném čase
+#### Translator + validation workflow
-Plus ověření protokolu se skutečnými klienty pomocí `npm run test:protocols:e2e`.
+The Translator area includes:
-> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Reference nástrojů, konfigurace IDE a příklady klientů
+- **Playground**: request transformation checks
+- **Chat Tester**: full request/response round-trip
+- **Test Bench**: multiple cases in one run
+- **Live Monitor**: real-time traffic view
+
+Plus protocol validation with real clients via `npm run test:protocols:e2e`.
+
+> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples
>
-> 📖**[A2A Server README](src/lib/a2a/README.md)**— Dovednosti, metody JSON-RPC, streamování a životní cyklus úloh## 🧪 Evaluations (Evals)
+> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle
-OmniRoute obsahuje vestavěný hodnotící rámec pro testování kvality odezvy LLM oproti zlaté sadě. Přistupte k němu přes**Analytics → Evals**na hlavním panelu.### Built-in Golden Set
+## 🧪 Evaluations (Evals)
-Předinstalovaná sada „OmniRoute Golden Set“ obsahuje testovací případy pro:
+OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard.
-- Pozdravy, matematika, zeměpis, generování kódu
-- Kompatibilita formátu JSON, překlad, generování markdown
-- Bezpečnostní odmítnutí (škodlivý obsah), počítání, booleovská logika### Evaluation Strategies
+### Built-in Golden Set
-| Strategie | Popis | Příklad |
-| ----------------- | ---------------------------------------------------------------------- | ------------------------------- | --- |
-| "přesný" | Výstup se musí přesně shodovat | "4"" |
-| "obsahuje" | Výstup musí obsahovat podřetězec (nerozlišují se malá a velká písmena) | "Paříž" |
-| "regulární výraz" | Výstup musí odpovídat vzoru regulárního výrazu | `"1.*2.*3"` |
-| "vlastní" | Vlastní funkce JS vrací true/false | `(výstup) => výstup.délka > 10` | --- |
+The pre-loaded "OmniRoute Golden Set" contains test cases for:
+
+- Greetings, math, geography, code generation
+- JSON format compliance, translation, markdown generation
+- Safety refusal (harmful content), counting, boolean logic
+
+### Evaluation Strategies
+
+| Strategy | Description | Example |
+| ---------- | ------------------------------------------------ | -------------------------------- |
+| `exact` | Output must match exactly | `"4"` |
+| `contains` | Output must contain substring (case-insensitive) | `"Paris"` |
+| `regex` | Output must match regex pattern | `"1.*2.*3"` |
+| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` |
+
+---
## 📖 Setup Guide
### Protocol Setup (MCP + A2A)
-
-🧩 Nastavení MCP (Model Context Protocol)
+
+🧩 MCP Setup (Model Context Protocol)
-Spusťte přenos MCP v režimu stdio:```bash
+Start MCP transport in stdio mode:
+
+```bash
omniroute --mcp
+```
-````
+Recommended validation flow:
-Doporučený postup ověření:
+1. Connect your MCP client over stdio.
+2. Run `omniroute_get_health`.
+3. Run `omniroute_list_combos`.
+4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit.
-1. Připojte svého MCP klienta přes stdio.
-2. Spusťte `omniroute_get_health`.
-3. Spusťte `omniroute_list_combos`.
-4. Otevřete `/dashboard/mcp` pro potvrzení prezenčního signálu, aktivity a auditu.
-
-Užitečná rozhraní API pro automatizaci:
+Useful APIs for automation:
- `GET /api/mcp/status`
- `GET /api/mcp/tools`
- `GET /api/mcp/audit`
-- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats`
-
-🤝 Nastavení A2A (Agent2Agent)
+
-Objevte agenta:```bash
+
+🤝 A2A Setup (Agent2Agent)
+
+Discover the agent:
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
-Odeslat úkol:```bash
+Send a task:
+
+```bash
curl -X POST http://localhost:20128/a2a \
- -H 'content-type: application/json' \
- -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+ -H 'content-type: application/json' \
+ -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+```
-````
-
-Správa životního cyklu:
+Manage lifecycle:
- `GET /api/a2a/status`
- `GET /api/a2a/tasks`
- `GET /api/a2a/tasks/:id`
- `POST /api/a2a/tasks/:id/cancel`
-Provozní uživatelské rozhraní:
+Operational UI:
-- `/dashboard/a2a` pro pozorování úkolu/stavu/toku a akce kouře
+- `/dashboard/a2a` for task/state/stream observability and smoke actions
-
-🧪 End-to-end validace protokolu
+
-Ověřte oba protokoly se skutečnými klienty:```bash
+
+🧪 End-to-end protocol validation
+
+Validate both protocols with real clients:
+
+```bash
npm run test:protocols:e2e
-````
+```
-Tím se ověřuje:
+This verifies:
-- Připojení/seznam/volání klienta MCP SDK
-- A2A objev/odeslat/streamovat/získat/zrušit
-- Křížová kontrola dat v MCP auditu a API pro správu úloh A2A
+- MCP SDK client connect/list/call
+- A2A discovery/send/stream/get/cancel
+- Cross-check data in MCP audit and A2A task management APIs
-
-💳 Poskytovatelé předplatného### Claude Code (Pro/Max)
+
+
+
+💳 Subscription Providers
+
+### Claude Code (Pro/Max)
```bash
Dashboard → Providers → Connect Claude Code
@@ -1402,7 +1602,9 @@ Models:
cc/claude-haiku-4-5-20251001
```
-**Tip pro profesionály:**Používejte Opus pro složité úkoly, Sonnet pro rychlost. OmniRoute sleduje kvótu na model!### OpenAI Codex (Plus/Pro)
+**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model!
+
+### OpenAI Codex (Plus/Pro)
```bash
Dashboard → Providers → Connect Codex
@@ -1416,20 +1618,22 @@ Models:
#### Codex Account Limit Management (5h + Weekly)
-Každý účet Codexu má nyní přepínače zásad v `Dashboard -> Providers`:
+Each Codex account now has policy toggles in `Dashboard -> Providers`:
-- `5h` (ZAP/VYP): vynutit zásadu prahu 5hodinového okna.
-- `Týdně` (ON/OFF): vynutit zásadu týdenního prahu okna.
-- Prahové chování: když povolené okno dosáhne využití >=90 %, daný účet je přeskočen.
-- Rotační chování: OmniRoute automaticky směruje na další způsobilý účet Codex.
-- Resetovat chování: po uplynutí času `resetAt` poskytovatele se účet automaticky znovu stane způsobilým.
+- `5h` (ON/OFF): enforce the 5-hour window threshold policy.
+- `Weekly` (ON/OFF): enforce the weekly window threshold policy.
+- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped.
+- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically.
+- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically.
-Scénáře:
+Scenarios:
-- `5h ON` + `Weekly ON`: účet je přeskočen, když kterékoli okno dosáhne prahové hodnoty.
-- `5h VYP` + `Týdně ZAP`: účet může zablokovat pouze používání týdně.
-- `5h ON` + `Týdenní OFF`: účet může zablokovat pouze 5 hodin používání.
-- `resetAt` prošlo: účet automaticky znovu vstoupí do rotace (bez ručního opětovného povolení).### Gemini CLI (FREE 180K/month!)
+- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold.
+- `5h OFF` + `Weekly ON`: only weekly usage can block the account.
+- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account.
+- `resetAt` passed: account re-enters rotation automatically (no manual re-enable).
+
+### Gemini CLI (FREE 180K/month!)
```bash
Dashboard → Providers → Connect Gemini CLI
@@ -1441,7 +1645,9 @@ Models:
gc/gemini-2.5-pro
```
-**Nejlepší hodnota:**Obrovská bezplatná úroveň! Použijte to před placenými úrovněmi.### GitHub Copilot
+**Best Value:** Huge free tier! Use this before paid tiers.
+
+### GitHub Copilot
```bash
Dashboard → Providers → Connect GitHub
@@ -1456,71 +1662,91 @@ Models:
-
-🔑 Poskytovatelé klíčů API### NVIDIA NIM (FREE developer access — 70+ models)
+
+🔑 API Key Providers
-1. Zaregistrujte se: [build.nvidia.com](https://build.nvidia.com)
-2. Získejte bezplatný klíč API (včetně 1000 kreditů pro odvození)
-3. Ovládací panel → Přidat poskytovatele → NVIDIA NIM:
- - Klíč API: `nvapi-your-key`
+### NVIDIA NIM (FREE developer access — 70+ models)
-**Modely:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` a 50+ dalších
+1. Sign up: [build.nvidia.com](https://build.nvidia.com)
+2. Get free API key (1000 inference credits included)
+3. Dashboard → Add Provider → NVIDIA NIM:
+ - API Key: `nvapi-your-key`
-**Tip pro profesionály:**API kompatibilní s OpenAI – bezproblémově funguje s překladem formátu OmniRoute!### DeepSeek
+**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more
-1. Zaregistrujte se: [platform.deepseek.com](https://platform.deepseek.com)
-2. Získejte API klíč
-3. Ovládací panel → Přidat poskytovatele → DeepSeek
+**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation!
-**Modely:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!)
+### DeepSeek
-1. Zaregistrujte se: [console.groq.com](https://console.groq.com)
-2. Získejte klíč API (včetně bezplatné úrovně)
-3. Ovládací panel → Přidat poskytovatele → Groq
+1. Sign up: [platform.deepseek.com](https://platform.deepseek.com)
+2. Get API key
+3. Dashboard → Add Provider → DeepSeek
-**Modely:**`groq/lama-3.3-70b`, `groq/mixtral-8x7b`
+**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder`
-**Tip pro profesionály:**Ultra rychlé vyvozování – nejlepší pro kódování v reálném čase!### OpenRouter (100+ Models)
+### Groq (Free Tier Available!)
-1. Zaregistrujte se: [openrouter.ai](https://openrouter.ai)
-2. Získejte API klíč
-3. Ovládací panel → Přidat poskytovatele → OpenRouter
+1. Sign up: [console.groq.com](https://console.groq.com)
+2. Get API key (free tier included)
+3. Dashboard → Add Provider → Groq
-**Modely:**Získejte přístup k více než 100 modelům od všech hlavních poskytovatelů prostřednictvím jediného klíče API.
+**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b`
-**Chování řídicího panelu:**Modely OpenRouter jsou spravovány z**Dostupných modelů**. Ruční přidání, import a automatická synchronizace aktualizují stejný seznam.
+**Pro Tip:** Ultra-fast inference — best for real-time coding!
-
-💰 Levní poskytovatelé (záložní)### GLM-4.7 (Daily reset, $0.6/1M)
+### OpenRouter (100+ Models)
-1. Zaregistrujte se: [Zhipu AI](https://open.bigmodel.cn/)
-2. Získejte API klíč z Coding Plan
-3. Ovládací panel → Přidat klíč API:
- - Poskytovatel: `glm`
- - Klíč API: `váš klíč`
+1. Sign up: [openrouter.ai](https://openrouter.ai)
+2. Get API key
+3. Dashboard → Add Provider → OpenRouter
-**Použití:**`glm/glm-4.7`
+**Models:** Access 100+ models from all major providers through a single API key.
-**Tip pro profesionály:**Kódovací plán nabízí 3× kvótu za 1/7 cenu! Resetovat denně v 10:00.### MiniMax M2.1 (5h reset, $0.20/1M)
+**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list.
-1. Zaregistrujte se: [MiniMax](https://www.minimax.io/)
-2. Získejte API klíč
-3. Ovládací panel → Přidat klíč API
+
-**Použití:**`minimax/MiniMax-M2.1`
+
+💰 Cheap Providers (Backup)
-**Tip pro profesionály:**Nejlevnější možnost pro dlouhý kontext (1 milion tokenů)!### Kimi K2 ($9/month flat)
+### GLM-4.7 (Daily reset, $0.6/1M)
-1. Přihlaste se k odběru: [Moonshot AI](https://platform.moonshot.ai/)
-2. Získejte API klíč
-3. Ovládací panel → Přidat klíč API
+1. Sign up: [Zhipu AI](https://open.bigmodel.cn/)
+2. Get API key from Coding Plan
+3. Dashboard → Add API Key:
+ - Provider: `glm`
+ - API Key: `your-key`
-**Použijte:**`kimi/kimi-latest`
+**Use:** `glm/glm-4.7`
-**Tip pro profesionály:**Pevná cena 9 $ měsíčně za 10 milionů tokenů = 0,90 $ / 1 milion efektivních nákladů!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM.
-
-🆓 ZDARMA poskytovatelé (nouzové zálohování)### Qoder (5 FREE models via OAuth)
+### MiniMax M2.1 (5h reset, $0.20/1M)
+
+1. Sign up: [MiniMax](https://www.minimax.io/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `minimax/MiniMax-M2.1`
+
+**Pro Tip:** Cheapest option for long context (1M tokens)!
+
+### Kimi K2 ($9/month flat)
+
+1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `kimi/kimi-latest`
+
+**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost!
+
+
+
+
+🆓 FREE Providers (Emergency Backup)
+
+### Qoder (5 FREE models via OAuth)
```bash
Dashboard → Connect Qoder
@@ -1561,8 +1787,10 @@ Models:
-
-🎨 Vytvořit komba### Example 1: Maximize Subscription → Cheap Backup
+
+🎨 Create Combos
+
+### Example 1: Maximize Subscription → Cheap Backup
```
Dashboard → Combos → Create New
@@ -1590,8 +1818,10 @@ Cost: $0 forever!
-
-🔧 Integrace CLI### Cursor IDE
+
+🔧 CLI Integration
+
+### Cursor IDE
```
Settings → Models → Advanced:
@@ -1602,7 +1832,9 @@ Settings → Models → Advanced:
### Claude Code
-Použijte stránku**CLI Tools**na řídicím panelu pro konfiguraci jedním kliknutím nebo upravte `~/.claude/settings.json` ručně.### Codex CLI
+Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually.
+
+### Codex CLI
```bash
export OPENAI_BASE_URL="http://localhost:20128"
@@ -1613,12 +1845,15 @@ codex "your prompt"
### OpenClaw
-**Možnost 1 – Hlavní panel (doporučeno):**```
+**Option 1 — Dashboard (recommended):**
+
+```
Dashboard → CLI Tools → OpenClaw → Select Model → Apply
+```
-````
+**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`:
-**Možnost 2 — Ručně:**Upravit `~/.openclaw/openclaw.json`:```json
+```json
{
"models": {
"providers": {
@@ -1630,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply
}
}
}
-````
+```
-> **Poznámka:**OpenClaw funguje pouze s místní OmniRoute. Použijte `127.0.0.1` místo `localhost`, abyste se vyhnuli problémům s rozlišením IPv6.### Cline / Continue / RooCode
+> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues.
+
+### Cline / Continue / RooCode
```
Settings → API Configuration:
@@ -1644,15 +1881,17 @@ Settings → API Configuration:
### OpenCode
-**Krok 1:**Přidejte OmniRoute jako vlastního poskytovatele:```bash
+**Step 1:** Add OmniRoute as a custom provider:
+
+```bash
opencode
/connect
-
# Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key
+```
-````
+**Step 2:** Create/edit `opencode.json` in your project root:
-**Krok 2:**Vytvořte/upravte soubor `opencode.json` v kořenovém adresáři projektu:```json
+```json
{
"$schema": "https://opencode.ai/config.json",
"provider": {
@@ -1670,118 +1909,130 @@ opencode
}
}
}
-````
+```
-**Krok 3:**Vyberte model v OpenCode:```bash
+**Step 3:** Select the model in OpenCode:
+
+```bash
/models
-
# Select any OmniRoute model from the list
+```
-````
+> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard.
->**Tip:**Přidejte jakýkoli model dostupný v koncovém bodu vašeho OmniRoute `/v1/models` do sekce `models`. Použijte formát `provider/model-id` z řídicího panelu OmniRoute.
+
---
## Řešení problémů
-
-Kliknutím rozbalíte průvodce odstraňováním problémů
+
+Click to expand troubleshooting guide
-**"Jazykový model neposkytoval zprávy"**
+**"Language model did not provide messages"**
-- Kvóta poskytovatele je vyčerpána → Zkontrolujte sledování kvót na řídicím panelu
-- Řešení: Použijte nouzovou kombinaci nebo přejděte na levnější úroveň
+- Provider quota exhausted → Check dashboard quota tracker
+- Solution: Use combo fallback or switch to cheaper tier
-**Omezení sazby**
+**Rate limiting**
-- Vyčerpaná kvóta předplatného → Záložní režim GLM/MiniMax
-- Přidejte kombinaci: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Subscription quota out → Fallback to GLM/MiniMax
+- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-**Platnost tokenu OAuth vypršela**
+**OAuth token expired**
-- Automaticky obnovováno OmniRoute
-- Pokud problémy přetrvávají: Řídicí panel → Poskytovatel → Znovu připojit
+- Auto-refreshed by OmniRoute
+- If issues persist: Dashboard → Provider → Reconnect
-**Vysoké náklady**
+**High costs**
-- Zkontrolujte statistiky využití v Dashboard → Náklady
-- Přepněte primární model na GLM/MiniMax
-- Používejte bezplatnou vrstvu (Gemini CLI, Qoder) pro nekritické úkoly
+- Check usage stats in Dashboard → Costs
+- Switch primary model to GLM/MiniMax
+- Use free tier (Gemini CLI, Qoder) for non-critical tasks
-**Porty řídicího panelu/API jsou chybné**
+**Dashboard/API ports are wrong**
-- `PORT` je kanonický základní port (a port API ve výchozím nastavení)
-- `API_PORT` přepíše pouze posluchače API kompatibilní s OpenAI
-- `DASHBOARD_PORT` přepíše pouze posluchače dashboard/Next.js
-– Nastavte „NEXT_PUBLIC_BASE_URL“ na svůj řídicí panel/veřejnou adresu URL (pro zpětná volání OAuth)
+- `PORT` is the canonical base port (and API port by default)
+- `API_PORT` overrides only OpenAI-compatible API listener
+- `DASHBOARD_PORT` overrides only dashboard/Next.js listener
+- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks)
-**Chyby synchronizace cloudu**
+**Cloud sync errors**
-- Ověřte, že `BASE_URL` odkazuje na vaši spuštěnou instanci
-- Ověřte, že `CLOUD_URL` odkazuje na očekávaný koncový bod cloudu
-- Udržujte hodnoty `NEXT_PUBLIC_*` zarovnané s hodnotami na straně serveru
+- Verify `BASE_URL` points to your running instance
+- Verify `CLOUD_URL` points to your expected cloud endpoint
+- Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**První přihlášení nefunguje**
+**First login not working**
-- Zkontrolujte `INITIAL_PASSWORD` v `.env`
-- Pokud není nastaveno, záložní heslo je `123456`
+- Check `INITIAL_PASSWORD` in `.env`
+- If unset, fallback password is `123456`
-**Žádné protokoly požadavků**
+**No request logs**
-- Artefakty požadavku se zapisují do `DATA_DIR/call_logs/` jako jeden soubor JSON na požadavek
-- Povolte zachycení potrubí z řídicího panelu → Protokoly → Protokoly žádostí, pokud potřebujete podrobné užitečné zatížení pro jednotlivé fáze
-- Nastavte `APP_LOG_TO_FILE=true`, pokud chcete také protokoly konzoly aplikace v `logs/application/app.log`
-– Podle potřeby upravte `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` a `CALL_LOG_MAX_ENTRIES`
+- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request
+- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads
+- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log`
+- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed
-**Test připojení ukazuje „Neplatné“ pro poskytovatele kompatibilní s OpenAI**
+**Connection test shows "Invalid" for OpenAI-compatible providers**
-- Mnoho poskytovatelů nevystavuje koncový bod `/models`
-- OmniRoute v1.0.6+ zahrnuje nouzové ověření prostřednictvím dokončení chatu
-- Zajistěte, aby základní adresa URL obsahovala příponu `/v1`### 🔐 OAuth on a Remote Server
+- Many providers don't expose a `/models` endpoint
+- OmniRoute v1.0.6+ includes fallback validation via chat completions
+- Ensure base URL includes `/v1` suffix
+
+### 🔐 OAuth on a Remote Server
->**⚠️ Důležité pro uživatele provozující OmniRoute na VPS, Dockeru nebo jakémkoli vzdáleném serveru**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
+> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server**
-Poskytovatelé**Antigravity**a**Gemini CLI**používají**Google OAuth 2.0**. Google vyžaduje, aby parametr `redirect_uri` v toku OAuth přesně odpovídal jednomu z předem registrovaných URI v Google Cloud Console aplikace.
+#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
-Přihlašovací údaje OAuth dodávané v OmniRoute jsou registrovány**pouze pro `localhost`**. Když přistupujete k OmniRoute na vzdáleném serveru (např. `https://omniroute.myserver.com`), Google odmítne ověření pomocí:```
+The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console.
+
+The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solution: Configure your own OAuth credentials
-Ve službě Google Cloud Console musíte vytvořit**OAuth 2.0 Client ID**s identifikátorem URI vašeho serveru.#### Step-by-step
+You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI.
-**1. Otevřít Google Cloud Console**
+#### Step-by-step
-Přejděte na: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
+**1. Open Google Cloud Console**
-**2. Vytvořit nové ID klienta OAuth 2.0**
+Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-– Klikněte na**"+ Vytvořit přihlašovací údaje"**→**"ID klienta OAuth"**
+**2. Create a new OAuth 2.0 Client ID**
-- Typ aplikace:**"Webová aplikace"**
-- Název: cokoliv se vám líbí (např. `OmniRoute Remote`)
+- Click **"+ Create Credentials"** → **"OAuth client ID"**
+- Application type: **"Web application"**
+- Name: anything you like (e.g. `OmniRoute Remote`)
-**3. Přidat identifikátory URI autorizovaného přesměrování**
+**3. Add Authorized Redirect URIs**
-Do pole**"URI autorizovaného přesměrování"**přidejte:```
+In the **"Authorized redirect URIs"** field, add:
+
+```
https://your-server.com/callback
+```
-````
+> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`).
-> Nahraďte `vas-server.com` doménou nebo IP svého serveru (v případě potřeby uveďte port, např. `http://45.33.32.156:20128/callback`).
+**4. Save and copy the credentials**
-**4. Uložte a zkopírujte přihlašovací údaje**
+After creating, Google will show the **Client ID** and **Client Secret**.
-Po vytvoření Google zobrazí**Client ID**a**Client Secret**.
+**5. Set environment variables**
-**5. Nastavit proměnné prostředí**
+In your `.env` (or Docker environment variables):
-Ve vašem `.env` (nebo proměnných prostředí Docker):```bash
+```bash
# For Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
@@ -1790,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
-````
+```
-**6. Restartujte OmniRoute**```bash
+**6. Restart OmniRoute**
+```bash
# npm:
-
npm run dev
# Docker:
-
docker restart omniroute
+```
-````
+**7. Try connecting again**
-**7. Zkuste se připojit znovu**
+Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth
-Ovládací panel → Poskytovatelé → Antigravitace (nebo Gemini CLI) → OAuth
+Google will now redirect correctly to `https://your-server.com/callback`.
-Google se nyní správně přesměruje na `https://vas-server.com/callback`.---
+---
#### Temporary workaround (without custom credentials)
-Pokud si nyní nechcete nastavovat vlastní přihlašovací údaje, můžete stále použít**ruční postup URL**:
+If you don't want to set up your own credentials right now, you can still use the **manual URL flow**:
-1. OmniRoute otevře autorizační URL Google
-2. Po autorizaci se Google pokusí přesměrovat na `localhost` (který selže na vzdáleném serveru)
-3.**Zkopírujte celou adresu URL**z adresního řádku prohlížeče (i když se stránka nenačte)
-4. Vložte tuto adresu URL do pole zobrazeného v modálu připojení OmniRoute
-5. Klikněte na**"Připojit"**
+1. OmniRoute opens the Google authorization URL
+2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server)
+3. **Copy the full URL** from your browser's address bar (even if the page doesn't load)
+4. Paste that URL into the field shown in the OmniRoute connection modal
+5. Click **"Connect"**
-> Funguje to, protože autorizační kód v adrese URL je platný bez ohledu na to, zda se stránka přesměrování načetla.---
+> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded.
-
-🇧🇷 Versão em Português
#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+---
-Osvedčuje**Antigravity**a**Gemini CLI**používáme**Google OAuth 2.0**pro autenticitu. O Google exige que a `redirect_uri` usada no fluxo OAuth seja**exatamente**uma das URIs pré-cadastradas no Google Cloud Console to use.
+
+🇧🇷 Versão em Português
-Jako credenciais OAuth embutidas no OmniRoute estão cadastradas**apenas para `localhost`**. Quando você acessa o OmniRoute um um servidor remote (ex: `https://omniroute.meuservidor.com`), nebo Google rejeita a autenticação com:```
+#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+
+Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo.
+
+As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solução: Configure suas próprias credenciais OAuth
-Você precisa criar um**OAuth 2.0 Client ID**no Google Cloud Console com a URI do seu server.#### Passo a passo
+Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor.
-**1. Přístup ke službě Google Cloud Console**
+#### Passo a passo
+
+**1. Acesse o Google Cloud Console**
Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
**2. Crie um novo OAuth 2.0 Client ID**
-- Klikněte na**"+ Vytvořit přihlašovací údaje"**→**"ID klienta OAuth"**
-- Tipo de aplicativo:**"Webová aplikace"**
-- Nome: escolha qualquer nome (např.: `OmniRoute Remote`)
+- Clique em **"+ Create Credentials"** → **"OAuth client ID"**
+- Tipo de aplicativo: **"Web application"**
+- Nome: escolha qualquer nome (ex: `OmniRoute Remote`)
-**3. Adicione jako Authorized Redirect URI**
+**3. Adicione as Authorized Redirect URIs**
-Žádné pole**"URI autorizovaného přesměrování"**, adicione:```
+No campo **"Authorized redirect URIs"**, adicione:
+
+```
https://seu-servidor.com/callback
+```
-````
+> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`).
-> Substitua `seu-servidor.com` pelo domínio nebo IP do seu servidor (včetně portu se necessário, např.: `http://45.33.32.156:20128/callback`).
+**4. Salve e copie as credenciais**
-**4. Uložit a zkopírovat jako credenciais**
+Após criar, o Google mostrará o **Client ID** e o **Client Secret**.
-Após criar, o Google mostrará o**Client ID**e o**Client Secret**.
+**5. Configure as variáveis de ambiente**
-**5. Konfigurovat jako variáveis de ambiente**
+No seu `.env` (ou nas variáveis de ambiente do Docker):
-No seu `.env` (ou nas variáveis de ambiente do Docker):```bash
+```bash
# Para Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
@@ -1869,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
-````
+```
-**6. Reinicie nebo OmniRoute**```bash
+**6. Reinicie o OmniRoute**
+```bash
# Se usando npm:
-
npm run dev
# Se usando Docker:
-
docker restart omniroute
-
-````
+```
**7. Tente conectar novamente**
-Dashboard → Poskytovatelé → Antigravitace (nebo Gemini CLI) → OAuth
+Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth
-Agora nebo Google redirecionamente corretamente para `https://seu-servidor.com/callback` a autenticação funcionará.---
+Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará.
+
+---
#### Workaround temporário (sem configurar credenciais próprias)
-Zjistěte, jaké jsou údaje o vaší kreditní kartě, a je možné, že použijete fluxo**příručku URL**:
+Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**:
-1. O OmniRoute abrirá a URL autorização Google
+1. O OmniRoute abrirá a URL de autorização do Google
2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto)
-3.**Zkopírujte úplnou adresu URL**da barra de endereço do seu browser (mesmo que a pagina não carregue)
+3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue)
4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute
-5. Klikněte na**"Připojit"**
+5. Clique em **"Connect"**
-> Toto řešení funguje pomocí autorizačního kódu na URL a nezávislého přesměrování.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+
+
---
@@ -1907,64 +2171,73 @@ Zjistěte, jaké jsou údaje o vaší kreditní kartě, a je možné, že použi
## 🛠️ Tech Stack
-
-Kliknutím rozbalíte podrobnosti o technologickém zásobníku
+
+Click to expand tech stack details
--**Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ není**podporován**– nativní binární soubory `better-sqlite3` jsou nekompatibilní)
--**Jazyk**: TypeScript 5.9 —**100% TypeScript**napříč `src/` a `open-sse/` (nula `any` v základních modulech od verze 2.0)
--**Framework**: Next.js 16 + React 19 + Tailwind CSS 4
--**Databáze**: LowDB (JSON) + SQLite (stav domény + protokoly proxy + audit MCP + rozhodnutí o směrování)
--**Schémata**: Zod (ověření I/O nástroje MCP, smlouvy API)
--**Protokoly**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
--**Streamování**: Server-Sent Events (SSE)
--**Auth**: OAuth 2.0 (PKCE) + JWT + API klíče + MCP Scoped Authorization
--**Testování**: Testovací program Node.js + Vitest (více než 900 testů včetně jednotky, integrace, E2E)
--**CI/CD**: Akce GitHub (automatické publikování npm + Docker Hub při vydání)
--**Web**: [omniroute.online](https://omniroute.online)
--**Balík**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
--**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
--**Odolnost**: Jistič, exponenciální ústup, stádo proti hromům, TLS spoofing, auto-kombo samoléčení
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible)
+- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0)
+- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4
+- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions)
+- **Schemas**: Zod (MCP tool I/O validation, API contracts)
+- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
+- **Streaming**: Server-Sent Events (SSE)
+- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization
+- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E)
+- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release)
+- **Website**: [omniroute.online](https://omniroute.online)
+- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
+- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
+- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing
+
+
---
## Dokumentace
-| Dokument | Popis |
-| ---------------------------------------------- | ---------------------------------------------------- |
-| [Uživatelská příručka](docs/USER_GUIDE.md) | Poskytovatelé, komba, integrace CLI, nasazení |
-| [Reference API](docs/API_REFERENCE.md) | Všechny koncové body s příklady |
-| [Server MCP](open-sse/mcp-server/README.md) | 16 MCP nástroje, konfigurace IDE, klienti Python/TS/Go |
-| [Server A2A](src/lib/a2a/README.md) | Protokol JSON-RPC 2.0, dovednosti, streamování, správa úloh |
-| [Auto-Combo Engine](docs/auto-combo.md) | 6faktorové bodování, balíčky režimů, samoléčení |
-| [Odstraňování problémů](docs/TROUBLESHOOTING.md) | Běžné problémy a řešení |
-| [Architektura](docs/ARCHITECTURE.md) | Architektura systému a vnitřní části |
-| [Přispívá](CONTRIBUTING.md) | Vývojové nastavení a pokyny |
-| [Specifikace OpenAPI](docs/openapi.yaml) | Specifikace OpenAPI 3.0 |
-| [Bezpečnostní zásady](SECURITY.md) | Hlášení zranitelnosti a bezpečnostní postupy |
-| [Deployment VM](docs/VM_DEPLOYMENT_GUIDE.md) | Kompletní průvodce: Nastavení VM + nginx + Cloudflare |
-| [Galerie funkcí](docs/FEATURES.md) | Vizuální prohlídka řídicího panelu se snímky obrazovky |
-| [Kontrolní seznam vydání](docs/RELEASE_CHECKLIST.md) | Kroky ověření před vydáním |---
+| Document | Description |
+| ---------------------------------------------- | --------------------------------------------------- |
+| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment |
+| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples |
+| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients |
+| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt |
+| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing |
+| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation |
+| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions |
+| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals |
+| [Contributing](CONTRIBUTING.md) | Development setup and guidelines |
+| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification |
+| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices |
+| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup |
+| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots |
+| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps |
+
+---
## 🗺️ Roadmap
-OmniRoute má naplánováno**210+ funkcí**v několika fázích vývoje. Zde jsou klíčové oblasti:
+OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas:
-| Kategorie | Plánované funkce | Hlavní body |
-| ------------------------------ | ----------------- | -------------------------------------------------------------------------------------- |
-| 🧠**Směrování a inteligence**| 25+ | Směrování s nejnižší latencí, směrování založené na značkách, předběžná kontrola kvót, výběr účtu P2C |
-| 🔒**Zabezpečení a dodržování předpisů**| 20+ | Zpevnění SSRF, maskování pověření, rychlostní limit na koncový bod, stanovení rozsahu klíče managementu |
-| 📊**Pozorovatelnost**| 15+ | Integrace OpenTelemetry, sledování kvót v reálném čase, sledování nákladů na model |
-| 🔄**Integrace poskytovatelů**| 20+ | Registr dynamického modelu, cooldowny poskytovatelů, kodex pro více účtů, analýza kvót Copilota |
-| ⚡**Výkon**| 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API |
-| 🌐**Ekosystém**| 10+ | WebSocket API, konfigurace hot-reload, distribuované úložiště konfigurace, komerční režim |### 🔜 Coming Soon
+| Category | Planned Features | Highlights |
+| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- |
+| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection |
+| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping |
+| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model |
+| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing |
+| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API |
+| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode |
-- 🔗**Integrace OpenCode**– Podpora nativního poskytovatele pro IDE kódování OpenCode AI
-- 🔗**Integrace TRAE**— Plná podpora pro vývojový rámec TRAE AI
-- 📦**Batch API**— Asynchronní dávkové zpracování pro hromadné požadavky
-- 🎯**Směrování založené na značkách**– Směrování požadavků na základě vlastních značek a metadat
-- 💰**Strategie nejnižších nákladů**— Automaticky vyberte nejlevnějšího dostupného poskytovatele
+### 🔜 Coming Soon
-> 📝 Úplné specifikace funkcí dostupné v [`docs/new-features/`](docs/new-features/) (217 podrobných specifikací)---
+- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE
+- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework
+- 📦 **Batch API** — Asynchronous batch processing for bulk requests
+- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata
+- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider
+
+> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs)
+
+---
## 👥 Contributors
@@ -1972,18 +2245,20 @@ OmniRoute má naplánováno**210+ funkcí**v několika fázích vývoje. Zde jso
### How to Contribute
-1. Rozdělte úložiště
-2. Vytvořte si větev funkcí (`git checkout -b feature/amazing-feature`)
-3. Potvrďte změny (`git commit -m 'Přidat úžasnou funkci'`)
-4. Push do větve (`git Push origin feature/amazing-feature`)
-5. Otevřete žádost o stažení
+1. Fork the repository
+2. Create your feature branch (`git checkout -b feature/amazing-feature`)
+3. Commit your changes (`git commit -m 'Add amazing feature'`)
+4. Push to the branch (`git push origin feature/amazing-feature`)
+5. Open a Pull Request
-Podrobné pokyny najdete na [CONTRIBUTING.md](CONTRIBUTING.md).### Releasing a New Version
+See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines.
+
+### Releasing a New Version
```bash
# Create a release — npm publish happens automatically
gh release create v2.0.0 --title "v2.0.0" --generate-notes
-````
+```
---
@@ -1995,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes
## 🙏 Acknowledgments
-Zvláštní poděkování patří**[9router](https://github.com/decolua/9router)**od**[decolua](https://github.com/decolua)**— původnímu projektu, který inspiroval tento fork. OmniRoute staví na tomto neuvěřitelném základu s dalšími funkcemi, multimodálními API a úplným přepsáním TypeScriptu.
+Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite.
-Zvláštní poděkování patří**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**– původní implementaci Go, která inspirovala tento port JavaScriptu.---
+Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port.
+
+---
## Licence
-Licence MIT – podrobnosti viz [LICENCE](LICENCE).---
+MIT License - see [LICENSE](LICENSE) for details.
+
+---
Built with ❤️ for developers who code 24/7
diff --git a/docs/i18n/cs/docs/ARCHITECTURE.md b/docs/i18n/cs/docs/ARCHITECTURE.md
index e6d8fdc60c..587e9fe20c 100644
--- a/docs/i18n/cs/docs/ARCHITECTURE.md
+++ b/docs/i18n/cs/docs/ARCHITECTURE.md
@@ -4,80 +4,93 @@
---
-_Poslední aktualizace: 28.03.2026_## Executive Summary
-OmniRoute je místní AI směrovací brána a řídicí panel postavený na Next.js.
-Poskytuje jeden koncový bod kompatibilní s OpenAI (`/v1/*`) a směruje provoz přes několik upstreamových poskytovatelů s překladem, nouzovým obnovením, obnovením tokenu a sledováním využití.
-Základní schopnosti:
+_Last updated: 2026-03-28_
-- OpenAI kompatibilní povrch API pro CLI/nástroje (28 poskytovatelů)
-- Překlad požadavku/odpovědi mezi formáty poskytovatelů
-- Záložní kombinace modelů (sekvence více modelů)
- – Záloha na úrovni účtu (více účtů na poskytovatele)
-- Správa připojení poskytovatele OAuth + API klíče
-- Generování vkládání pomocí `/v1/embeddings` (6 poskytovatelů, 9 modelů)
-- Generování obrázků pomocí `/v1/images/generations` (4 poskytovatelé, 9 modelů)
-- Myslete na analýzu značek (`
...`) pro modely uvažování
-- Dezinfekce odezvy pro přísnou kompatibilitu OpenAI SDK
-- Normalizace rolí (vývojář→systém, systém→uživatel) pro kompatibilitu mezi poskytovateli
-- Konverze strukturovaného výstupu (json_schema → Gemini responseSchema)
-- Místní perzistence pro poskytovatele, klíče, aliasy, komba, nastavení, ceny
-- Sledování využití/nákladů a protokolování požadavků
-- Volitelná cloudová synchronizace pro synchronizaci mezi více zařízeními/stavy
-- Seznam povolených/blokovaných IP pro řízení přístupu k API
-- Myslet na správu rozpočtu (průchozí/automatické/vlastní/adaptivní)
-- Okamžité vstřikování globálního systému
-- Sledování relací a snímání otisků prstů
-- Rozšířené omezení sazeb na účet pomocí profilů specifických pro poskytovatele
-- Vzor jističe pro odolnost poskytovatele
-- Ochrana stáda proti hromu s mutexovým zamykáním
-- Mezipaměť deduplikace požadavků na základě podpisu
-- Doménová vrstva: dostupnost modelu, nákladová pravidla, záložní politika, politika uzamčení
-- Perzistence stavu domény (mezipaměť pro zápis SQLite pro záložní, rozpočty, uzamčení, jističe)
-- Modul zásad pro centralizované vyhodnocování požadavků (uzamčení → rozpočet → záložní)
-- Vyžádejte si telemetrii s agregací latence p50/p95/p99
-- ID korelace (X-Request-Id) pro end-to-end trasování
-- Protokolování auditu shody s odhlášením podle klíče API
-- Eval rámec pro zajištění kvality LLM
-- Řídicí panel Resilience UI se stavem jističe v reálném čase
-- Modulární poskytovatelé OAuth (12 jednotlivých modulů pod `src/lib/oauth/providers/`)
+## Executive Summary
-Primární runtime model:
+OmniRoute is a local AI routing gateway and dashboard built on Next.js.
+It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking.
-- Trasy aplikací Next.js pod `src/app/api/*` implementují jak rozhraní API řídicího panelu, tak rozhraní API pro kompatibilitu
-- Sdílené jádro SSE/směrování v `src/sse/*` + `open-sse/*` se stará o provádění poskytovatele, překlad, streamování, zálohování a používání## Scope and Boundaries
+Core capabilities:
+
+- OpenAI-compatible API surface for CLI/tools (28 providers)
+- Request/response translation across provider formats
+- Model combo fallback (multi-model sequence)
+- Account-level fallback (multi-account per provider)
+- OAuth + API-key provider connection management
+- Embedding generation via `/v1/embeddings` (6 providers, 9 models)
+- Image generation via `/v1/images/generations` (4 providers, 9 models)
+- Think tag parsing (`
...`) for reasoning models
+- Response sanitization for strict OpenAI SDK compatibility
+- Role normalization (developer→system, system→user) for cross-provider compatibility
+- Structured output conversion (json_schema → Gemini responseSchema)
+- Local persistence for providers, keys, aliases, combos, settings, pricing
+- Usage/cost tracking and request logging
+- Optional cloud sync for multi-device/state sync
+- IP allowlist/blocklist for API access control
+- Thinking budget management (passthrough/auto/custom/adaptive)
+- Global system prompt injection
+- Session tracking and fingerprinting
+- Per-account enhanced rate limiting with provider-specific profiles
+- Circuit breaker pattern for provider resilience
+- Anti-thundering herd protection with mutex locking
+- Signature-based request deduplication cache
+- Domain layer: model availability, cost rules, fallback policy, lockout policy
+- Context Relay: session handoff summaries for account rotation continuity
+- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers)
+- Policy engine for centralized request evaluation (lockout → budget → fallback)
+- Request telemetry with p50/p95/p99 latency aggregation
+- Correlation ID (X-Request-Id) for end-to-end tracing
+- Compliance audit logging with opt-out per API key
+- Eval framework for LLM quality assurance
+- Resilience UI dashboard with real-time circuit breaker status
+- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`)
+
+Primary runtime model:
+
+- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs
+- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage
+
+## Scope and Boundaries
### In Scope
-- Runtime místní brány
-- Rozhraní API pro správu řídicích panelů
-- Ověření poskytovatele a obnovení tokenu
-- Vyžádejte si překlad a streamování SSE
-- Místní stav + perzistence používání
-- Volitelná orchestrace synchronizace s cloudem### Out of Scope
+- Local gateway runtime
+- Dashboard management APIs
+- Provider authentication and token refresh
+- Request translation and SSE streaming
+- Local state + usage persistence
+- Optional cloud sync orchestration
-- Implementace cloudové služby za `NEXT_PUBLIC_CLOUD_URL`
-- Poskytovatel SLA/řídící rovina mimo místní proces
-- Samotné externí binární soubory CLI (Claude CLI, Codex CLI atd.)## Dashboard Surface (Current)
+### Out of Scope
-Hlavní stránky pod `src/app/(dashboard)/dashboard/`:
+- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL`
+- Provider SLA/control plane outside local process
+- External CLI binaries themselves (Claude CLI, Codex CLI, etc.)
-- `/dashboard` — rychlý start + přehled poskytovatele
-- `/dashboard/endpoint` — proxy koncového bodu + MCP + A2A + karty koncového bodu API
-- `/dashboard/providers` — připojení a přihlašovací údaje poskytovatele
-- `/dashboard/combos` — kombo strategie, šablony, pravidla směrování modelů
-- `/dashboard/costs` — agregace nákladů a viditelnost cen
-- `/dashboard/analytics` — analýzy a vyhodnocení využití
-- `/dashboard/limits` — kontroly kvót/sazeb
-- `/dashboard/cli-tools` — CLI onboarding, runtime detekce, generování konfigurace
-- `/dashboard/agents` — detekovaní agenti AKT + vlastní registrace agenta
-- `/dashboard/media` — hřiště pro obrázky/video/hudbu
-- `/dashboard/search-tools` — testování a historie poskytovatelů vyhledávání
-- `/dashboard/health` — doba provozuschopnosti, jističe, limity sazeb
-- `/dashboard/logs` — protokoly požadavku/proxy/audit/konzole
-- `/dashboard/settings` — karty nastavení systému (obecné, směrování, výchozí kombinace atd.)
-- `/dashboard/api-manager` — životní cyklus klíče API a oprávnění k modelu## High-Level System Context
+## Dashboard Surface (Current)
+
+Main pages under `src/app/(dashboard)/dashboard/`:
+
+- `/dashboard` — quick start + provider overview
+- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs
+- `/dashboard/providers` — provider connections and credentials
+- `/dashboard/combos` — combo strategies, templates, model routing rules
+- `/dashboard/costs` — cost aggregation and pricing visibility
+- `/dashboard/analytics` — usage analytics and evaluations
+- `/dashboard/limits` — quota/rate controls
+- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation
+- `/dashboard/agents` — detected ACP agents + custom agent registration
+- `/dashboard/media` — image/video/music playground
+- `/dashboard/search-tools` — search provider testing and history
+- `/dashboard/health` — uptime, circuit breakers, rate limits
+- `/dashboard/logs` — request/proxy/audit/console logs
+- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.)
+- `/dashboard/api-manager` — API key lifecycle and model permissions
+
+## High-Level System Context
```mermaid
flowchart LR
@@ -129,139 +142,151 @@ flowchart LR
## 1) API and Routing Layer (Next.js App Routes)
-Hlavní adresáře:
+Main directories:
-- `src/app/api/v1/*` a `src/app/api/v1beta/*` pro rozhraní API pro kompatibilitu
-- `src/app/api/*` pro správu/konfiguraci API
-- Další přepíše mapu `next.config.mjs` `/v1/*` na `/api/v1/*`
+- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs
+- `src/app/api/*` for management/configuration APIs
+- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*`
-Důležité cesty kompatibility:
+Important compatibility routes:
- `src/app/api/v1/chat/completions/route.ts`
- `src/app/api/v1/messages/route.ts`
- `src/app/api/v1/responses/route.ts`
-- `src/app/api/v1/models/route.ts` — zahrnuje vlastní modely s `custom: true`
-- `src/app/api/v1/embeddings/route.ts` — generování vložení (6 poskytovatelů)
-- `src/app/api/v1/images/generations/route.ts` — generování obrázků (4+ poskytovatelé včetně Antigravity/Nebius)
+- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true`
+- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers)
+- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius)
- `src/app/api/v1/messages/count_tokens/route.ts`
- – `src/app/api/v1/providers/[poskytovatel]/chat/completions/route.ts` – vyhrazený chat pro jednotlivé poskytovatele
- – `src/app/api/v1/providers/[poskytovatel]/embeddings/route.ts` – vyhrazená vložení pro jednotlivé poskytovatele
-- `src/app/api/v1/providers/[poskytovatel]/images/generations/route.ts` – vyhrazené obrázky pro jednotlivé poskytovatele
+- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat
+- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings
+- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images
- `src/app/api/v1beta/models/route.ts`
-- `src/app/api/v1beta/models/[...cesta]/route.ts`
+- `src/app/api/v1beta/models/[...path]/route.ts`
-Domény správy:
+Management domains:
- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*`
-- Poskytovatelé/připojení: `src/app/api/providers*`
-- Uzly poskytovatele: `src/app/api/provider-nodes*`
-- Vlastní modely: `src/app/api/provider-models` (GET/POST/DELETE)
-- Katalog modelů: `src/app/api/models/route.ts` (GET)
-- Konfigurace proxy: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
+- Providers/connections: `src/app/api/providers*`
+- Provider nodes: `src/app/api/provider-nodes*`
+- Custom models: `src/app/api/provider-models` (GET/POST/DELETE)
+- Model catalog: `src/app/api/models/route.ts` (GET)
+- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
- OAuth: `src/app/api/oauth/*`
-- Klíče/aliasy/komba/cena: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing`
-- Použití: `src/app/api/usage/*`
+- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing`
+- Usage: `src/app/api/usage/*`
- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*`
-- Pomocníci nástrojů CLI: `src/app/api/cli-tools/*`
-- IP filtr: `src/app/api/settings/ip-filter` (GET/PUT)
+- CLI tooling helpers: `src/app/api/cli-tools/*`
+- IP filter: `src/app/api/settings/ip-filter` (GET/PUT)
- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT)
-- Systémová výzva: `src/app/api/settings/system-prompt` (GET/PUT)
-- Relace: `src/app/api/sessions` (GET)
-- Sazbové limity: `src/app/api/rate-limits` (GET)
-- Odolnost: `src/app/api/resilience` (GET/PATCH) — profily poskytovatelů, jistič, stav omezení rychlosti
-- Resetování odolnosti: `src/app/api/resilience/reset` (POST) — resetujte jističe + cooldowny
-- Statistiky mezipaměti: `src/app/api/cache/stats` (GET/DELETE)
-- Dostupnost modelu: `src/app/api/models/availability` (GET/POST)
-- Telemetrie: `src/app/api/telemetry/summary` (GET)
- – Rozpočet: `src/app/api/usage/budget` (GET/POST)
-- Záložní řetězce: `src/app/api/fallback/chains` (GET/POST/DELETE)
-- Audit souladu: `src/app/api/compliance/audit-log` (GET)
-- Hodnoty: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET)
-- Zásady: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core
+- System prompt: `src/app/api/settings/system-prompt` (GET/PUT)
+- Sessions: `src/app/api/sessions` (GET)
+- Rate limits: `src/app/api/rate-limits` (GET)
+- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state
+- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns
+- Cache stats: `src/app/api/cache/stats` (GET/DELETE)
+- Model availability: `src/app/api/models/availability` (GET/POST)
+- Telemetry: `src/app/api/telemetry/summary` (GET)
+- Budget: `src/app/api/usage/budget` (GET/POST)
+- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE)
+- Compliance audit: `src/app/api/compliance/audit-log` (GET)
+- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET)
+- Policies: `src/app/api/policies` (GET/POST)
-Hlavní průtokové moduly:
+## 2) SSE + Translation Core
-- Záznam: `src/sse/handlers/chat.ts`
-- Základní orchestrace: `open-sse/handlers/chatCore.ts`
-- Spouštěcí adaptéry poskytovatele: `open-sse/executors/*`
-- Detekce formátu/konfigurace poskytovatele: `open-sse/services/provider.ts`
-- Parse/resolve modelu: `src/sse/services/model.ts`, `open-sse/services/model.ts`
-- Logika záložního účtu: `open-sse/services/accountFallback.ts`
-- Registr překladů: `open-sse/translator/index.ts`
-- Transformace streamu: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
-- Extrakce/normalizace použití: `open-sse/utils/usageTracking.ts`
-- Analyzátor značek Think: `open-sse/utils/thinkTagParser.ts`
-- Obsluha vkládání: `open-sse/handlers/embeddings.ts`
-- Registr poskytovatele vkládání: `open-sse/config/embeddingRegistry.ts`
-- Ovladač generování obrázků: `open-sse/handlers/imageGeneration.ts`
-- Registr poskytovatele obrázků: `open-sse/config/imageRegistry.ts`
-- Dezinfekce odezvy: `open-sse/handlers/responseSanitizer.ts`
-- Normalizace rolí: `open-sse/services/roleNormalizer.ts`
+Main flow modules:
-Služby (obchodní logika):
+- Entry: `src/sse/handlers/chat.ts`
+- Core orchestration: `open-sse/handlers/chatCore.ts`
+- Provider execution adapters: `open-sse/executors/*`
+- Format detection/provider config: `open-sse/services/provider.ts`
+- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts`
+- Account fallback logic: `open-sse/services/accountFallback.ts`
+- Translation registry: `open-sse/translator/index.ts`
+- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
+- Usage extraction/normalization: `open-sse/utils/usageTracking.ts`
+- Think tag parser: `open-sse/utils/thinkTagParser.ts`
+- Embedding handler: `open-sse/handlers/embeddings.ts`
+- Embedding provider registry: `open-sse/config/embeddingRegistry.ts`
+- Image generation handler: `open-sse/handlers/imageGeneration.ts`
+- Image provider registry: `open-sse/config/imageRegistry.ts`
+- Response sanitization: `open-sse/handlers/responseSanitizer.ts`
+- Role normalization: `open-sse/services/roleNormalizer.ts`
-- Výběr účtu/bodování: `open-sse/services/accountSelector.ts`
-- Kontextová správa životního cyklu: `open-sse/services/contextManager.ts`
-- Vynucení filtru IP: `open-sse/services/ipFilter.ts`
-- Sledování relací: `open-sse/services/sessionManager.ts`
-- Žádost o deduplikaci: `open-sse/services/signatureCache.ts`
-- Vložení příkazu systému: `open-sse/services/systemPrompt.ts`
+Services (business logic):
+
+- Account selection/scoring: `open-sse/services/accountSelector.ts`
+- Context lifecycle management: `open-sse/services/contextManager.ts`
+- IP filter enforcement: `open-sse/services/ipFilter.ts`
+- Session tracking: `open-sse/services/sessionManager.ts`
+- Request deduplication: `open-sse/services/signatureCache.ts`
+- System prompt injection: `open-sse/services/systemPrompt.ts`
- Thinking budget management: `open-sse/services/thinkingBudget.ts`
-- Směrování modelu se zástupnými znaky: `open-sse/services/wildcardRouter.ts`
-- Správa limitu sazby: `open-sse/services/rateLimitManager.ts`
-- Jistič: `open-sse/services/circuitBreaker.ts`
+- Wildcard model routing: `open-sse/services/wildcardRouter.ts`
+- Rate limit management: `open-sse/services/rateLimitManager.ts`
+- Circuit breaker: `open-sse/services/circuitBreaker.ts`
+- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy
+- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions
-Moduly vrstvy domény:
+Domain layer modules:
-- Dostupnost modelu: `src/lib/domain/modelAvailability.ts`
-- Pravidla/rozpočty nákladů: `src/lib/domain/costRules.ts`
-- Záložní zásady: `src/lib/domain/fallbackPolicy.ts`
+- Model availability: `src/lib/domain/modelAvailability.ts`
+- Cost rules/budgets: `src/lib/domain/costRules.ts`
+- Fallback policy: `src/lib/domain/fallbackPolicy.ts`
- Combo resolver: `src/lib/domain/comboResolver.ts`
-- Zásady uzamčení: `src/lib/domain/lockoutPolicy.ts`
-- Modul zásad: `src/domain/policyEngine.ts` — centralizované uzamčení → rozpočet → záložní vyhodnocení
-- Katalog chybových kódů: `src/lib/domain/errorCodes.ts`
-- ID požadavku: `src/lib/domain/requestId.ts`
-- Časový limit načtení: `src/lib/domain/fetchTimeout.ts`
-- Žádost o telemetrii: `src/lib/domain/requestTelemetry.ts`
-- Soulad/audit: `src/lib/domain/compliance/index.ts`
+- Lockout policy: `src/lib/domain/lockoutPolicy.ts`
+- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation
+- Error codes catalog: `src/lib/domain/errorCodes.ts`
+- Request ID: `src/lib/domain/requestId.ts`
+- Fetch timeout: `src/lib/domain/fetchTimeout.ts`
+- Request telemetry: `src/lib/domain/requestTelemetry.ts`
+- Compliance/audit: `src/lib/domain/compliance/index.ts`
- Eval runner: `src/lib/domain/evalRunner.ts`
-- Trvalost stavu domény: `src/lib/db/domainState.ts` — SQLite CRUD pro záložní řetězce, rozpočty, historii nákladů, stav uzamčení, jističe
+- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers
-Moduly poskytovatele OAuth (12 samostatných souborů pod `src/lib/oauth/providers/`):
+OAuth provider modules (12 individual files under `src/lib/oauth/providers/`):
-- Index registru: `src/lib/oauth/providers/index.ts`
- – Jednotliví poskytovatelé: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilo`codes`, `kilo`code
-- Tenký obal: `src/lib/oauth/providers.ts` — reexporty z jednotlivých modulů## 3) Persistence Layer
+- Registry index: `src/lib/oauth/providers/index.ts`
+- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
+- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules
-Primární stav DB (SQLite):
+## 3) Persistence Layer
-- Základní jádro: `src/lib/db/core.ts` (better-sqlite3, migrace, WAL)
-- Fasáda pro reexport: `src/lib/localDb.ts` (tenká vrstva kompatibility pro volající)
-- soubor: `${DATA_DIR}/storage.sqlite` (nebo `$XDG_CONFIG_HOME/omniroute/storage.sqlite`, pokud je nastaven, jinak `~/.omniroute/storage.sqlite`)
-- entity (tabulky + jmenné prostory KV): providerConnections, providerNodes, modelAliases, komba, apiKeys, nastavení, ceny,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt**
+Primary state DB (SQLite):
-Perzistence při používání:
+- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL)
+- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers)
+- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`)
+- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt**
-- fasáda: `src/lib/usageDb.ts` (rozložené moduly v `src/lib/usage/*`)
-- SQLite tabulky v `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs`
-- volitelné artefakty souborů zůstávají kvůli kompatibilitě/ladění (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `
/logs/...`)
-- starší soubory JSON jsou migrovány do SQLite migrací při spuštění, pokud jsou k dispozici
+Usage persistence:
-Stavová databáze domény (SQLite):
+- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`)
+- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs`
+- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`)
+- legacy JSON files are migrated to SQLite by startup migrations when present
-- `src/lib/db/domainState.ts` — operace CRUD pro stav domény
- – Tabulky (vytvořené v `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
-- Vzor mezipaměti pro zápis: mapy v paměti jsou autoritativní za běhu; mutace se zapisují synchronně do SQLite; stav je obnoven z DB při studeném startu## 4) Auth + Security Surfaces
+Domain State DB (SQLite):
-- Ověření souboru cookie řídicího panelu: `src/proxy.ts`, `src/app/api/auth/login/route.ts`
-- Generování/ověření klíče API: `src/shared/utils/apiKey.ts`
-- Tajné informace poskytovatele zůstaly v položkách `providerConnections`
-- Podpora odchozích proxy přes `open-sse/utils/proxyFetch.ts` (env vars) a `open-sse/utils/networkProxy.ts` (konfigurovatelné pro jednotlivé poskytovatele nebo globální)## 5) Cloud Sync
+- `src/lib/db/domainState.ts` — CRUD operations for domain state
+- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
+- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start
-- Init plánovače: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
-- Pravidelný úkol: `src/shared/services/cloudSyncScheduler.ts`
-- Pravidelný úkol: `src/shared/services/modelSyncScheduler.ts`
-- Řídící cesta: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`)
+## 4) Auth + Security Surfaces
+
+- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts`
+- API key generation/verification: `src/shared/utils/apiKey.ts`
+- Provider secrets persisted in `providerConnections` entries
+- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global)
+
+## 5) Cloud Sync
+
+- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
+- Periodic task: `src/shared/services/cloudSyncScheduler.ts`
+- Periodic task: `src/shared/services/modelSyncScheduler.ts`
+- Control route: `src/app/api/sync/cloud/route.ts`
+
+## Request Lifecycle (`/v1/chat/completions`)
```mermaid
sequenceDiagram
@@ -338,7 +363,9 @@ flowchart TD
Q -- No --> R[Return all unavailable]
```
-Záložní rozhodnutí jsou řízena `open-sse/services/accountFallback.ts` pomocí stavových kódů a heuristiky chybových zpráv. Kombinované směrování přidává ještě jednu ochranu: 400s v rozsahu poskytovatele, jako jsou selhání blokování obsahu a ověřování rolí, jsou považovány za lokální selhání modelu, takže pozdější kombinované cíle mohou stále běžet.## OAuth Onboarding and Token Refresh Lifecycle
+Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run.
+
+## OAuth Onboarding and Token Refresh Lifecycle
```mermaid
sequenceDiagram
@@ -368,7 +395,9 @@ sequenceDiagram
Test-->>UI: validation result
```
-Obnovení během živého provozu se provádí uvnitř `open-sse/handlers/chatCore.ts` prostřednictvím spouštěče `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable)
+Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`.
+
+## Cloud Sync Lifecycle (Enable / Sync / Disable)
```mermaid
sequenceDiagram
@@ -400,7 +429,9 @@ sequenceDiagram
Sync-->>UI: disabled
```
-Pravidelnou synchronizaci spouští „CloudSyncScheduler“, když je povolen cloud.## Data Model and Storage Map
+Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled.
+
+## Data Model and Storage Map
```mermaid
erDiagram
@@ -501,12 +532,14 @@ erDiagram
}
```
-Soubory fyzického úložiště:
+Physical storage files:
-- primární runtime DB: `${DATA_DIR}/storage.sqlite`
-- řádky protokolu požadavku: `${DATA_DIR}/log.txt` (artefakt compat/debug)
-- archivy strukturovaného obsahu volání: `${DATA_DIR}/call_logs/`
-- volitelné relace ladění překladatele/požadavku: `/logs/...`## Deployment Topology
+- primary runtime DB: `${DATA_DIR}/storage.sqlite`
+- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact)
+- structured call payload archives: `${DATA_DIR}/call_logs/`
+- optional translator/request debug sessions: `/logs/...`
+
+## Deployment Topology
```mermaid
flowchart LR
@@ -541,205 +574,249 @@ flowchart LR
### Route and API Modules
-- `src/app/api/v1/*`, `src/app/api/v1beta/*`: rozhraní API pro kompatibilitu
-- `src/app/api/v1/providers/[poskytovatel]/*`: vyhrazené trasy pro jednotlivé poskytovatele (chat, vkládání, obrázky)
-- `src/app/api/providers*`: poskytovatel CRUD, ověření, testování
-- `src/app/api/provider-nodes*`: vlastní kompatibilní správa uzlů
-- `src/app/api/provider-models`: správa vlastních modelů (CRUD)
-- `src/app/api/models/route.ts`: API katalogu modelů (aliasy + vlastní modely)
-- `src/app/api/oauth/*`: toky OAuth/kódu zařízení
-- `src/app/api/keys*`: životní cyklus místního klíče API
-- `src/app/api/models/alias`: správa aliasů
-- `src/app/api/combos*`: správa záložních kombinací
-- `src/app/api/pricing`: přepisy cen pro výpočet nákladů
-- `src/app/api/settings/proxy`: konfigurace proxy (GET/PUT/DELETE)
-- `src/app/api/settings/proxy/test`: test odchozího proxy připojení (POST)
-- `src/app/api/usage/*`: využití a protokoly API
-- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloudová synchronizace a pomocníci pro cloud
-- `src/app/api/cli-tools/*`: místní zapisovače/kontroly konfigurace CLI
-- `src/app/api/settings/ip-filter`: seznam povolených/blokovaných IP adres (GET/PUT)
-- `src/app/api/settings/thinking-budget`: konfigurace rozpočtu tokenu myšlení (GET/PUT)
-- `src/app/api/settings/system-prompt`: globální systémová výzva (GET/PUT)
-- `src/app/api/sessions`: seznam aktivních relací (GET)
-- `src/app/api/rate-limits`: stav limitu sazby na účet (GET)### Routing and Execution Core
+- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs
+- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images)
+- `src/app/api/providers*`: provider CRUD, validation, testing
+- `src/app/api/provider-nodes*`: custom compatible node management
+- `src/app/api/provider-models`: custom model management (CRUD)
+- `src/app/api/models/route.ts`: model catalog API (aliases + custom models)
+- `src/app/api/oauth/*`: OAuth/device-code flows
+- `src/app/api/keys*`: local API key lifecycle
+- `src/app/api/models/alias`: alias management
+- `src/app/api/combos*`: fallback combo management
+- `src/app/api/pricing`: pricing overrides for cost calculation
+- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE)
+- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST)
+- `src/app/api/usage/*`: usage and logs APIs
+- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers
+- `src/app/api/cli-tools/*`: local CLI config writers/checkers
+- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT)
+- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT)
+- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT)
+- `src/app/api/sessions`: active session listing (GET)
+- `src/app/api/rate-limits`: per-account rate limit status (GET)
-- `src/sse/handlers/chat.ts`: analýza požadavků, zpracování kombinací, smyčka výběru účtu
-- `open-sse/handlers/chatCore.ts`: překlad, odeslání exekutora, zpracování opakování/obnovení, nastavení streamu
-- `open-sse/executors/*`: chování sítě a formátu specifické pro poskytovatele### Translation Registry and Format Converters
+### Routing and Execution Core
-- `open-sse/translator/index.ts`: registr a orchestrace překladatelů
-- Požadavek na překladatele: `open-sse/translator/request/*`
-- Překladače odpovědí: `open-sse/translator/response/*`
-- Formátové konstanty: `open-sse/translator/formats.ts`### Persistence
+- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop
+- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup
+- `open-sse/executors/*`: provider-specific network and format behavior
-- `src/lib/db/*`: trvalá trvalá konfigurace/stav a doména na SQLite
-- `src/lib/localDb.ts`: reexport kompatibility pro moduly DB
-- `src/lib/usageDb.ts`: fasáda historie použití/protokolů volání nad tabulkami SQLite## Provider Executor Coverage (Strategy Pattern)
+### Translation Registry and Format Converters
-Každý poskytovatel má specializovaný spouštěč rozšiřující `BaseExecutor` (v `open-sse/executors/base.ts`), který poskytuje vytváření URL, konstrukci záhlaví, opakování s exponenciálním stažením, háky pro obnovení pověření a metodu orchestrace `execute()`.
+- `open-sse/translator/index.ts`: translator registry and orchestration
+- Request translators: `open-sse/translator/request/*`
+- Response translators: `open-sse/translator/response/*`
+- Format constants: `open-sse/translator/formats.ts`
-| Exekutor | Poskytovatel(é) | Speciální manipulace |
-| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------------------------------------------------------- |
-| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamická konfigurace URL/záhlaví na poskytovatele |
-| "AntigravityExecutor" | Google Antigravity | Vlastní ID projektů/relací, Opakovat po analýze |
-| "CodexExecutor" | Kodex OpenAI | Vkládá systémové instrukce, nutí k logickému úsilí |
-| `CursorExecutor` | Kurzor IDE | Protokol ConnectRPC, kódování Protobuf, podepisování požadavků pomocí kontrolního součtu |
-| "GithubExecutor" | GitHub Copilot | Obnovení tokenu druhého pilota, hlavičky napodobující VSCode |
-| "KiroExecutor" | AWS CodeWhisperer/Kiro | Binární formát AWS EventStream → konverze SSE |
-| "GeminiCLIExecutor" | Gemini CLI | Cyklus obnovení tokenu Google OAuth |
+### Persistence
-Všichni ostatní poskytovatelé (včetně vlastních kompatibilních uzlů) používají `DefaultExecutor`.## Provider Compatibility Matrix
+- `src/lib/db/*`: persistent config/state and domain persistence on SQLite
+- `src/lib/localDb.ts`: compatibility re-export for DB modules
+- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables
-| Poskytovatel | Formát | Auth | Stream | Nestreamovat | Obnovení tokenu | Použití API |
-| ---------------- | ---------------- | ------------------------ | -------------------- | ------------ | --------------- | ------------------- | ------------------------------ |
-| Claude | claude | Klíč API / OAuth | ✅ | ✅ | ✅ | ⚠️ Pouze správce |
-| Blíženci | Blíženci | Klíč API / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloudová konzole |
-| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloudová konzole |
-| Antigravitace | antigravitace | OAuth | ✅ | ✅ | ✅ | ✅ Plná kvóta API |
-| OpenAI | openai | API klíč | ✅ | ✅ | ❌ | ❌ |
-| Codex | openai-responses | OAuth | ✅ nuceně | ❌ | ✅ | ✅ Sazbové limity |
-| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Snímky kvót |
-| Kurzor | kurzor | Vlastní kontrolní součet | ✅ | ✅ | ❌ | ❌ |
-| Kiro | kiro | AWS SSO OIDC | ✅ (Stream událostí) | ❌ | ✅ | ✅ Limity použití |
-| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Na vyžádání |
-| Qoder | openai | OAuth (základní) | ✅ | ✅ | ✅ | ⚠️ Na vyžádání |
-| OpenRouter | openai | API klíč | ✅ | ✅ | ❌ | ❌ |
-| GLM/Kimi/MiniMax | claude | API klíč | ✅ | ✅ | ❌ | ❌ |
-| DeepSeek | openai | API klíč | ✅ | ✅ | ❌ | ❌ |
-| Groq | openai | API klíč | ✅ | ✅ | ❌ | ❌ |
-| xAI (Grok) | openai | API klíč | ✅ | ✅ | ❌ | ❌ |
-| Mistral | openai | API klíč | ✅ | ✅ | ❌ | ❌ |
-| Zmatenost | openai | API klíč | ✅ | ✅ | ❌ | ❌ |
-| Společně AI | openai | Klíč API | ✅ | ✅ | ❌ | ❌ |
-| Ohňostroje AI | openai | API klíč | ✅ | ✅ | ❌ | ❌ |
-| Cerebras | openai | API klíč | ✅ | ✅ | ❌ | ❌ |
-| Cohere | openai | API klíč | ✅ | ✅ | ❌ | ❌ |
-| NVIDIA NIM | openai | API klíč | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage |
+## Provider Executor Coverage (Strategy Pattern)
-Mezi zjištěné zdrojové formáty patří:
+Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method.
-- "openai".
-- "openai-odpovědi".
-- "claude".
-- "blíženci".
+| Executor | Provider(s) | Special Handling |
+| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- |
+| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider |
+| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing |
+| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort |
+| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum |
+| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers |
+| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion |
+| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle |
-Mezi cílové formáty patří:
+All other providers (including custom compatible nodes) use the `DefaultExecutor`.
-- OpenAI chat/odpovědi
+## Provider Compatibility Matrix
+
+| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API |
+| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ |
+| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only |
+| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API |
+| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits |
+| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots |
+| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ |
+| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits |
+| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request |
+| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request |
+| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ |
+| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+
+## Format Translation Coverage
+
+Detected source formats include:
+
+- `openai`
+- `openai-responses`
+- `claude`
+- `gemini`
+
+Target formats include:
+
+- OpenAI chat/Responses
- Claude
-- Gemini/Gemini-CLI/Antigravitační obálka
+- Gemini/Gemini-CLI/Antigravity envelope
- Kiro
-- Kurzor
+- Cursor
-Překlady používají**OpenAI jako formát centra**— všechny konverze procházejí přes OpenAI jako prostředník:```
+Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate:
+
+```
Source Format → OpenAI (hub) → Target Format
+```
-````
+Translations are selected dynamically based on source payload shape and provider target format.
-Překlady jsou vybírány dynamicky na základě tvaru zdrojové užitečné zátěže a cílového formátu poskytovatele.
+Additional processing layers in the translation pipeline:
-Další vrstvy zpracování v překladovém potrubí:
+- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance
+- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE)
+- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field
+- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema`
--**Dezinfekce odezvy**– Odstraňuje nestandardní pole z odpovědí ve formátu OpenAI (streamovaných i nestreamovaných), aby byla zajištěna přísná shoda se sadou SDK
--**Normalizace rolí**— Převádí `vývojář` → `systém` pro jiné cíle než OpenAI; sloučí `systém` → `uživatel` pro modely, které odmítají systémovou roli (GLM, ERNIE)
-–**Extrakce značek Think**– analyzuje bloky „...“ z obsahu do pole „reasoning_content“
-–**Strukturovaný výstup**– Převádí OpenAI `response_format.json_schema` na Gemini `responseMimeType` + `responseSchema`## Supported API Endpoints
+## Supported API Endpoints
-| Koncový bod | Formát | Psovod |
-| --------------------------------------------------- | ------------------- | -------------------------------------------------------------------- |
-| `POST /v1/chat/completions` | Chat OpenAI | `src/sse/handlers/chat.ts` |
-| `POST /v1/messages` | Claude Messages | Stejná obsluha (automaticky zjištěna) |
-| `POST /v1/responses` | Odezvy OpenAI | `open-sse/handlers/responsesHandler.ts` |
-| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` |
-| `ZÍSKAT /v1/embeddings` | Seznam modelů | Cesta API |
-| `POST /v1/images/generations` | Obrázky OpenAI | `open-sse/handlers/imageGeneration.ts` |
-| `ZÍSKAT /v1/images/generations` | Seznam modelů | Cesta API |
-| `POST /v1/providers/{provider}/chat/completions` | Chat OpenAI | Vyhrazené na poskytovatele s ověřením modelu |
-| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Vyhrazené na poskytovatele s ověřením modelu |
-| `POST /v1/providers/{poskytovatel}/images/generations` | Obrázky OpenAI | Vyhrazené na poskytovatele s ověřením modelu |
-| `POST /v1/messages/count_tokens` | Počet tokenů Claude | Cesta API |
-| `GET /v1/models` | Seznam modelů OpenAI | Cesta API (chat + vkládání + obrázek + vlastní modely) |
-| `GET /api/models/catalog` | Katalog | Všechny modely seskupené podle poskytovatele + typ |
-| `POST /v1beta/models/*:streamGenerateContent` | Blíženec domorodec | Cesta API |
-| `GET/PUT/DELETE /api/settings/proxy` | Konfigurace proxy | Konfigurace síťového proxy |
-| `POST /api/settings/proxy/test` | Připojení proxy | Koncový bod testu stavu proxy/konektivity |
-| `GET/POST/DELETE /api/provider-models` | Modely poskytovatelů | Vlastní a spravované dostupné modely podporují metadata modelu poskytovatele |## Bypass Handler
+| Endpoint | Format | Handler |
+| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- |
+| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` |
+| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) |
+| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` |
+| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` |
+| `GET /v1/embeddings` | Model listing | API route |
+| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` |
+| `GET /v1/images/generations` | Model listing | API route |
+| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation |
+| `POST /v1/messages/count_tokens` | Claude Token Count | API route |
+| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) |
+| `GET /api/models/catalog` | Catalog | All models grouped by provider + type |
+| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route |
+| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration |
+| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint |
+| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models |
-Obslužná rutina bypassu (`open-sse/utils/bypassHandler.ts`) zachycuje známé požadavky na „zahození“ od Claude CLI – zahřívací pingy, extrakce titulů a počty tokenů – a vrací**falešnou odpověď**, aniž by spotřebovával tokeny poskytovatele upstream. To se spustí pouze v případě, že `User-Agent` obsahuje `claude-cli`.## Request Logger Pipeline
+## Bypass Handler
-Záznamník požadavků (`open-sse/utils/requestLogger.ts`) poskytuje 7fázový kanál protokolování ladění, který je ve výchozím nastavení vypnutý, povolený pomocí `ENABLE_REQUEST_LOGS=true`:```
+The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`.
+
+## Request Logger Pipeline
+
+The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`:
+
+```
1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json
→ 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt
-````
+```
-Soubory se zapisují do `/logs//` pro každou relaci požadavku.## Failure Modes and Resilience
+Files are written to `/logs//` for each request session.
+
+## Failure Modes and Resilience
## 1) Account/Provider Availability
-- Cooldown účtu poskytovatele při přechodných chybách/chybách rychlosti/autorizace
-- záložní účet před neúspěšným žádostí
-- Záloha kombinovaného modelu, když je vyčerpána aktuální cesta modelu/poskytovatele## 2) Token Expiry
+- provider account cooldown on transient/rate/auth errors
+- account fallback before failing request
+- combo model fallback when current model/provider path is exhausted
-- Předběžná kontrola a obnovení s opakovaným pokusem pro poskytovatele obnovitelných zdrojů
-- 401/403 opakování po pokusu o obnovení v cestě jádra## 3) Stream Safety
+## 2) Token Expiry
-- řadič toku s vědomím odpojení
-- překladový proud s vyprázdněním konce proudu a zpracováním `[DONE]`
-- záložní odhad využití, když chybí metadata využití poskytovatele## 4) Cloud Sync Degradation
+- pre-check and refresh with retry for refreshable providers
+- 401/403 retry after refresh attempt in core path
-- Objeví se chyby synchronizace, ale místní běh pokračuje
-- plánovač má logiku umožňující opakování, ale periodické spouštění aktuálně standardně volá synchronizaci na jeden pokus## 5) Data Integrity
+## 3) Stream Safety
-- Migrace schémat SQLite a automatické upgrady při spuštění
-- starší cesta ke kompatibilitě migrace JSON → SQLite## Observability and Operational Signals
+- disconnect-aware stream controller
+- translation stream with end-of-stream flush and `[DONE]` handling
+- usage estimation fallback when provider usage metadata is missing
-Zdroje viditelnosti za běhu:
+## 4) Cloud Sync Degradation
-- protokoly konzoly z `src/sse/utils/logger.ts`
-- agregáty využití na žádost v SQLite (`usage_history`, `call_logs`, `proxy_logs`)
-- čtyřfázové podrobné zachycení užitečného zatížení v SQLite (`request_detail_logs`), když `settings.detailed_logs_enabled=true`
-- textový protokol o stavu požadavku v `log.txt` (nepovinné/kompatibilní)
-- volitelné protokoly hlubokých požadavků/překladů pod `logs/`, když `ENABLE_REQUEST_LOGS=true`
-- koncové body využití řídicího panelu (`/api/usage/*`) pro spotřebu uživatelského rozhraní
+- sync errors are surfaced but local runtime continues
+- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default
-Podrobné zachycení datové části požadavku ukládá až čtyři fáze datové zátěže JSON na směrované volání:
+## 5) Data Integrity
-- nezpracovaný požadavek přijatý od klienta
-- přeložená žádost skutečně odeslaná proti proudu
-- odpověď poskytovatele rekonstruovaná jako JSON; streamované odpovědi jsou komprimovány do konečného shrnutí plus metadata streamu
-- konečná odpověď klienta vrácená OmniRoute; streamované odpovědi jsou uloženy ve stejném kompaktním souhrnném formuláři## Security-Sensitive Boundaries
+- SQLite schema migrations and auto-upgrade hooks at startup
+- legacy JSON → SQLite migration compatibility path
-- Tajný klíč JWT (`JWT_SECRET`) zajišťuje ověřování/podepisování souborů cookie relace řídicího panelu
-- Počáteční zaváděcí heslo (`INITIAL_PASSWORD`) by mělo být explicitně nakonfigurováno pro zřizování při prvním spuštění
-- Tajný klíč API HMAC (`API_KEY_SECRET`) zabezpečuje vygenerovaný formát lokálního klíče API
-- Tajné informace poskytovatele (klíče/tokeny API) jsou uloženy v místní databázi a měly by být chráněny na úrovni souborového systému
-- Koncové body synchronizace cloudu se spoléhají na sémantiku klíče API + ID počítače## Environment and Runtime Matrix
+## Observability and Operational Signals
-Proměnné prostředí aktivně používané kódem:
+Runtime visibility sources:
-- Aplikace/auth: `JWT_SECRET`, `INITIAL_PASSWORD`
-- Úložiště: `DATA_DIR`
-- Kompatibilní chování uzlu: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
-- Volitelné přepsání základny úložiště (Linux/macOS, když není `DATA_DIR` nastaveno): `XDG_CONFIG_HOME`
- – Bezpečnostní hash: `API_KEY_SECRET`, `MACHINE_ID_SALT`
-- Protokolování: `ENABLE_REQUEST_LOGS`
- – Synchronizace/cloudové URL: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
- – Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` a varianty s malými písmeny
-- Příznaky funkce SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
-- Pomocníci platformy/běhu (nikoli konfigurace specifická pro aplikaci): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes
+- console logs from `src/sse/utils/logger.ts`
+- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`)
+- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true`
+- textual request status log in `log.txt` (optional/compat)
+- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true`
+- dashboard usage endpoints (`/api/usage/*`) for UI consumption
-1. `usageDb` a `localDb` sdílejí stejnou zásadu základního adresáře (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) se starší migrací souborů.
-2. `/api/v1/route.ts` deleguje stejný tvůrce jednotného katalogu, který používá `/api/v1/models` (`src/app/api/v1/models/catalog.ts`), aby se zabránilo sémantickému posunu.
-3. Pokud je povoleno, zapisovač požadavku zapisuje celé záhlaví/tělo; považovat adresář log za citlivý.
-4. Chování cloudu závisí na správné dosažitelnosti koncového bodu cloudu „NEXT_PUBLIC_BASE_URL“.
-5. Adresář `open-sse/` je publikován jako balíček `@omniroute/open-sse`**npm workspace**. Zdrojový kód jej importuje přes `@omniroute/open-sse/...` (vyřešeno Next.js `transpilePackages`). Cesty k souborům v tomto dokumentu stále používají název adresáře `open-sse/` kvůli konzistenci.
-6. Grafy v řídicím panelu používají**Recharts**(založené na SVG) pro přístupné, interaktivní analytické vizualizace (sloupcové grafy využití modelu, tabulky rozdělení poskytovatelů s mírou úspěšnosti).
-7. E2E testy používají**Playwright**(`tests/e2e/`), spouštěné přes `npm run test:e2e`. Unit testy používají**Node.js test runner**(`tests/unit/`), spouštějí se přes `npm run test:unit`. Zdrojový kód pod `src/` je**TypeScript**(`.ts`/`.tsx`); pracovní prostor `open-sse/` zůstává JavaScriptem (`.js`).
-8. Stránka Nastavení je uspořádána do 5 záložek: Zabezpečení, Směrování (6 globálních strategií: fill-first, round-robin, p2c, náhodné, nejméně používané, nákladově optimalizované), Odolnost (upravitelné rychlostní limity, jistič, zásady), AI (rozpočet myšlení, systémová výzva, mezipaměť výzvy), Pokročilé (proxy).## Operational Verification Checklist
+Detailed request payload capture stores up to four JSON payload stages per routed call:
-- Sestavení ze zdroje: `npm run build`
-- Sestavení obrazu Dockeru: `docker build -t omniroute .`
-- Spusťte službu a ověřte:
+- raw request received from the client
+- translated request actually sent upstream
+- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata
+- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form
+
+## Security-Sensitive Boundaries
+
+- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing
+- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning
+- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format
+- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level
+- Cloud sync endpoints rely on API key auth + machine id semantics
+
+## Environment and Runtime Matrix
+
+Environment variables actively used by code:
+
+- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD`
+- Storage: `DATA_DIR`
+- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
+- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME`
+- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT`
+- Logging: `ENABLE_REQUEST_LOGS`
+- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
+- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants
+- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
+- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`
+
+## Known Architectural Notes
+
+1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration.
+2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift.
+3. Request logger writes full headers/body when enabled; treat log directory as sensitive.
+4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability.
+5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency.
+6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates).
+7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`).
+8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy).
+9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed.
+10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22.
+11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible.
+
+## Operational Verification Checklist
+
+- Build from source: `npm run build`
+- Build Docker image: `docker build -t omniroute .`
+- Start service and verify:
- `GET /api/settings`
- `GET /api/v1/models`
-- Základní adresa URL cíle CLI by měla být `http://:20128/v1`, když `PORT=20128`
+- CLI target base URL should be `http://:20128/v1` when `PORT=20128`
diff --git a/docs/i18n/cs/docs/FEATURES.md b/docs/i18n/cs/docs/FEATURES.md
index cb8205cd0f..0b538261e5 100644
--- a/docs/i18n/cs/docs/FEATURES.md
+++ b/docs/i18n/cs/docs/FEATURES.md
@@ -4,102 +4,168 @@
---
-Vizuální průvodce každou částí řídicího panelu OmniRoute.---
+
+
+Visual guide to every section of the OmniRoute dashboard.
+
+---
## 🔌 Providers
-Správa připojení poskytovatelů AI: poskytovatelé OAuth (Claude Code, Codex, Gemini CLI), poskytovatelé klíčů API (Groq, DeepSeek, OpenRouter) a bezplatní poskytovatelé (Qoder, Qwen, Kiro). Účty Kiro zahrnují sledování zůstatku kreditu – zbývající kredity, celkový příspěvek a datum obnovení jsou viditelné v Dashboard → Použití.
+Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage.
+
+
---
## 🎨 Combos
-Vytvářejte komba směrování modelů se 6 strategiemi: prioritní, vážená, cyklická, náhodná, nejméně používaná a nákladově optimalizovaná. Každé kombo řetězí více modelů s automatickým nouzovým návratem a zahrnuje rychlé šablony a kontroly připravenosti.
+Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks.
+
+
---
## 📊 Analytics
-Komplexní analýzy využití se spotřebou tokenů, odhady nákladů, teplotní mapy aktivit, týdenní distribuční grafy a rozpisy podle poskytovatelů.
+Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns.
+
+
---
## 🏥 System Health
-Monitorování v reálném čase: doba provozuschopnosti, paměť, verze, percentily latence (p50/p95/p99), statistika mezipaměti a stavy jističe poskytovatele.
+Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states.
+
+
---
## 🔧 Translator Playground
-Čtyři režimy pro ladění překladů API:**Playground**(konvertor formátů),**Chat Tester**(živé požadavky),**Test Bench**(dávkové testy) a**Live Monitor**(stream v reálném čase).
+Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream).
+
+
---
## 🎮 Model Playground _(v2.0.9+)_
-Otestujte jakýkoli model přímo z palubní desky. Vyberte poskytovatele, model a koncový bod, pište výzvy pomocí editoru Monaco, streamujte odpovědi v reálném čase, rušte uprostřed streamu a zobrazujte metriky časování.---
+Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics.
+
+---
## 🎨 Themes _(v2.0.5+)_
-Přizpůsobitelné barevné motivy pro celý přístrojový panel. Vyberte si ze 7 přednastavených barev (korálová, modrá, červená, zelená, fialová, oranžová, azurová) nebo si vytvořte vlastní motiv výběrem libovolné šestihranné barvy. Podporuje světlý, tmavý a systémový režim.---
+Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode.
+
+---
## ⚙️ Settings
-Komplexní panel nastavení s kartami:
+Comprehensive settings panel with tabs:
--**Obecné**— Systémové úložiště, správa zálohování (export/import databáze) -**Vzhled**— Volič motivu (tmavý/světlý/systém), přednastavení barevných motivů a vlastní barvy, viditelnost zdravotního deníku, ovládací prvky viditelnosti položek na postranním panelu -**Zabezpečení**— Ochrana koncových bodů API, blokování vlastního poskytovatele, filtrování IP, informace o relaci -**Směrování**— Modelové aliasy, degradace úloh na pozadí -**Odolnost**- Perzistence rychlostního limitu, ladění jističe, automatické deaktivace zakázaných účtů, sledování expirace poskytovatele -**Advanced**– Přepisy konfigurace, auditní záznam konfigurace, režim degradace záložního řešení
+- **General** — System storage, backup management (export/import database)
+- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls
+- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info
+- **Routing** — Model aliases, background task degradation
+- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration
+- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode
+
+
---
## 🔧 CLI Tools
-Konfigurace jedním kliknutím pro nástroje pro kódování AI: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor a Factory Droid. Obsahuje automatické nastavení konfigurace/resetování, profily připojení a mapování modelu.
+One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping.
+
+
---
## 🤖 CLI Agents _(v2.0.11+)_
-Dashboard pro zjišťování a správu agentů CLI. Zobrazuje mřížku 14 vestavěných agentů (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) s:
+Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with:
--**Stav instalace**— Instalováno / Nenalezeno s detekcí verze -**Odznaky protokolu**— stdio, HTTP atd. -**Vlastní agenti**— Zaregistrujte jakýkoli nástroj CLI prostřednictvím formuláře (název, binární soubor, příkaz verze, spawn args) -**CLI Fingerprint Matching**– Přepínání na jednotlivé poskytovatele, aby odpovídalo nativním podpisům požadavků CLI, čímž se snižuje riziko zákazu při zachování IP adresy proxy---
+- **Installation status** — Installed / Not Found with version detection
+- **Protocol badges** — stdio, HTTP, etc.
+- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args)
+- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP
+
+---
+
+## 🔗 Context Relay _(v3.5.5+)_
+
+A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context.
+
+Configurable via combo-level or global settings:
+- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%)
+- **Max Messages For Summary** — How much recent history to condense
+- **Summary Model** — Optional override model for generating the handoff summary
+
+Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md).
+
+---
+
+## 🛡️ Proxy Hardening _(v3.5.5+)_
+
+Comprehensive proxy configuration enforcement across the entire request pipeline:
+
+- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments
+- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings
+- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22
+- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS
+
+---
## 🖼️ Media _(v2.0.3+)_
-Generujte obrázky, videa a hudbu z řídicího panelu. Podporuje OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open a MusicGen.---
+Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen.
+
+---
## 📝 Request Logs
-Protokolování požadavků v reálném čase s filtrováním podle poskytovatele, modelu, účtu a klíče API. Zobrazuje stavové kódy, využití tokenu, latenci a podrobnosti o odpovědi.
+Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details.
+
+
---
## 🌐 API Endpoint
-Váš sjednocený koncový bod API s rozdělením schopností: Dokončení chatu, API odpovědí, vkládání, generování obrázků, změna pořadí, přepis zvuku, převod textu na řeč, moderování a registrované klíče rozhraní API. Integrace Cloudflare Quick Tunnel a podpora cloudového proxy pro vzdálený přístup.
+Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access.
+
+
---
## 🔑 API Key Management
-Vytvářejte, upravujte a rušte klíče API. Každý klíč může být omezen na konkrétní modely/poskytovatele s plným přístupem nebo oprávněním pouze pro čtení. Vizuální správa klíčů se sledováním využití.---
+Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking.
+
+---
## 📋 Audit Log
-Sledování administrativních akcí s filtrováním podle typu akce, aktéra, cíle, IP adresy a časového razítka. Úplná historie událostí zabezpečení.---
+Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history.
+
+---
## 🖥️ Desktop Application
-Nativní desktopová aplikace Electron pro Windows, macOS a Linux. Spusťte OmniRoute jako samostatnou aplikaci s integrací na systémové liště, offline podporou, automatickou aktualizací a instalací jedním kliknutím.
+Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install.
-Klíčové vlastnosti:
+Key features:
-- Dotazování připravenosti serveru (žádná prázdná obrazovka při studeném startu)
-- Systémová lišta se správou portů
-- Zásady zabezpečení obsahu
-- Jednoinstanční zámek
-- Automatická aktualizace při restartu
-- Platformově podmíněné uživatelské rozhraní (semafory macOS, výchozí titulek Windows/Linux)
-- Balení sestavení Hardened Electron – symbolicky propojené `node_modules` v samostatném balíčku jsou detekovány a odmítnuty před zabalením, čímž se zabrání závislosti běhu na sestavení (v2.5.5+)
+- Server readiness polling (no blank screen on cold start)
+- System tray with port management
+- Content Security Policy
+- Single-instance lock
+- Auto-update on restart
+- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar)
+- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+)
-📖 Úplnou dokumentaci naleznete v [`electron/README.md`](../electron/README.md).
+📖 See [`electron/README.md`](../electron/README.md) for full documentation.
diff --git a/docs/i18n/cs/docs/TROUBLESHOOTING.md b/docs/i18n/cs/docs/TROUBLESHOOTING.md
index a6efcaa9fa..8b2b5824f2 100644
--- a/docs/i18n/cs/docs/TROUBLESHOOTING.md
+++ b/docs/i18n/cs/docs/TROUBLESHOOTING.md
@@ -4,69 +4,142 @@
---
-Běžné problémy a řešení pro OmniRoute.---
+
+
+Common problems and solutions for OmniRoute.
+
+---
## Quick Fixes
-| Problém | Řešení |
-| ---------------------------------------- | ------------------------------------------------------------------------------------- | --- |
-| První přihlášení nefunguje | Nastavit `INITIAL_PASSWORD` v `.env` (žádné napevno zakódované výchozí nastavení) |
-| Dashboard se otevírá na nesprávném portu | Nastavit `PORT=20128` a `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
-| Žádné záznamy požadavků pod `logs/` | Nastavte `ENABLE_REQUEST_LOGS=true` |
-| EACCES: povolení odepřeno | Nastavte `DATA_DIR=/cesta/k/zapisovatelnému/adresáři` tak, aby přepsal `~/.omniroute` |
-| Strategie směrování se neukládá | Aktualizace na v1.4.11+ (oprava schématu Zod pro trvalost nastavení) | --- |
+| Problem | Solution |
+| ----------------------------- | ------------------------------------------------------------------ |
+| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) |
+| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
+| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` |
+| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` |
+| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) |
+| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below |
+| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below |
+
+---
+
+## Node.js Compatibility
+
+
+
+### Login page crashes or shows "Module self-registration" error
+
+**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database.
+
+**Symptoms:**
+- Login page shows a blank screen or a server error
+- Console shows `Error: Module did not self-register` or similar native binding errors
+- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected
+
+**Fix:**
+
+1. Install Node.js 22 LTS (recommended):
+ ```bash
+ nvm install 22
+ nvm use 22
+ ```
+2. Verify your version: `node --version` should show `v22.x.x`
+3. Reinstall OmniRoute: `npm install -g omniroute`
+4. Restart: `omniroute`
+
+> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**.
+
+---
+
+## Proxy Issues
+
+
+
+### Provider validation shows "fetch failed"
+
+**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing.
+
+**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically.
+
+### Token health check fails with "fetch failed"
+
+**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection.
+
+**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+.
+
+### SOCKS5 proxy returns "invalid onRequestStart method"
+
+**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation.
+
+**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+.
+
+---
## Provider Issues
### "Language model did not provide messages"
-**Příčina:**Kvóta poskytovatele je vyčerpána.
+**Cause:** Provider quota exhausted.
-**Oprava:**
+**Fix:**
-1. Zkontrolujte sledování kvót na řídicím panelu
-2. Použijte kombinaci se záložními úrovněmi
-3. Přejděte na levnější/bezplatnou úroveň### Rate Limiting
+1. Check dashboard quota tracker
+2. Use a combo with fallback tiers
+3. Switch to cheaper/free tier
-**Příčina:**Vyčerpaná kvóta předplatného.
+### Rate Limiting
-**Oprava:**
+**Cause:** Subscription quota exhausted.
-– Přidejte záložní: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+**Fix:**
-- Použijte GLM/MiniMax jako levnou zálohu### OAuth Token Expired
+- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Use GLM/MiniMax as cheap backup
-OmniRoute automaticky obnovuje tokeny. Pokud problémy přetrvávají:
+### OAuth Token Expired
-1. Ovládací panel → Poskytovatel → Znovu připojit
-2. Odstraňte a znovu přidejte připojení poskytovatele---
+OmniRoute auto-refreshes tokens. If issues persist:
+
+1. Dashboard → Provider → Reconnect
+2. Delete and re-add the provider connection
+
+---
## Cloud Issues
### Cloud Sync Errors
-1. Ověřte, že `BASE_URL` odkazuje na vaši spuštěnou instanci (např. `http://localhost:20128`)
-2. Ověřte, že `CLOUD_URL` odkazuje na váš koncový bod cloudu (např. `https://omniroute.dev`)
-3. Udržujte hodnoty `NEXT_PUBLIC_*` zarovnané s hodnotami na straně serveru### Cloud `stream=false` Returns 500
+1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`)
+2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`)
+3. Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**Příznak:**`Neočekávaný token 'd'...` na koncovém bodu cloudu pro nestreamovaná volání.
+### Cloud `stream=false` Returns 500
-**Příčina:**Upstream vrací užitečné zatížení SSE, zatímco klient očekává JSON.
+**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls.
-**Řešení:**Pro přímá cloudová volání použijte `stream=true`. Místní běhové prostředí zahrnuje záložní SSE→JSON.### Cloud Says Connected but "Invalid API key"
+**Cause:** Upstream returns SSE payload while client expects JSON.
-1. Vytvořte nový klíč z místního řídicího panelu (`/api/keys`)
-2. Spusťte synchronizaci s cloudem: Povolte cloud → Synchronizovat nyní
-3. Staré/nesynchronizované klíče mohou v cloudu stále vracet „401“.---
+**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback.
+
+### Cloud Says Connected but "Invalid API key"
+
+1. Create a fresh key from local dashboard (`/api/keys`)
+2. Run cloud sync: Enable Cloud → Sync Now
+3. Old/non-synced keys can still return `401` on cloud
+
+---
## Docker Issues
### CLI Tool Shows Not Installed
-1. Zkontrolujte pole runtime: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
-2. Pro přenosný režim: použijte cíl obrazu `runner-cli` (přibalená rozhraní CLI)
-3. Pro režim připojení hostitele: nastavte `CLI_EXTRA_PATHS` a připojte adresář hostitele bin jako pouze pro čtení
-4. Pokud `installed=true` a `runnable=false`: binární soubor byl nalezen, ale neprošel zdravotní kontrolou### Quick Runtime Validation
+1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
+2. For portable mode: use image target `runner-cli` (bundled CLIs)
+3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only
+4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck
+
+### Quick Runtime Validation
```bash
curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
@@ -80,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,
### High Costs
-1. Zkontrolujte statistiky využití v Dashboard → Usage
-2. Přepněte primární model na GLM/MiniMax
-3. Pro nekritické úkoly používejte bezplatnou vrstvu (Gemini CLI, Qoder).
-4. Nastavte rozpočty nákladů na klíč API: Dashboard → API Keys → Budget---
+1. Check usage stats in Dashboard → Usage
+2. Switch primary model to GLM/MiniMax
+3. Use free tier (Gemini CLI, Qoder) for non-critical tasks
+4. Set cost budgets per API key: Dashboard → API Keys → Budget
+
+---
## Debugging
### Enable Request Logs
-V souboru `.env` nastavte `ENABLE_REQUEST_LOGS=true`. Protokoly se zobrazují v adresáři `logs/`.### Check Provider Health
+Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory.
+
+### Check Provider Health
```bash
# Health dashboard
@@ -101,106 +178,135 @@ curl http://localhost:20128/api/monitoring/health
### Runtime Storage
-- Hlavní stav: `${DATA_DIR}/storage.sqlite` (poskytovatelé, komba, aliasy, klíče, nastavení)
-- Použití: SQLite tabulky v `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + volitelné `${DATA_DIR}/log.txt` a `${DATA_DIR}/call_logs/`
-- Protokoly požadavků: `/logs/...` (když `ENABLE_REQUEST_LOGS=true`)---
+- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings)
+- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/`
+- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`)
+
+---
## Circuit Breaker Issues
### Provider stuck in OPEN state
-Když je jistič poskytovatele OTEVŘENÝ, požadavky jsou blokovány, dokud nevyprší cooldown.
+When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires.
-**Oprava:**
+**Fix:**
-1. Přejděte na**Hlavní panel → Nastavení → Odolnost**
-2. Zkontrolujte kartu jističe pro dotčeného poskytovatele
-3. Kliknutím na**Resetovat vše**vymažete všechny jističe nebo počkejte, až vyprší cooldown
-4. Před resetováním ověřte, zda je poskytovatel skutečně dostupný### Provider keeps tripping the circuit breaker
+1. Go to **Dashboard → Settings → Resilience**
+2. Check the circuit breaker card for the affected provider
+3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire
+4. Verify the provider is actually available before resetting
-Pokud poskytovatel opakovaně přejde do stavu OTEVŘENO:
+### Provider keeps tripping the circuit breaker
-1. Zkontrolujte**Dashboard → Health → Provider Health**pro vzor selhání
-2. Přejděte na**Nastavení → Odolnost → Profily poskytovatelů**a zvyšte práh selhání
-3. Zkontrolujte, zda poskytovatel nezměnil limity API nebo vyžaduje opětovné ověření
-4. Zkontrolujte telemetrii latence – vysoká latence může způsobit selhání na základě časového limitu---
+If a provider repeatedly enters OPEN state:
+
+1. Check **Dashboard → Health → Provider Health** for the failure pattern
+2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold
+3. Check if the provider has changed API limits or requires re-authentication
+4. Review latency telemetry — high latency may cause timeout-based failures
+
+---
## Audio Transcription Issues
### "Unsupported model" error
-- Ujistěte se, že používáte správnou předponu: `deepgram/nova-3` nebo `assemblyai/best`
- – Ověřte, že je poskytovatel připojen v**Dashboard → Providers**### Transcription returns empty or fails
+- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best`
+- Verify the provider is connected in **Dashboard → Providers**
-- Zkontrolujte podporované zvukové formáty: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm`
-- Ověřte, zda je velikost souboru v rámci limitů poskytovatele (obvykle < 25 MB)
-- Zkontrolujte platnost klíče API poskytovatele na kartě poskytovatele---
+### Transcription returns empty or fails
+
+- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm`
+- Verify file size is within provider limits (typically < 25MB)
+- Check provider API key validity in the provider card
+
+---
## Translator Debugging
-K ladění problémů s překladem formátu použijte**Dashboard → Translator**:
+Use **Dashboard → Translator** to debug format translation issues:
-| Režim | Kdy použít |
-| -------------------- | ----------------------------------------------------------------------------------------------------------- | ------------------------ |
-| **Hřiště** | Porovnejte vstupní/výstupní formáty vedle sebe — vložte neúspěšný požadavek, abyste viděli, jak se překládá |
-| **Chat Tester** | Odesílejte živé zprávy a kontrolujte celý obsah požadavku/odpovědi včetně záhlaví |
-| **Zkušební stolice** | Spusťte dávkové testy napříč kombinacemi formátů, abyste zjistili, které překlady jsou poškozené |
-| **Živý monitor** | Sledujte tok požadavků v reálném čase, abyste zachytili občasné problémy s překladem | ### Common format issues |
+| Mode | When to Use |
+| ---------------- | -------------------------------------------------------------------------------------------- |
+| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates |
+| **Chat Tester** | Send live messages and inspect the full request/response payload including headers |
+| **Test Bench** | Run batch tests across format combinations to find which translations are broken |
+| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues |
--**Značky myšlení se nezobrazují**— Zkontrolujte, zda cílový poskytovatel podporuje myšlení a nastavení rozpočtu na myšlení -**Přerušení volání nástroje**— Některé překlady formátů mohou odstranit nepodporovaná pole; ověřit v režimu Playground -**Chybí systémová výzva**– Claude a Gemini zacházejí s výzvami systému odlišně; zkontrolovat překladový výstup
-–**SDK vrací surový řetězec místo objektu**– Opraveno ve verzi 1.1.0: sanitizér odpovědi nyní odstraňuje nestandardní pole (`x_groq`, `usage_breakdown` atd.), která způsobují selhání ověření OpenAI SDK Pydantic -**GLM/ERNIE odmítá `systémovou` roli**— Opraveno ve verzi 1.1.0: normalizátor rolí automaticky spojuje systémové zprávy do uživatelských zpráv pro nekompatibilní modely
+### Common format issues
-- Role**`vývojáře` nebyla rozpoznána**— Opraveno ve verzi 1.1.0: automaticky převedeno na `systém` pro poskytovatele mimo OpenAI -**`json_schema` nefunguje s Gemini**– Opraveno ve verzi 1.1.0: `response_format` je nyní převeden na Gemini `responseMimeType` + `responseSchema`---
+- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting
+- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode
+- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output
+- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures
+- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models
+- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers
+- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema`
+
+---
## Resilience Settings
### Auto rate-limit not triggering
-– Automatický limit sazby se vztahuje pouze na poskytovatele klíčů API (nikoli OAuth/předplatné)
+- Auto rate-limit only applies to API key providers (not OAuth/subscription)
+- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled
+- Check if the provider returns `429` status codes or `Retry-After` headers
-- Ověřte, zda je v**Nastavení → Odolnost → Profily poskytovatelů**povolen automatický limit rychlosti
-- Zkontrolujte, zda poskytovatel vrací stavové kódy `429` nebo záhlaví `Retry-After`### Tuning exponential backoff
+### Tuning exponential backoff
-Profily poskytovatelů podporují tato nastavení:
+Provider profiles support these settings:
--**Základní zpoždění**— Počáteční doba čekání po prvním selhání (výchozí: 1s)
-–**Max. zpoždění**– Maximální doba čekání (výchozí: 30 s) -**Multiplikátor**– o kolik se má prodloužit zpoždění při po sobě jdoucím selhání (výchozí: 2x)### Anti-thundering herd
+- **Base delay** — Initial wait time after first failure (default: 1s)
+- **Max delay** — Maximum wait time cap (default: 30s)
+- **Multiplier** — How much to increase delay per consecutive failure (default: 2x)
-Když mnoho souběžných požadavků zasáhne poskytovatele s omezenou rychlostí, OmniRoute použije mutex + automatické omezování rychlosti k serializaci požadavků a prevenci kaskádových selhání. To je automatické pro poskytovatele klíčů API.---
+### Anti-thundering herd
+
+When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers.
+
+---
## Optional RAG / LLM failure taxonomy (16 problems)
-Někteří uživatelé OmniRoute umístí bránu před RAG nebo zásobníky agentů. V těchto nastaveních je běžné vidět podivný vzorec: OmniRoute vypadá zdravě (poskytovatelé jsou v pořádku, směrovací profily jsou v pořádku, žádná upozornění na omezení rychlosti), ale konečná odpověď je stále špatná.
+Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong.
-V praxi tyto incidenty obvykle pocházejí z navazujícího potrubí RAG, nikoli ze samotné brány.
+In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself.
-Pokud chcete sdílený slovník pro popis těchto selhání, můžete použít WFGY ProblemMap, externí textový zdroj licence MIT, který definuje šestnáct opakujících se vzorců selhání RAG / LLM. Na vysoké úrovni pokrývá:
+If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers:
-- posun vyhledávání a porušené hranice kontextu
-- prázdné nebo zastaralé indexy a vektorová úložiště
-- vkládání versus sémantický nesoulad
-- rychlé sestavení a problémy s kontextovým oknem
-- logický kolaps a příliš sebevědomé odpovědi
-- selhání koordinace dlouhých řetězců a agentů
-- multiagentní paměť a posun rolí
-- problémy s nasazením a objednáním bootstrapu
+- retrieval drift and broken context boundaries
+- empty or stale indexes and vector stores
+- embedding versus semantic mismatch
+- prompt assembly and context window issues
+- logic collapse and overconfident answers
+- long chain and agent coordination failures
+- multi agent memory and role drift
+- deployment and bootstrap ordering problems
-Myšlenka je jednoduchá:
+The idea is simple:
-1. Když prozkoumáte špatnou odpověď, zachyťte:
- - uživatelský úkol a požadavek
- - kombinace trasy nebo poskytovatele v OmniRoute
- - jakýkoli kontext RAG použitý po proudu (načtené dokumenty, volání nástrojů atd.)
-2. Namapujte incident na jedno nebo dvě čísla WFGY ProblemMap (`č.1` … `č.16`).
-3. Uložte číslo na svůj vlastní řídicí panel, runbook nebo sledovač incidentů vedle protokolů OmniRoute.
-4. Použijte příslušnou stránku WFGY k rozhodnutí, zda potřebujete změnit strategii zásobníku RAG, retrieveru nebo směrování.
+1. When you investigate a bad response, capture:
+ - user task and request
+ - route or provider combo in OmniRoute
+ - any RAG context used downstream (retrieved documents, tool calls, etc)
+2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`).
+3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs.
+4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy.
-Celý text a konkrétní recepty jsou k dispozici zde (licence MIT, pouze text):
+Full text and concrete recipes live here (MIT license, text only):
-[SOUBOR WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md)
+[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md)
-Tuto sekci můžete ignorovat, pokud za OmniRoute nespouštíte RAG nebo agenty.---
+You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute.
+
+---
## Still Stuck?
-–**Problémy s GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architecture**: Interní podrobnosti viz [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) -**Reference API**: Všechny koncové body viz [`docs/API_REFERENCE.md`](API_REFERENCE.md) -**Health Dashboard**: Zkontrolujte**Dashboard → Health**pro stav systému v reálném čase -**Translator**: K ladění problémů s formátem použijte**Dashboard → Translator**
+- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details
+- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints
+- **Health Dashboard**: Check **Dashboard → Health** for real-time system status
+- **Translator**: Use **Dashboard → Translator** to debug format issues
diff --git a/docs/i18n/cs/llm.txt b/docs/i18n/cs/llm.txt
new file mode 100644
index 0000000000..79760c56fd
--- /dev/null
+++ b/docs/i18n/cs/llm.txt
@@ -0,0 +1,476 @@
+# OmniRoute (Čeština)
+
+🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt)
+
+---
+
+
+> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app.
+
+## Přehled
+
+OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free.
+
+**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost.
+
+**Current version:** 3.5.5
+
+## Tech Stack
+
+- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`)
+- **Framework:** Next.js 16 (App Router) with TypeScript 5.9
+- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations)
+- **State management:** Zustand (client), SQLite (server persistence)
+- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons
+- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth
+- **Schemas:** Zod v4 for all API / MCP input validation
+- **Background jobs:** Custom token health check scheduler, 24h model auto-sync
+- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses
+- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine
+- **i18n:** next-intl with 30 languages
+- **Desktop:** Electron (cross-platform: Windows, macOS, Linux)
+- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`)
+
+## Project Structure
+
+```
+/
+├── src/ # Main application source
+│ ├── app/ # Next.js App Router pages and API routes
+│ │ ├── (dashboard)/ # Dashboard UI pages
+│ │ │ └── dashboard/
+│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents)
+│ │ │ ├── analytics/ # Usage analytics and charts
+│ │ │ ├── api-manager/ # API key management
+│ │ │ ├── audit/ # Audit logs
+│ │ │ ├── auto-combo/ # Auto-combo engine dashboard
+│ │ │ ├── cache/ # Cache dashboard (semantic cache stats)
+│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.)
+│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates)
+│ │ │ ├── costs/ # Cost tracking per provider/model
+│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs
+│ │ │ ├── health/ # System health (uptime, circuit breakers, latency)
+│ │ │ ├── limits/ # Rate limits dashboard
+│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed)
+│ │ │ ├── media/ # Image/video/music generation + transcription
+│ │ │ ├── memory/ # Memory system dashboard
+│ │ │ ├── onboarding/ # Onboarding wizard
+│ │ │ ├── playground/ # Model playground (Monaco editor, streaming)
+│ │ │ ├── providers/ # Provider management (OAuth + API key + free)
+│ │ │ ├── search-tools/ # Search tools configuration
+│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced)
+│ │ │ ├── skills/ # Skills system dashboard
+│ │ │ ├── translator/ # Format translator + debug tools
+│ │ │ └── usage/ # Usage history
+│ │ ├── api/ # REST API endpoints (51 route directories)
+│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings,
+│ │ │ │ # images, audio, videos, music, moderations, rerank, search,
+│ │ │ │ # responses, messages, registered-keys, quotas, accounts)
+│ │ │ ├── v1beta/ # Gemini-compatible API
+│ │ │ ├── a2a/ # A2A agent management API
+│ │ │ ├── acp/ # ACP agent management API
+│ │ │ ├── oauth/ # OAuth flows per provider
+│ │ │ ├── providers/ # Provider CRUD and batch testing
+│ │ │ ├── models/ # Dashboard model listing and aliases
+│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains)
+│ │ │ ├── memory/ # Memory system API
+│ │ │ ├── skills/ # Skills system API
+│ │ │ ├── evals/ # Eval runner API
+│ │ │ ├── mcp/ # MCP HTTP transport API
+│ │ │ ├── search/ # Search provider API
+│ │ │ ├── webhooks/ # Webhook management
+│ │ │ ├── tunnels/ # Cloudflare tunnel management
+│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.)
+│ │ ├── landing/ # Landing page
+│ │ ├── login/ # Login page
+│ │ ├── forgot-password/ # Password recovery
+│ │ ├── status/ # Status page
+│ │ └── docs/ # In-app documentation
+│ ├── domain/ # Domain types and policy engine
+│ │ ├── policyEngine.ts # Central policy engine
+│ │ ├── comboResolver.ts # Combo resolution logic
+│ │ ├── costRules.ts # Cost calculation rules
+│ │ ├── degradation.ts # Graceful degradation
+│ │ ├── fallbackPolicy.ts # Fallback behavior
+│ │ ├── lockoutPolicy.ts # Account lockout logic
+│ │ ├── modelAvailability.ts # Model availability checks
+│ │ ├── providerExpiration.ts # Provider credential expiration
+│ │ ├── quotaCache.ts # Quota caching layer
+│ │ ├── configAudit.ts # Configuration auditing
+│ │ └── responses.ts # Domain response types
+│ ├── i18n/ # Internationalization
+│ │ └── messages/ # 30 language JSON files
+│ ├── lib/ # Core libraries
+│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server
+│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting)
+│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup
+│ │ │ └── streaming.ts # SSE streaming for A2A
+│ │ ├── acp/ # Agent Communication Protocol registry and manager
+│ │ ├── compliance/ # Compliance policy engine
+│ │ ├── db/ # SQLite database layer (21 modules + migrations)
+│ │ │ ├── core.ts # Database initialization, connection, schema
+│ │ │ ├── providers.ts # Provider connection CRUD
+│ │ │ ├── models.ts # Model catalog management
+│ │ │ ├── combos.ts # Combo configuration
+│ │ │ ├── apiKeys.ts # API key management
+│ │ │ ├── settings.ts # Settings persistence
+│ │ │ ├── backup.ts # Database backup/restore
+│ │ │ ├── proxies.ts # Proxy registry
+│ │ │ ├── prompts.ts # Prompt templates
+│ │ │ ├── webhooks.ts # Webhook subscriptions
+│ │ │ ├── detailedLogs.ts # Detailed request logging
+│ │ │ ├── domainState.ts # Domain state persistence
+│ │ │ ├── registeredKeys.ts # Registered API keys with quotas
+│ │ │ ├── quotaSnapshots.ts # Quota snapshot history
+│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings
+│ │ │ ├── cliToolState.ts # CLI tool state tracking
+│ │ │ ├── encryption.ts # Data encryption
+│ │ │ ├── readCache.ts # Read-through cache layer
+│ │ │ ├── secrets.ts # Secrets management
+│ │ │ ├── stateReset.ts # State reset utilities
+│ │ │ ├── migrationRunner.ts # Schema migration runner
+│ │ │ └── migrations/ # 16 SQL migration files
+│ │ ├── evals/ # Eval runner and scheduler
+│ │ ├── memory/ # Persistent conversational memory
+│ │ │ ├── extraction.ts # Memory extraction from conversations
+│ │ │ ├── injection.ts # Memory injection into context
+│ │ │ ├── retrieval.ts # Memory retrieval/search
+│ │ │ ├── store.ts # Memory persistence layer
+│ │ │ └── summarization.ts # Memory summarization
+│ │ ├── oauth/ # OAuth providers, services, and utilities
+│ │ │ ├── constants/ # Default OAuth credentials (overridable via env)
+│ │ │ ├── providers/ # Provider-specific OAuth configs
+│ │ │ ├── services/ # Provider-specific token exchange logic
+│ │ │ └── utils/ # PKCE, callback server, token helpers
+│ │ ├── plugins/ # Plugin system
+│ │ ├── skills/ # Extensible skill framework
+│ │ │ ├── registry.ts # Skill registration
+│ │ │ ├── executor.ts # Skill execution engine
+│ │ │ ├── sandbox.ts # Skill sandbox environment
+│ │ │ ├── builtin/ # Built-in skills
+│ │ │ ├── interception.ts # Skill request interception
+│ │ │ └── injection.ts # Skill context injection
+│ │ ├── usage/ # Usage tracking system
+│ │ │ ├── callLogs.ts # Call log persistence
+│ │ │ ├── costCalculator.ts # Cost calculation engine
+│ │ │ └── usageHistory.ts # Usage history queries
+│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers
+│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management
+│ │ ├── pricingSync.ts # LiteLLM pricing data sync
+│ │ ├── semanticCache.ts # Semantic caching layer
+│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler
+│ │ ├── webhookDispatcher.ts # Webhook event dispatcher
+│ │ └── localDb.ts # Unified re-export layer for all DB modules
+│ ├── middleware/ # Request middleware
+│ │ └── promptInjectionGuard.ts # Prompt injection detection
+│ ├── mitm/ # MITM proxy capability
+│ │ ├── cert/ # Certificate management
+│ │ ├── dns/ # DNS handling
+│ │ ├── targets/ # Target routing
+│ │ └── manager.ts # MITM proxy manager
+│ ├── shared/ # Shared utilities, components, and constants
+│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.)
+│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes
+│ │ ├── contracts/ # Shared API contracts
+│ │ ├── hooks/ # React hooks
+│ │ ├── middleware/ # Shared middleware utilities
+│ │ ├── schemas/ # Shared Zod schemas
+│ │ ├── services/ # Shared services
+│ │ ├── types/ # Shared TypeScript types
+│ │ ├── validation/ # Zod schemas (settings, providers, routes)
+│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID)
+│ ├── sse/ # SSE proxy pipeline
+│ │ ├── services/ # Auth resolution, format translation, response handling
+│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency
+│ ├── store/ # Zustand client-side stores (theme, providers, etc.)
+│ └── types/ # TypeScript type definitions
+├── open-sse/ # Standalone SSE server (npm workspace)
+│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video,
+│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models)
+│ ├── executors/ # Provider-specific request executors (14 executors)
+│ │ ├── base.ts # Base executor with shared logic
+│ │ ├── default.ts # Default OpenAI-compatible executor
+│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum)
+│ │ ├── codex.ts # OpenAI Codex CLI
+│ │ ├── antigravity.ts # Antigravity IDE
+│ │ ├── github.ts # GitHub Copilot
+│ │ ├── gemini-cli.ts # Gemini CLI
+│ │ ├── kiro.ts # Kiro AI
+│ │ ├── qoder.ts # Qoder AI
+│ │ ├── vertex.ts # Vertex AI (Service Account JSON)
+│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI
+│ │ ├── opencode.ts # OpenCode Zen/Go
+│ │ ├── pollinations.ts # Pollinations AI
+│ │ └── puter.ts # Puter AI
+│ ├── handlers/ # Request handlers per API type (11 handlers)
+│ │ ├── chatCore.ts # Main chat completions handler
+│ │ ├── responsesHandler.ts # OpenAI Responses API handler
+│ │ ├── embeddings.ts # Embedding generation
+│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.)
+│ │ ├── videoGeneration.ts # Video generation
+│ │ ├── musicGeneration.ts # Music generation
+│ │ ├── audioSpeech.ts # Text-to-speech
+│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI)
+│ │ ├── moderations.ts # Content moderation
+│ │ ├── rerank.ts # Reranking API
+│ │ └── search.ts # Web search API
+│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP)
+│ │ ├── server.ts # MCP server core (tool registration, scope enforcement)
+│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools)
+│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a)
+│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes)
+│ │ ├── audit.ts # Tool call audit logging
+│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat
+│ │ └── httpTransport.ts # HTTP transport handler
+│ ├── services/ # 36+ service modules
+│ │ ├── combo.ts # Core routing engine
+│ │ ├── usage.ts # Usage tracking
+│ │ ├── tokenRefresh.ts # OAuth token refresh
+│ │ ├── rateLimitManager.ts # Rate limit management
+│ │ ├── accountFallback.ts # Multi-account fallback
+│ │ ├── sessionManager.ts # Session management
+│ │ ├── wildcardRouter.ts # Wildcard model routing
+│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration)
+│ │ ├── intentClassifier.ts # Request intent classification
+│ │ ├── taskAwareRouter.ts # Task-aware routing
+│ │ ├── thinkingBudget.ts # Thinking budget management
+│ │ ├── contextManager.ts # Context window management
+│ │ ├── modelDeprecation.ts # Model deprecation handling
+│ │ ├── modelFamilyFallback.ts # Intra-family model fallback
+│ │ ├── emergencyFallback.ts # Emergency fallback
+│ │ ├── workflowFSM.ts # Workflow state machine
+│ │ ├── backgroundTaskDetector.ts # Background task detection
+│ │ ├── ipFilter.ts # IP-based access control
+│ │ ├── signatureCache.ts # CLI signature caching
+│ │ ├── volumeDetector.ts # Request volume detection
+│ │ ├── contextHandoff.ts # Context relay handoff generation and injection
+│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay
+│ │ └── ... # Additional services (14 more modules)
+│ ├── transformer/ # Responses API transformer
+│ │ └── responsesTransformer.ts
+│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek)
+│ │ ├── request/ # Request translators per provider
+│ │ ├── response/ # Response translators per provider
+│ │ ├── helpers/ # Translation helpers
+│ │ └── image/ # Image format translation
+│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.)
+├── electron/ # Electron desktop app (cross-platform)
+│ ├── main.js # Electron main process
+│ ├── preload.js # Preload script (IPC bridge)
+│ └── assets/ # App icons and assets
+├── tests/ # Test suites
+│ ├── unit/ # 122 unit test files
+│ ├── integration/ # Integration tests
+│ ├── e2e/ # Playwright E2E tests
+│ ├── security/ # Security tests
+│ ├── translator/ # Translator-specific tests
+│ └── load/ # Load tests
+├── docs/ # Documentation
+│ ├── i18n/ # 30-language translated docs
+│ ├── ARCHITECTURE.md # Full architecture documentation
+│ ├── API_REFERENCE.md # API reference
+│ ├── USER_GUIDE.md # User guide
+│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview
+│ ├── CLI-TOOLS.md # CLI tools integration guide
+│ ├── A2A-SERVER.md # A2A agent protocol documentation
+│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring)
+│ ├── MCP-SERVER.md # MCP server (25 tools)
+│ ├── TROUBLESHOOTING.md # Troubleshooting guide
+│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide
+│ ├── openapi.yaml # OpenAPI specification
+│ └── screenshots/ # Dashboard screenshots
+├── bin/ # CLI entry points (omniroute, reset-password)
+├── scripts/ # Build and utility scripts
+└── .env.example # Environment variable template
+```
+
+## Key Features (v3.5.5)
+
+### Core Proxy
+- **60+ AI providers** with automatic format translation
+- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible)
+- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay
+- **4-tier fallback**: Subscription → API Key → Cheap → Free
+- **Context Relay strategy**: Session handoff summaries on account rotation for continuity
+- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown
+- **Semantic caching** with cache hit/miss headers
+- **Idempotency** with configurable dedup window
+- **Circuit breaker** per provider with configurable thresholds
+- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback
+- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers
+- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement
+- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization
+- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills
+- **Prompt Injection Guard**: Middleware-level prompt injection detection
+- **MITM Proxy**: Certificate management, DNS handling, and target routing
+- **Cloudflare Tunnels**: Managed tunnel creation for remote access
+- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches)
+
+### Bezpečnost
+- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs)
+- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()`
+- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()`
+- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection
+- **CLI Fingerprint Matching** — Per-provider request signature matching
+- **Prompt injection guard** — Request middleware detection
+- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`)
+- **PII sanitizer** — Sensitive data scrubbing in logs
+
+### Dashboard Pages (23 sections)
+- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons
+- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies
+- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics
+- **Analytics** — Token consumption, cost, heatmaps, distributions
+- **Health** — Uptime, memory, latency percentiles, circuit breakers
+- **Logs** — Request, Proxy, Audit, Console (tabbed)
+- **Audit** — Audit trail and compliance logging
+- **Costs** — Cost tracking per provider/model
+- **Limits** — Rate limit monitoring
+- **Cache** — Semantic cache statistics and management
+- **CLI Tools** — One-click configuration for 10+ AI CLI tools
+- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration
+- **Playground** — Test any model with Monaco editor, streaming responses
+- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files)
+- **Search Tools** — Search provider configuration and testing
+- **Memory** — Memory system management and visualization
+- **Skills** — Skills framework management and execution
+- **Translator** — Format debugging: playground, chat tester, test bench, live monitor
+- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced
+- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed)
+- **Onboarding** — Setup wizard for new users
+- **Usage** — Usage history and analytics
+- **API Manager** — API key management with scoped permissions
+
+### Protocol Support
+- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations`
+- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens`
+- **OpenAI Responses** — `/v1/responses`
+- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}`
+- **Ollama** — `/v1/api/chat`, `/api/tags`
+- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily)
+- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP)
+- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills)
+- **ACP** — Agent Communication Protocol registry and manager
+
+### MCP Server (25 Tools)
+| Category | Tools |
+|-----------|-------|
+| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` |
+| Memory (3) | `memory_search`, `memory_add`, `memory_clear` |
+| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` |
+
+**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience`
+
+### Provider Categories
+
+**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI
+
+**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline
+
+**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan
+
+**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs
+
+### Internationalization
+- 30 languages for UI (all dashboard pages)
+- 30 translated documentation sets in docs/i18n/
+- Language switcher in documentation
+
+## Key Architectural Decisions
+
+1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints.
+
+2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently.
+
+3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation.
+
+4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity.
+
+5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client.
+
+6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes.
+
+7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`).
+
+8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages.
+
+9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations.
+
+10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`.
+
+11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation.
+
+12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline.
+
+13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast.
+
+## Main Flows
+
+### Proxy Request Flow
+1. Client sends OpenAI-format request to `/v1/chat/completions`
+2. API key validation
+3. Model resolution: direct model or combo lookup
+4. For combos: iterate through models with selected strategy
+5. Auth resolution: get credentials for the target provider
+6. Format translation: OpenAI → provider native format
+7. CLI fingerprint matching (if enabled for provider)
+8. Upstream request with circuit breaker and rate limiting
+9. Response translation: provider → OpenAI format
+10. omniModel tag sanitization (strip internal tags)
+11. SSE streaming back to client
+12. Memory extraction (if memory system enabled)
+13. Usage logging and cost calculation
+
+### OAuth Flow
+1. Dashboard initiates `/api/oauth/[provider]/authorize`
+2. User completes OAuth login in browser
+3. Callback hits `/api/oauth/[provider]/exchange`
+4. Tokens stored as a provider connection in SQLite
+5. Background job refreshes tokens before expiry
+
+## Important Notes for LLMs
+
+1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only).
+
+2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`).
+
+3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services.
+
+4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`.
+
+5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module.
+
+6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`).
+
+7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes.
+
+8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB.
+
+9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown.
+
+10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130).
+
+11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux).
+
+12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint.
+
+13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API.
+
+14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API.
+
+15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time.
+
+16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`.
+
+17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22.
+
+18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false.
+
+## Links
+
+- Repository: https://github.com/diegosouzapw/OmniRoute
+- Website: https://omniroute.online
+- npm: https://www.npmjs.com/package/omniroute
+- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute
+- Documentation: See `/docs/` directory
diff --git a/docs/i18n/da/README.md b/docs/i18n/da/README.md
index 07c3e3ee50..06a0559457 100644
--- a/docs/i18n/da/README.md
+++ b/docs/i18n/da/README.md
@@ -4,11 +4,14 @@
---
+
### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback.
-_Din universelle API-proxy — ét slutpunkt, 60+ udbydere, ingen nedetid. Nu med**MCP Server (25 værktøjer)**,**A2A Protocol**,**Memory/Skills Systems**&**Electron Desktop App**._
+_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._
-**Chatafslutninger • Indlejringer • Billedgenerering • Video • Musik • Lyd • Genrangering •**Websøgning**• MCP-server • A2A-protokol • 100 % TypeScript**---
+**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript**
+
+---
@@ -39,9 +42,13 @@ _Din universelle API-proxy — ét slutpunkt, 60+ udbydere, ingen nedetid. Nu me
[](https://omniroute.online)
[](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-[🌐 Hjemmeside](https://omniroute.online) • [🚀 Lynstart](#-hurtig-start) • [💡 Funktioner](#-nøglefunktioner) • [📖 Docs](#-dokumentation) • [💰 Priser](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-🌐**Tilgængelig på:**🇺🇸 [engelsk](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Tysk](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [English](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesien](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [filippinsk](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)---
+
+
+🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)
+
+---
## 🖼️ Main Dashboard
@@ -53,28 +60,30 @@ _Din universelle API-proxy — ét slutpunkt, 60+ udbydere, ingen nedetid. Nu me
## 📸 Dashboard Preview
-
-Klik for at se skærmbilleder af dashboard
+
+Click to see dashboard screenshots
-| Side | Skærmbillede |
-| ----------------- | -------------------------------------------------- | ---------- |
-| **Udbydere** |  |
-| **Komboer** |  |
-| **Analyse** |  |
-| **Sundhed** |  |
-| **Oversætter** |  |
-| **Indstillinger** |  |
-| **CLI-værktøjer** |  |
-| **Brugslogfiler** |  |
-| **Endpunkter** |  | |
+| Page | Screenshot |
+| -------------- | ------------------------------------------------- |
+| **Providers** |  |
+| **Combos** |  |
+| **Analytics** |  |
+| **Health** |  |
+| **Translator** |  |
+| **Settings** |  |
+| **CLI Tools** |  |
+| **Usage Logs** |  |
+| **Endpoints** |  |
+
+
---
### 🤖 Free AI Provider for your favorite coding agents
-_Tilslut ethvert AI-drevet IDE- eller CLI-værktøj gennem OmniRoute - gratis API-gateway til ubegrænset kodning._
+_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._
-
+
@@ -88,28 +97,28 @@ _Tilslut ethvert AI-drevet IDE- eller CLI-værktøj gennem OmniRoute - gratis AP

NanoBot
- ⭐ 20,9K
+ ⭐ 20.9K
|

PicoClaw
- ⭐ 14,6K
+ ⭐ 14.6K
|

ZeroClaw
- ⭐ 9,9K
+ ⭐ 9.9K
|

IronClaw
- ⭐ 2,1K
+ ⭐ 2.1K
|
@@ -125,481 +134,555 @@ _Tilslut ethvert AI-drevet IDE- eller CLI-værktøj gennem OmniRoute - gratis AP

Codex CLI
- ⭐ 60,8K
+ ⭐ 60.8K

Claude Code
- ⭐ 67,3K
+ ⭐ 67.3K
|

Gemini CLI
- ⭐ 94,7K
+ ⭐ 94.7K
|
- 
- Kilokode
+ 
+ Kilo Code
- ⭐ 15,5K
+ ⭐ 15.5K
|
-📡 Alle agenter forbinder via http://localhost:20128/v1 eller http://cloud.omniroute.online/v1 - én konfiguration, ubegrænset modeller og kvote---
+📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota
+
+---
## 🤔 Why OmniRoute?
-**Stop med at spilde penge og nå grænser:**
+**Stop wasting money and hitting limits:**
--
Abonnementskvoten udløber ubrugt hver måned
--
Satsgrænser stopper dig med at midtkode
--
Dyre API'er ($20-50/måned pr. udbyder)
--
Manuel skift mellem udbydere
+-
Subscription quota expires unused every month
+-
Rate limits stop you mid-coding
+-
Expensive APIs ($20-50/month per provider)
+-
Manual switching between providers
-**OmniRoute løser dette:**
+**OmniRoute solves this:**
-- ✅**Maksimer abonnementer**- Spor kvote, brug hver bit før nulstilling
-- ✅**Automatisk fallback**- Abonnement → API-nøgle → Billig → Gratis, ingen nedetid
-- ✅**Multi-konto**- Round-robin mellem konti pr. udbyder
-- ✅**Universal**- Virker med Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, ethvert CLI-værktøj---
+- ✅ **Maximize subscriptions** - Track quota, use every bit before reset
+- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime
+- ✅ **Multi-account** - Round-robin between accounts per provider
+- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool
+
+---
## 📧 Support
-> 💬**Tilmeld dig vores fællesskab!**[WhatsApp-gruppe](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Få hjælp, del tips, og hold dig opdateret.
+> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated.
--**Websted**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problemer**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Fællesskabsgruppe](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Bidrager**: Se [CONTRIBUTING.md](CONTRIBUTING.md), åbn en PR, eller vælg et "godt første nummer" -**Originalt projekt**: [9router af decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug?
+- **Website**: [omniroute.online](https://omniroute.online)
+- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute)
+- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue`
+- **Original Project**: [9router by decolua](https://github.com/decolua/9router)
-Når du åbner et problem, skal du køre kommandoen systeminfo og vedhæfte den genererede fil:```bash
+### 🐛 Reporting a Bug?
+
+When opening an issue, please run the system-info command and attach the generated file:
+
+```bash
npm run system-info
-
```
-Dette genererer en `system-info.txt` med din Node.js-version, OmniRoute-version, OS-detaljer, installerede CLI-værktøjer (qoder, gemini, claude, codex, antigravity, droid osv.), Docker/PM2-status og systempakker - alt hvad vi har brug for for hurtigt at reproducere dit problem. Vedhæft filen direkte til dit GitHub-problem.---
+This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue.
+
+---
## 🔄 How It Works
```
-
┌─────────────┐
-│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
-│ Tool │
+│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
+│ Tool │
└──────┬──────┘
-│ http://localhost:20128/v1
-↓
+ │ http://localhost:20128/v1
+ ↓
┌─────────────────────────────────────────┐
-│ OmniRoute (Smart Router) │
-│ • Format translation (OpenAI ↔ Claude) │
-│ • Quota tracking + Embeddings + Images │
-│ • Auto token refresh │
+│ OmniRoute (Smart Router) │
+│ • Format translation (OpenAI ↔ Claude) │
+│ • Quota tracking + Embeddings + Images │
+│ • Auto token refresh │
└──────┬──────────────────────────────────┘
-│
-├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
-│ ↓ quota exhausted
-├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
-│ ↓ budget limit
-├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
-│ ↓ budget limit
-└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
+ │
+ ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
+ │ ↓ quota exhausted
+ ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
+ │ ↓ budget limit
+ ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
+ │ ↓ budget limit
+ └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
Result: Never stop coding, minimal cost
-
-````
+```
---
## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases
->**Alle udviklere, der bruger AI-værktøjer, står over for disse problemer dagligt.**OmniRoute blev bygget til at løse dem alle - fra omkostningsoverskridelser til regionale blokke, fra ødelagte OAuth-flows til protokoloperationer og observerbarhed i virksomheden.
+> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability.
-
-💸 1. "Jeg betaler for et dyrt abonnement, men bliver stadig afbrudt af grænser"
+
+💸 1. "I pay for an expensive subscription but still get interrupted by limits"
-Udviklere betaler $20-200/måned for Claude Pro, Codex Pro eller GitHub Copilot. Selv ved betaling har kvoten et loft - 5 timers brug, ugentlige grænser eller satsgrænser pr. minut. Mid-coding session, udbyderen holder op med at svare, og udvikleren mister flow og produktivitet.
+Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity.
-**Sådan løser OmniRoute det:**
+**How OmniRoute solves it:**
--**Smart 4-Tier Fallback**— Hvis abonnementskvoten løber ud, omdirigeres automatisk til API Key → Billig → Gratis uden manuel indgriben
--**Sporing af udbydergrænser**— Cachelagrede kvote-øjebliksbilleder opdateres på en server-sideplan (standard `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) med manuel opdatering tilgængelig i brugergrænsefladen
--**Multi-Account Support**- Flere konti pr. udbyder med automatisk round-robin - når den ene løber tør, skifter til den næste
--**Brugerdefinerede kombinationer**— Tilpasselige fallback-kæder med 9 balanceringsstrategier (prioritet, vægtet, fill-first, round-robin, P2C, tilfældig, mindst brugt, omkostningsoptimeret, strengt tilfældig)
--**Codex Business Quotas**— Business/Team Workspace kvoteovervågning direkte i dashboardet
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention
+- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI
+- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next
+- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**)
+- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard
-
-🔌 2. "Jeg skal bruge flere udbydere, men hver har en anden API"
+
-OpenAI bruger et format, Claude (Antropisk) bruger et andet, Gemini endnu et andet. Hvis en udvikler ønsker at teste modeller fra forskellige udbydere eller fallback mellem dem, skal de omkonfigurere SDK'er, ændre slutpunkter, håndtere inkompatible formater. Tilpassede udbydere (FriendLI, NIM) har ikke-standardmodelslutpunkter.
+
+🔌 2. "I need to use multiple providers but each has a different API"
-**Sådan løser OmniRoute det:**
+OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints.
--**Unified Endpoint**— En enkelt `http://localhost:20128/v1` fungerer som proxy for alle 60+ udbydere
--**Formatoversættelse**— Automatisk og gennemsigtig: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
--**Response Sanitization**- Fjerner ikke-standardfelter (`x_groq`, `usage_breakdown`, `service_tier`), der bryder OpenAI SDK v1.83+
--**Rollenormalisering**— Konverterer `udvikler` → `system` for ikke-OpenAI-udbydere; `system` → `bruger` til GLM/ERNIE
--**Think Tag Extraction**— Udtrækker ""-blokke fra modeller som DeepSeek R1 til standardiseret "reasoning_content"
--**Structured Output for Gemini**— `json_schema` → `responseMimeType`/`responseSchema` automatisk konvertering
--**`stream` er standard til "false"**- Justerer med OpenAI-specifikationer, undgår uventede SSE i Python/Rust/Go SDK'er
+**How OmniRoute solves it:**
-
-🌐 3. "Min AI-udbyder blokerer mit område/land"
+- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers
+- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
+- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+
+- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE
+- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content`
+- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion
+- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs
-Udbydere som OpenAI/Codex blokerer adgang fra visse geografiske områder. Brugere får fejl som "unsupported_country_region_territory" under OAuth- og API-forbindelser. Dette er især frustrerende for udviklere fra udviklingslande.
+
-**Sådan løser OmniRoute det:**
+
+🌐 3. "My AI provider blocks my region/country"
--**3-Level Proxy Config**— Konfigurerbar proxy på 3 niveauer: global (al trafik), pr. udbyder (kun én udbyder) og pr. forbindelse/nøgle
--**Farvekodede proxy-badges**— Visuelle indikatorer: 🟢 global proxy, 🟡 udbyder proxy, 🔵 forbindelsesproxy, viser altid IP'en
--**OAuth-tokenudveksling gennem proxy**- OAuth-flowet går også gennem proxyen og løser "unsupported_country_region_territory".
--**Forbindelsestest via proxy**— Forbindelsestest bruger den konfigurerede proxy (ikke mere direkte omgåelse)
--**SOCKS5-understøttelse**— Fuld SOCKS5-proxy-understøttelse til udgående routing
--**TLS Fingerprint Spoofing**— Browserlignende TLS-fingeraftryk via 'wreq-js' for at omgå botdetektion
--**🔏 Matching af CLI-fingeraftryk**— Omarrangerer overskrifter og kropsfelter, så de matcher native CLI-binære signaturer, hvilket drastisk reducerer risikoen for kontoflaggning. Proxy-IP'en bevares - du får både stealth**og**IP-maskering samtidigt
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries.
-
-🆓 4. "Jeg vil bruge AI til kodning, men jeg har ingen penge"
+**How OmniRoute solves it:**
-Ikke alle kan betale $20-200/måned for AI-abonnementer. Studerende, udviklere fra vækstlande, hobbyfolk og freelancere har brug for adgang til kvalitetsmodeller uden omkostninger.
+- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key
+- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP
+- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory`
+- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass)
+- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection
+- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously
-**Sådan løser OmniRoute det:**
+
--**Free Tier Providers Indbygget**— Indbygget understøttelse af 100 % gratis udbydere: Qoder (5 ubegrænsede modeller via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited-modeller:-r-modeller:-r qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID gratis), Gemini CLI (180K tokens/måned gratis)
--**Ollama Cloud**— Cloud-hostede Ollama-modeller på `api.ollama.com` med gratis "Light usage"-niveau; brug `ollamacloud/` præfiks
--**Kun gratis kombinationer**— Kæde `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/måned uden nedetid
--**NVIDIA NIM Free Access**— ~40 RPM dev-forever gratis adgang til 70+ modeller på build.nvidia.com (overgang fra kreditter til rene hastighedsgrænser)
--**Cost Optimized Strategy**— Routingstrategi, der automatisk vælger den billigste tilgængelige udbyder
+
+🆓 4. "I want to use AI for coding but I have no money"
-
-🔒 5. "Jeg skal beskytte min AI-gateway mod uautoriseret adgang"
+Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost.
-Når en AI-gateway eksponeres for netværket (LAN, VPS, Docker), kan enhver med adressen forbruge udviklerens tokens/kvote. Uden beskyttelse er API'er sårbare over for misbrug, hurtig injektion og misbrug.
+**How OmniRoute solves it:**
-**Sådan løser OmniRoute det:**
+- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free)
+- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix
+- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime
+- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits)
+- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider
--**API Key Management**— Generering, rotation og scoping pr. udbyder med en dedikeret `/dashboard/api-manager`-side
--**Tilladelser på modelniveau**— Begræns API-nøgler til specifikke modeller ('openai/*', jokertegnsmønstre) med Tillad alt/Begræns-skift
--**API Endpoint Protection**— Kræv en nøgle til `/v1/modeller` og bloker specifikke udbydere fra listen
--**Auth Guard + CSRF Protection**— Alle dashboard-ruter beskyttet med 'withAuth' middleware + CSRF-tokens
--**Rate Limiter**— Per-IP hastighedsbegrænsning med konfigurerbare vinduer
--**IP-filtrering**— Tilladelsesliste/blokeringsliste til adgangskontrol
--**Prompt Injection Guard**— Sanering mod ondsindede promptmønstre
--**AES-256-GCM-kryptering**— Legitimationsoplysninger krypteret i hvile
+
-
-🛑 6. "Min udbyder gik ned, og jeg mistede mit kodningsflow"
+
+🔒 5. "I need to protect my AI gateway from unauthorized access"
-AI-udbydere kan blive ustabile, returnere 5xx-fejl eller ramme midlertidige hastighedsgrænser. Hvis en udvikler afhænger af en enkelt udbyder, bliver de afbrudt. Uden strømafbrydere kan gentagne genforsøg crashe programmet.
+When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse.
-**Sådan løser OmniRoute det:**
+**How OmniRoute solves it:**
--**Circuit Breaker pr. model**— Automatisk åbning/lukning med konfigurerbare tærskler og nedkøling (Lukket/Åben/Halv-Åben), omfang pr. model for at undgå kaskadeblokke
--**Eksponentiel backoff**— Progressive forsinkelser af genforsøg
--**Anti-tordenbesætning**— Mutex + semaforbeskyttelse mod samtidige genforsøgsstorme
--**Combo Fallback Chains**— Hvis den primære udbyder fejler, falder den automatisk gennem kæden uden indgriben
--**Combo Circuit Breaker**- Deaktiverer automatisk fejlende udbydere i en kombinationskæde
--**Health Dashboard**— Oppetidsovervågning, strømafbrydertilstande, lockouts, cachestatistik, p50/p95/p99 latency
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page
+- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle
+- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing
+- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens
+- **Rate Limiter** — Per-IP rate limiting with configurable windows
+- **IP Filtering** — Allowlist/blocklist for access control
+- **Prompt Injection Guard** — Sanitization against malicious prompt patterns
+- **AES-256-GCM Encryption** — Credentials encrypted at rest
-
-🔧 7. "Konfiguration af hvert AI-værktøj er trættende og gentagende"
+
-Udviklere bruger Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Hvert værktøj har brug for en anden konfiguration (API-endepunkt, nøgle, model). At omkonfigurere, når du skifter udbyder eller model, er spild af tid.
+
+🛑 6. "My provider went down and I lost my coding flow"
-**Sådan løser OmniRoute det:**
+AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application.
--**CLI Tools Dashboard**— Dedikeret side med et-klik opsætning til Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
--**GitHub Copilot Config Generator**— Genererer `chatLanguageModels.json` til VS-kode med bulk modelvalg
--**Onboarding Wizard**— Guidet 4-trins opsætning for førstegangsbrugere
--**Et slutpunkt, alle modeller**— Konfigurer `http://localhost:20128/v1` én gang, få adgang til 60+ udbydere
+**How OmniRoute solves it:**
-
-🔑 8. "Administration af OAuth-tokens fra flere udbydere er et helvede"
+- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks
+- **Exponential Backoff** — Progressive retry delays
+- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms
+- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention
+- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain
+- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency
-Claude Code, Codex, Gemini CLI, Copilot - alle bruger OAuth 2.0 med udløbende tokens. Udviklere skal genautentificere konstant, håndtere `client_secret is missing`, `redirect_uri_mismatch` og fejl på fjernservere. OAuth på LAN/VPS er særligt problematisk.
+
-**Sådan løser OmniRoute det:**
+
+🔧 7. "Configuring each AI tool is tedious and repetitive"
--**Automatisk tokenopdatering**— OAuth-tokens opdateres i baggrunden før udløb
--**OAuth 2.0 (PKCE) Indbygget**— Automatisk flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
--**Multi-Account OAuth**— Flere konti pr. udbyder via JWT/ID-tokenudtrækning
--**OAuth LAN/Remote Fix**— Privat IP-detektion for `redirect_uri` + manuel URL-tilstand for fjernservere
--**OAuth Behind Nginx**— Bruger `window.location.origin` til omvendt proxy-kompatibilitet
--**Remote OAuth Guide**— Trin-for-trin guide til Google Cloud-legitimationsoplysninger på VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time.
-
-📊 9. "Jeg ved ikke, hvor meget jeg bruger eller hvor"
+**How OmniRoute solves it:**
-Udviklere bruger flere betalte udbydere, men har ikke noget samlet syn på udgifter. Hver udbyder har sit eget faktureringsdashboard, men der er ingen konsolideret visning. Uventede omkostninger kan hobe sig op.
+- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
+- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection
+- **Onboarding Wizard** — Guided 4-step setup for first-time users
+- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers
-**Sådan løser OmniRoute det:**
+
--**Dashboard for omkostningsanalyse**— omkostningssporing pr. token og budgetstyring pr. udbyder
--**Budgetgrænser pr. niveau**— Udgiftsloft pr. niveau, der udløser automatisk fallback
--**Priskonfiguration pr. model**— Konfigurerbare priser pr. model
--**Brugsstatistik pr. API-nøgle**— Antal anmodninger og sidst anvendte tidsstempel pr. nøgle
--**Analytics Dashboard**— Statiske kort, modelbrugsdiagram, udbydertabel med succesrater og latens
+
+🔑 8. "Managing OAuth tokens from multiple providers is hell"
-
-🐛 10. "Jeg kan ikke diagnosticere fejl og problemer i AI-kald"
+Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic.
-Når et opkald mislykkes, ved udvikleren ikke, om det var en takstgrænse, udløbet token, forkert format eller udbyderfejl. Fragmenterede logfiler på tværs af forskellige terminaler. Uden observerbarhed er fejlfinding trial-and-error.
+**How OmniRoute solves it:**
-**Sådan løser OmniRoute det:**
+- **Auto Token Refresh** — OAuth tokens refresh in background before expiration
+- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
+- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction
+- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers
+- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility
+- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker
--**Unified Logs Dashboard**— 4 faner: Request Logs, Proxy Logs, Audit Logs, Console
--**Console Log Viewer**— Realtidsterminal-fremviser med farvekodede niveauer, automatisk rulning, søg, filtrer
--**SQLite Proxy Logs**— Vedvarende logfiler, der overlever servergenstarter
--**Oversætterlegeplads**— 4 fejlfindingstilstande: Legeplads (formatoversættelse), Chattester (rundtur), Testbænk (batch), Live Monitor (realtid)
--**Request Telemetri**— p50/p95/p99 latency + X-Request-Id-sporing
--**Filbaseret logning med rotation**— Applogfiler roterer efter størrelse, opbevaringsdage og arkivantal; opkaldslog-artefakter roterer efter opbevaringsdage og filantal
--**System Info Report**— `npm run system-info` genererer `system-info.txt` med dit fulde miljø (Nodeversion, OmniRoute-version, OS, CLI-værktøjer, Docker/PM2-status). Vedhæft det, når du rapporterer problemer til øjeblikkelig triage.
+
-
-🏗️ 11. "Deployering og vedligeholdelse af gatewayen er kompleks"
+
+📊 9. "I don't know how much I'm spending or where"
-Installation, konfiguration og vedligeholdelse af en AI-proxy på tværs af forskellige miljøer (lokalt, VPS, Docker, cloud) er arbejdskrævende. Problemer som hårdkodede stier, "EACCES" på mapper, portkonflikter og cross-platform builds tilføjer friktion.
+Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up.
-**Sådan løser OmniRoute det:**
+**How OmniRoute solves it:**
--**npm global installation**— `npm install -g omniroute && omniroute` — færdig
--**Docker Multi-Platform**— AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi)
--**Docker Compose Profiles**— 'base' (ingen CLI-værktøjer) og 'cli' (med Claude Code, Codex, OpenClaw)
--**Electron Desktop App**— Indbygget app til Windows/macOS/Linux med systembakke, autostart, offlinetilstand
--**Split-Port Mode**— API og Dashboard på separate porte til avancerede scenarier (omvendt proxy, containernetværk)
--**Cloud Sync**— Konfigurer synkronisering på tværs af enheder via Cloudflare Workers
--**DB Backups**— Automatisk backup, gendannelse, eksport og import af alle indstillinger med `DISABLE_SQLITE_AUTO_BACKUP` til eksternt administrerede sikkerhedskopier
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider
+- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback
+- **Per-Model Pricing Configuration** — Configurable prices per model
+- **Usage Statistics Per API Key** — Request count and last-used timestamp per key
+- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency
-
-🌍 12. "Grænsefladen er kun engelsk, og mit team taler ikke engelsk"
+
-Hold i ikke-engelsktalende lande, især i Latinamerika, Asien og Europa, kæmper med grænseflader, der kun er på engelsk. Sprogbarrierer reducerer adoption og øger konfigurationsfejl.
+
+🐛 10. "I can't diagnose errors and problems in AI calls"
-**Sådan løser OmniRoute det:**
+When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error.
--**Dashboard i18n — 30 sprog**— Alle 500+ taster oversat, inklusive arabisk, bulgarsk, dansk, tysk, spansk, finsk, fransk, hebraisk, hindi, ungarsk, indonesisk, italiensk, japansk, koreansk, malaysisk, hollandsk, norsk, polsk, portugisisk (PT/BR), rumænsk, russisk, ukrainsk, kinesisk, ukrainsk, kinesisk, kinesisk, ukrainsk, kinesisk, ukrainsk, kinesisk, ukrainsk, svensk, Vietnam, Vietnam
--**RTL-understøttelse**— Højre-til-venstre-understøttelse for arabisk og hebraisk
--**Multi-Language READMEs**— 30 komplette dokumentationsoversættelser
--**Sprogvælger**— Globusikon i overskriften til skift i realtid
+**How OmniRoute solves it:**
-
-🔄 13. "Jeg har brug for mere end chat – jeg har brug for indlejringer, billeder, lyd"
+- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console
+- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter
+- **SQLite Proxy Logs** — Persistent logs that survive server restarts
+- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time)
+- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing
+- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count
+- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage.
-AI er ikke bare fuldførelse af chat. Udviklere skal generere billeder, transskribere lyd, oprette indlejringer til RAG, omrangere dokumenter og moderere indhold. Hver API har et andet slutpunkt og format.
+
-**Sådan løser OmniRoute det:**
+
+🏗️ 11. "Deploying and maintaining the gateway is complex"
--**Embeddings**— `/v1/embeddings` med 6 udbydere og 9+ modeller
--**Billedgenerering**— `/v1/images/generations` med 10 udbydere og 20+ modeller (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
--**Tekst-til-video**— `/v1/videoer/generationer` — ComfyUI (AnimateDiff, SVD) og SD WebUI
--**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
--**Lydtransskription**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
--**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + eksisterende udbydere
--**Moderationer**— `/v1/moderations` — Indholdssikkerhedstjek
--**Reranking**— `/v1/rerank' — Reranking af dokumentrelevans
--**Responses API**— Fuld `/v1/responses`-understøttelse af Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction.
-
-🧪 14. "Jeg har ingen måde at teste og sammenligne kvalitet på tværs af modeller"
+**How OmniRoute solves it:**
-Udviklere vil gerne vide, hvilken model der er bedst til deres brug - kode, oversættelse, ræsonnement - men manuel sammenligning er langsom. Der findes ingen integrerede evalueringsværktøjer.
+- **npm global install** — `npm install -g omniroute && omniroute` — done
+- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi)
+- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw)
+- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode
+- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking)
+- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers
+- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups
-**Sådan løser OmniRoute det:**
+
--**LLM-evalueringer**— Gyldne sæt-test med 10 forudindlæste cases, der dækker hilsner, matematik, geografi, kodegenerering, JSON-overholdelse, oversættelse, markdown, sikkerhedsafvisning
--**4 matchstrategier**— 'præcis', 'indeholder', 'regex', 'brugerdefineret' (JS-funktion)
--**Translator Playground Test Bench**— Batchtest med flere input og forventede output, sammenligning på tværs af udbydere
--**Chattester**— Fuld rundtur med visuel responsgengivelse
--**Live Monitor**— Realtidsstream af alle anmodninger, der flyder gennem proxyen
+
+🌍 12. "The interface is English-only and my team doesn't speak English"
-
-📈 15. "Jeg har brug for at skalere uden at miste ydeevne"
+Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors.
-Efterhånden som forespørgselsvolumen vokser, genererer de samme spørgsmål duplikerede omkostninger uden cache. Uden idempotens, dublerede anmodninger om affaldsbehandling. Takstgrænser pr. udbyder skal overholdes.
+**How OmniRoute solves it:**
-**Sådan løser OmniRoute det:**
+- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English
+- **RTL Support** — Right-to-left support for Arabic and Hebrew
+- **Multi-Language READMEs** — 30 complete documentation translations
+- **Language Selector** — Globe icon in header for real-time switching
--**Semantisk cache**— To-lags cache (signatur + semantisk) reducerer omkostninger og latens
--**Request Idempotency**— 5s deduplikeringsvindue for identiske anmodninger
--**Detektion af hastighedsgrænse**— RPM pr. udbyder, min. gap og maks. samtidig sporing
--**Redigerbare hastighedsgrænser**— Konfigurerbare standardindstillinger i Indstillinger → Modstandsdygtighed med vedholdenhed
--**API Key Validation Cache**— 3-lags cache til produktionsydeevne
--**Health Dashboard med telemetri**— p50/p95/p99 latency, cachestatistik, oppetid
+
-
-🤖 16. "Jeg vil kontrollere modeladfærd globalt"
+
+🔄 13. "I need more than chat — I need embeddings, images, audio"
-Udviklere, der ønsker alle svar på et bestemt sprog, med en bestemt tone, eller ønsker at begrænse ræsonnementstokens. Det er upraktisk at konfigurere dette i hvert værktøj/anmodning.
+AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format.
-**Sådan løser OmniRoute det:**
+**How OmniRoute solves it:**
--**System Prompt Injection**— Global prompt anvendt på alle anmodninger
--**Thinking Budget Validation**— Reasoning token allocation control pr. anmodning (passthrough, auto, custom, adaptive)
--**9 Routing Strategies**— Globale strategier, der bestemmer, hvordan anmodninger distribueres
--**Wildcard Router**— `udbyder/*`-mønstre rutes dynamisk til enhver udbyder
--**Kombo Aktiver/Deaktiver Til/fra**— Skift kombinationer direkte fra dashboardet
--**Tilskiftning af udbyder**— Aktiver/deaktiver alle forbindelser for en udbyder med et enkelt klik
--**Blokerede udbydere**— Ekskluder specifikke udbydere fra `/v1/models` liste
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models
+- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
+- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI
+- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
+- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
+- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers
+- **Moderations** — `/v1/moderations` — Content safety checks
+- **Reranking** — `/v1/rerank` — Document relevance reranking
+- **Responses API** — Full `/v1/responses` support for Codex
-
-🧰 17. "Jeg har brug for MCP-værktøjer som førsteklasses produktegenskaber"
+
-Mange AI-gateways afslører kun MCP som en skjult implementeringsdetalje. Teams har brug for et synligt, overskueligt operationslag.
+
+🧪 14. "I have no way to test and compare quality across models"
-**Sådan løser OmniRoute det:**
+Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist.
-- MCP vises på fanen dashboardnavigation og endepunktsprotokol
-- Dedikeret MCP-administrationsside med proces, værktøjer, omfang og revision
-- Indbygget hurtigstart til `omniroute --mcp` og klient onboarding
+**How OmniRoute solves it:**
-
-🧠 18. "Jeg har brug for A2A-orkestrering med synkronisering + stream opgavestier"
+- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal
+- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function)
+- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison
+- **Chat Tester** — Full round-trip with visual response rendering
+- **Live Monitor** — Real-time stream of all requests flowing through the proxy
-Agentarbejdsgange kræver både direkte svar og langvarig streamet udførelse med livscykluskontrol.
+
-**Sådan løser OmniRoute det:**
+
+📈 15. "I need to scale without losing performance"
-- A2A JSON-RPC-slutpunkt ('POST /a2a') med 'message/send' og 'message/stream'
-- SSE-streaming med udbredelse af terminaltilstand
-- Opgavelivscyklus API'er for "opgaver/hent" og "opgaver/annuller".
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected.
-
-🛰️ 19. "Jeg har brug for ægte MCP-processundhed, ikke gættet status"
+**How OmniRoute solves it:**
-Operationelle teams skal vide, om MCP faktisk er i live, ikke kun om en API er tilgængelig.
+- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency
+- **Request Idempotency** — 5s deduplication window for identical requests
+- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking
+- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence
+- **API Key Validation Cache** — 3-tier cache for production performance
+- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime
-**Sådan løser OmniRoute det:**
+
-- Runtime-hjerteslagsfil med PID, tidsstempler, transport, værktøjstælling og omfangstilstand
-- MCP status API, der kombinerer hjerteslag + seneste aktivitet
-- UI-statuskort til proces/oppetid/hjerteslagsfriskhed
+
+🤖 16. "I want to control model behavior globally"
-
-📋 20. "Jeg har brug for auditable MCP-værktøjsudførelse"
+Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical.
-Når værktøjer muterer konfiguration eller udløser ops-handlinger, har teams brug for retsmedicinsk sporbarhed.
+**How OmniRoute solves it:**
-**Sådan løser OmniRoute det:**
+- **System Prompt Injection** — Global prompt applied to all requests
+- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive)
+- **9 Routing Strategies** — Global strategies that determine how requests are distributed
+- **Wildcard Router** — `provider/*` patterns route dynamically to any provider
+- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard
+- **Provider Toggle** — Enable/disable all connections for a provider with one click
+- **Blocked Providers** — Exclude specific providers from `/v1/models` listing
-- SQLite-støttet revisionslogning for MCP-værktøjsopkald
-- Filtrerer efter værktøj, succes/fiasko, API-nøgle og paginering
-- Dashboard revisionstabel + statistik slutpunkter til automatisering
+
-
-🔐 21. "Jeg har brug for scoped MCP-tilladelser pr. integration"
+
+🧰 17. "I need MCP tools as first-class product capabilities"
-Forskellige klienter bør have mindst privilegeret adgang til værktøjskategorier.
+Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer.
-**Sådan løser OmniRoute det:**
+**How OmniRoute solves it:**
-- 10 granulære MCP-skoper til kontrolleret værktøjsadgang
-- Håndhævelse af omfang og synlighed i MCP management UI
-- Sikker standardstilling for operationelt værktøj
+- MCP appears in the dashboard navigation and endpoint protocol tab
+- Dedicated MCP management page with process, tools, scopes, and audit
+- Built-in quick-start for `omniroute --mcp` and client onboarding
-
-⚙️ 22. "Jeg har brug for operationelle kontroller uden omfordeling"
+
-Teams har brug for hurtige runtime-ændringer under hændelser eller omkostningsbegivenheder.
+
+🧠 18. "I need A2A orchestration with sync + stream task paths"
-**Sådan løser OmniRoute det:**
+Agent workflows need both direct replies and long-running streamed execution with lifecycle control.
-- Skift kombinationsaktivering direkte fra MCP-dashboard
-- Anvend modstandsdygtighedsprofiler fra foruddefinerede politikpakker
-- Nulstil strømafbrydertilstand fra det samme betjeningspanel
+**How OmniRoute solves it:**
-
-🔄 23. "Jeg har brug for live A2A opgave livscyklus synlighed og annullering"
+- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream`
+- SSE streaming with terminal state propagation
+- Task lifecycle APIs for `tasks/get` and `tasks/cancel`
-Uden livscyklussynlighed bliver opgavehændelser svære at triage.
+
-**Sådan løser OmniRoute det:**
+
+🛰️ 19. "I need real MCP process health, not guessed status"
-- Opgaveliste/filtrering efter tilstand/færdighed med paginering
-- Drill-down på opgavemetadata, hændelser og artefakter
-- Slutpunkt for annullering af opgave og UI-handling med bekræftelse
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable.
-
-🌊 24. "Jeg har brug for aktive stream-metrics for A2A-indlæsning"
+**How OmniRoute solves it:**
-Streaming-arbejdsgange kræver operationel indsigt i samtidighed og live-forbindelser.
+- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode
+- MCP status API combining heartbeat + recent activity
+- UI status cards for process/uptime/heartbeat freshness
-**Sådan løser OmniRoute det:**
+
-- Aktive stream-tællere integreret i A2A-status
-- Tidsstempel for sidste opgave og tæller pr. stat
-- A2A dashboard-kort til operationsovervågning i realtid
+
+📋 20. "I need auditable MCP tool execution"
-
-🪪 25. "Jeg har brug for standardagentopdagelse til klienter"
+When tools mutate config or trigger ops actions, teams need forensic traceability.
-Eksterne klienter og orkestratorer har brug for maskinlæsbare metadata til onboarding.
+**How OmniRoute solves it:**
-**Sådan løser OmniRoute det:**
+- SQLite-backed audit logging for MCP tool calls
+- Filters by tool, success/failure, API key, and pagination
+- Dashboard audit table + stats endpoints for automation
-- Agentkort afsløret på `/.well-known/agent.json`
-- Evner og færdigheder vist i ledelsens brugergrænseflade
-- A2A status API inkluderer opdagelsesmetadata til automatisering
+
-
-🧭 26. "Jeg har brug for protokolsynlighed i produktets UX"
+
+🔐 21. "I need scoped MCP permissions per integration"
-Hvis brugere ikke kan opdage protokoloverflader, falder kvaliteten af adoption og support.
+Different clients should have least-privilege access to tool categories.
-**Sådan løser OmniRoute det:**
+**How OmniRoute solves it:**
-- Konsolideret**Endpoints**-side med faner til Proxy, MCP, A2A og API Endpoints
-- Inline service status skifter (Online/Offline) for MCP og A2A
-- Links fra oversigt til dedikerede administrationsfaner
+- 10 granular MCP scopes for controlled tool access
+- Scope enforcement and visibility in MCP management UI
+- Safe default posture for operational tooling
-
-🧪 27. "Jeg har brug for end-to-end protokolvalidering med rigtige klienter"
+
-Mock-tests er ikke nok til at validere protokolkompatibilitet før frigivelse.
+
+⚙️ 22. "I need operational controls without redeploying"
-**Sådan løser OmniRoute det:**
+Teams need quick runtime changes during incidents or cost events.
-- E2E-pakke, der starter app og bruger ægte MCP SDK-klienttransport
-- A2A klient tester for opdagelse, send, stream, hent og annuller flows
-- Krydstjek påstande mod MCP-revision og A2A-opgaver API'er
+**How OmniRoute solves it:**
-
-📡 28. "Jeg har brug for samlet observerbarhed på tværs af alle grænseflader"
+- Switch combo activation directly from MCP dashboard
+- Apply resilience profiles from pre-defined policy packs
+- Reset circuit breaker state from the same operations panel
-Opdeling af observerbarhed efter protokol skaber blinde pletter og længere MTTR.
+
-**Sådan løser OmniRoute det:**
+
+🔄 23. "I need live A2A task lifecycle visibility and cancellation"
-- Samlede dashboards/logfiler/analyse i ét produkt
-- Health + audit + request telemetri på tværs af OpenAI, MCP og A2A lag
-- Operationelle API'er til status og automatisering
+Without lifecycle visibility, task incidents become hard to triage.
-
-💼 29. "Jeg har brug for én runtime til proxy + værktøjer + agentorkestrering"
+**How OmniRoute solves it:**
-At køre mange separate tjenester øger driftsomkostninger og fejltilstande.
+- Task listing/filtering by state/skill with pagination
+- Drill-down on task metadata, events, and artifacts
+- Task cancellation endpoint and UI action with confirmation
-**Sådan løser OmniRoute det:**
+
-- OpenAI-kompatibel proxy, MCP-server og A2A-server i én stak
-- Delt godkendelse, robusthed, datalager og observerbarhed
-- Ensartet politikmodel på tværs af alle interaktionsflader
+
+🌊 24. "I need active stream metrics for A2A load"
-
-🚀 30. "Jeg skal sende agentiske arbejdsgange uden limkodesprawl"
+Streaming workflows require operational insight into concurrency and live connections.
-Hold mister hastighed, når de sammensætter flere ad-hoc-tjenester og scripts.
+**How OmniRoute solves it:**
-**Sådan løser OmniRoute det:**
+- Active stream counters integrated into A2A status
+- Last task timestamp and per-state counts
+- A2A dashboard cards for real-time ops monitoring
-- Ensartet slutpunktsstrategi for kunder og agenter
-- Indbygget protokolstyring UI'er og røgvalideringsstier
-- Produktionsklare fundamenter (sikkerhed, logning, robusthed, backup)
+
+
+
+🪪 25. "I need standard agent discovery for clients"
+
+External clients and orchestrators need machine-readable metadata for onboarding.
+
+**How OmniRoute solves it:**
+
+- Agent Card exposed at `/.well-known/agent.json`
+- Capabilities and skills shown in management UI
+- A2A status API includes discovery metadata for automation
+
+
+
+
+🧭 26. "I need protocol discoverability in the product UX"
+
+If users cannot discover protocol surfaces, adoption and support quality drop.
+
+**How OmniRoute solves it:**
+
+- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints
+- Inline service status toggles (Online/Offline) for MCP and A2A
+- Links from overview to dedicated management tabs
+
+
+
+
+🧪 27. "I need end-to-end protocol validation with real clients"
+
+Mock tests are not enough to validate protocol compatibility before release.
+
+**How OmniRoute solves it:**
+
+- E2E suite that boots app and uses real MCP SDK client transport
+- A2A client tests for discovery, send, stream, get, and cancel flows
+- Cross-check assertions against MCP audit and A2A tasks APIs
+
+
+
+
+📡 28. "I need unified observability across all interfaces"
+
+Splitting observability by protocol creates blind spots and longer MTTR.
+
+**How OmniRoute solves it:**
+
+- Unified dashboards/logs/analytics in one product
+- Health + audit + request telemetry across OpenAI, MCP, and A2A layers
+- Operational APIs for status and automation
+
+
+
+
+💼 29. "I need one runtime for proxy + tools + agent orchestration"
+
+Running many separate services increases operational cost and failure modes.
+
+**How OmniRoute solves it:**
+
+- OpenAI-compatible proxy, MCP server, and A2A server in one stack
+- Shared auth, resilience, data store, and observability
+- Consistent policy model across all interaction surfaces
+
+
+
+
+🚀 30. "I need to ship agentic workflows without glue-code sprawl"
+
+Teams lose velocity when stitching multiple ad-hoc services and scripts.
+
+**How OmniRoute solves it:**
+
+- Unified endpoint strategy for clients and agents
+- Built-in protocol management UIs and smoke validation paths
+- Production-ready foundations (security, logging, resilience, backup)
+
+
### Example Playbooks (Integrated Use Cases)
-**Playbook A: Maksimer betalt abonnement + billig backup**```txt
+**Playbook A: Maximize paid subscription + cheap backup**
+
+```txt
Combo: "maximize-claude"
1. cc/claude-opus-4-6
2. glm/glm-4.7
@@ -607,21 +690,23 @@ Combo: "maximize-claude"
Monthly cost: $20 + small backup spend
Outcome: higher quality, near-zero interruption
-````
+```
-**Playbook B: Kodningsstak uden omkostninger**```txt
+**Playbook B: Zero-cost coding stack**
+
+```txt
Combo: "free-forever"
-
-1. gc/gemini-3-flash
-2. if/kimi-k2-thinking
-3. qw/qwen3-coder-plus
+ 1. gc/gemini-3-flash
+ 2. if/kimi-k2-thinking
+ 3. qw/qwen3-coder-plus
Monthly cost: $0
Outcome: stable free coding workflow
+```
-````
+**Playbook C: 24/7 always-on fallback chain**
-**Playbook C: 24/7 altid aktiv reservekæde**```txt
+```txt
Combo: "always-on"
1. cc/claude-opus-4-6
2. cx/gpt-5.2-codex
@@ -630,122 +715,134 @@ Combo: "always-on"
5. if/kimi-k2-thinking
Outcome: deep fallback depth for deadline-critical workloads
-````
+```
-**Playbook D: Agent ops med MCP + A2A**```txt
+**Playbook D: Agent ops with MCP + A2A**
-1. Start MCP transport (`omniroute --mcp`) for tool-driven operations
-2. Run A2A tasks via `message/send` and `message/stream`
-3. Observe via /dashboard/endpoint (MCP and A2A tabs)
-4. Toggle services via inline status controls
-
-````
+```txt
+1) Start MCP transport (`omniroute --mcp`) for tool-driven operations
+2) Run A2A tasks via `message/send` and `message/stream`
+3) Observe via /dashboard/endpoint (MCP and A2A tabs)
+4) Toggle services via inline status controls
+```
---
## 🆓 Start Free — Zero Configuration Cost
-> Konfigurer AI-kodning på få minutter til**$0/måned**. Tilslut disse gratis konti, og brug den indbyggede**Free Stack**-kombination.
+> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo.
-| Trin | Handling | Udbydere ulåst |
-| ---- | -------------------------------------------------- | -------------------------------------------------------------------------- |
-| 1 | Tilslut**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**ubegrænset**|
-| 2 | Tilslut**Qoder**(Google OAuth) | kimi-k2-tænkning, qwen3-coder-plus, deepseek-r1... —**ubegrænset**|
-| 3 | Tilslut**Qwen**(enhedskode) | qwen3-coder-plus, qwen3-coder-flash... —**ubegrænset**|
-| 4 | Tilslut**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/md gratis**|
-| 5 | `/dashboard/combos` →**Gratis stak ($0)**skabelon | Round-robin alle gratis udbydere automatisk |
+| Step | Action | Providers Unlocked |
+| ---- | -------------------------------------------------- | ------------------------------------------------------------------ |
+| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** |
+| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** |
+| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** |
+| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** |
+| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically |
-**Peg enhver IDE/CLI til:**`http://localhost:20128/v1` · API-nøgle: `any-string` · Udført.
+**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done.
->**Valgfri ekstra dækning (også gratis):**Groq API-nøgle (30 RPM gratis), NVIDIA NIM (40 RPM gratis, 70+ modeller), Cerebras (1M tok/dag), LongCat API-nøgle (50M tokens/dag!), Cloudflare Workers AI (10K Neurons/day, 50+ modeller).## Kom hurtigt i gang
+> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models).
+
+## Kom hurtigt i gang
### 1) Install and run
```bash
npm install -g omniroute
omniroute
-````
+```
-> **pnpm-brugere:**Kør `pnpm approve-builds -g` efter installation for at aktivere native build-scripts, der kræves af `better-sqlite3` og `@swc/core`:
+> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`:
>
> ```bash
-> pnpm installer -g omniroute
-> pnpm approve-builds -g # Vælg alle pakker → godkend
+> pnpm install -g omniroute
+> pnpm approve-builds -g # Select all packages → approve
> omniroute
> ```
-Dashboard åbner på `http://localhost:20128` og API-base-URL er `http://localhost:20128/v1`.
+Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`.
-| Kommando | Beskrivelse |
+| Command | Description |
| ----------------------- | ----------------------------------------------------------- |
-| `omniroute` | Start server (`PORT=20128`, API og dashboard på samme port) |
-| `omniroute --port 3000` | Indstil kanonisk/API-port til 3000 |
-| `omniroute --mcp` | Start MCP-server (stdio-transport) |
-| `omniroute --no-open` | Åbn ikke browseren automatisk |
-| `omniroute --hjælp` | Vis hjælp |
+| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) |
+| `omniroute --port 3000` | Set canonical/API port to 3000 |
+| `omniroute --mcp` | Start MCP server (stdio transport) |
+| `omniroute --no-open` | Don't auto-open browser |
+| `omniroute --help` | Show help |
-Valgfri split-port-tilstand:```bash
+Optional split-port mode:
+
+```bash
PORT=20128 DASHBOARD_PORT=20129 omniroute
-
-# API: http://localhost:20128/v1
-
+# API: http://localhost:20128/v1
# Dashboard: http://localhost:20129
-
-````
+```
### Long-Running Streaming Timeouts
-Til de fleste implementeringer behøver du kun:
+For most deployments, you only need:
-| Variabel | Standard | Formål |
-| -------------------------- | ------------------------------ | ---------------------------------------------------------------------------------------------------------------------------------- |
-| `REQUEST_TIMEOUT_MS` | `600000` | Delt baseline for upstream-hentning, skjulte Undici-timeouts, TLS-fingeraftryksanmodninger og API-broanmodninger/proxy-timeouts |
-| `STREAM_IDLE_TIMEOUT_MS` | arver `REQUEST_TIMEOUT_MS` | Maksimalt mellemrum mellem streamingstykker, før OmniRoute afbryder SSE-strømmen |
+| Variable | Default | Purpose |
+| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- |
+| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts |
+| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream |
-Bagudkompatibilitet er bevaret: eksisterende `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` og andre timeoutvarianter pr. lag fungerer stadig og tilsidesætter den delte basislinje.
+Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline.
-Avancerede tilsidesættelser er tilgængelige, hvis du har brug for bedre kontrol:| Variabel | Standard | Formål |
-| ------------------------------------------ | ------------------------------------------ | ---------------------------------------------------------------------------- |
-| `FETCH_TIMEOUT_MS` | arver `REQUEST_TIMEOUT_MS` | Total upstream-anmodningstimeout brugt af hovedhentningsafbrydelsessignalet |
-| `FETCH_HEADERS_TIMEOUT_MS` | arver `FETCH_TIMEOUT_MS` | Undici tidsgrænse for modtagelse af opstrøms svaroverskrifter |
-| `FETCH_BODY_TIMEOUT_MS` | arver `FETCH_TIMEOUT_MS` | Undici tidsgrænse mellem opstrøms kropsstykker (`0` deaktiverer det) |
-| `FETCH_CONNECT_TIMEOUT_MS` | `30.000` | Undici TCP forbindelse timeout |
-| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout |
-| `TLS_CLIENT_TIMEOUT_MS` | arver `FETCH_TIMEOUT_MS` | Timeout for TLS-fingeraftryksanmodninger foretaget via `wreq-js` |
-| `API_BRIDGE_PROXY_TIMEOUT_MS` | arver `REQUEST_TIMEOUT_MS` eller `30000` | Timeout for `/v1` proxy-videresendelse fra API-port til dashboard-port |
-| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Timeout for indgående anmodning på API-broserveren |
-| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60.000` | Timeout for indgående header på API-broserveren |
-| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout på API-broserveren |
-| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inaktivitet timeout på API-broserveren (`0` deaktiverer den) |
+Advanced overrides are available if you need finer control:
-Hvis du kører OmniRoute bag Nginx, Caddy, Cloudflare eller en anden omvendt proxy, skal du sørge for, at proxyen
-timeouts er også højere end dine OmniRoute stream/hente timeouts.### 2) Connect providers and create your API key
+| Variable | Default | Purpose |
+| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- |
+| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal |
+| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers |
+| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) |
+| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout |
+| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout |
+| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` |
+| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port |
+| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server |
+| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server |
+| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server |
+| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) |
-1. Åbn Dashboard → `Providers` og tilslut mindst én udbyder (OAuth- eller API-nøgle).
-2. Åbn Dashboard → `Endpoints` og opret en API-nøgle.
-3. (Valgfrit) Åbn Dashboard → `Combos` og indstil din reservekæde.### 3) Point your coding tool to OmniRoute
+If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy
+timeouts are also higher than your OmniRoute stream/fetch timeouts.
+
+### 2) Connect providers and create your API key
+
+1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key).
+2. Open Dashboard → `Endpoints` and create an API key.
+3. (Optional) Open Dashboard → `Combos` and set your fallback chain.
+
+### 3) Point your coding tool to OmniRoute
```txt
Base URL: http://localhost:20128/v1
API Key: [copy from Endpoint page]
Model: if/kimi-k2-thinking (or any provider/model prefix)
-````
+```
-Fungerer med Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode og OpenAI-kompatible SDK'er.### 4) Enable and validate protocols (v2.0)
+Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs.
-**MCP (til værktøjsdrevne operationer):**```bash
+### 4) Enable and validate protocols (v2.0)
+
+**MCP (for tool-driven operations):**
+
+```bash
omniroute --mcp
+```
-````
-
-Tilslut derefter din MCP-klient over 'stdio' og test værktøjer som:
+Then connect your MCP client over `stdio` and test tools like:
- `omniroute_get_health`
- `omniroute_list_combos`
-**A2A (for agent-til-agent arbejdsgange):**```bash
+**A2A (for agent-to-agent workflows):**
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
```bash
curl -X POST http://localhost:20128/a2a \
@@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \
npm run test:protocols:e2e
```
-Denne suite validerer rigtige MCP- og A2A-klientstrømme mod en kørende app.### Alternative: run from source
+This suite validates real MCP and A2A client flows against a running app.
+
+### Alternative: run from source
```bash
cp .env.example .env
@@ -767,13 +866,13 @@ npm install
PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev
```
-
-Ugyldig Linux (`xbps-src`-skabelon)
+
+Void Linux (`xbps-src` template)
-For Void Linux-brugere kan du bygge en indbygget pakke ved hjælp af `xbps-src`. Gem denne blok som `srcpkgs/omniroute/template`:```bash
+For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`:
+```bash
# Template file for 'omniroute'
-
pkgname=omniroute
version=3.4.1
revision=1
@@ -785,7 +884,7 @@ license="MIT"
homepage="https://github.com/diegosouzapw/OmniRoute"
distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz"
checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b
-system_accounts="\_omniroute"
+system_accounts="_omniroute"
omniroute_homedir="/var/lib/omniroute"
export NODE_ENV=production
export npm_config_engine_strict=false
@@ -793,71 +892,70 @@ export npm_config_loglevel=error
export npm_config_fund=false
export npm_config_audit=false
-do_build() { # Determine target CPU arch for node-gyp
-local \_gyp_arch
-case "$XBPS_TARGET_MACHINE" in
-aarch64*) \_gyp_arch=arm64 ;;
-armv7*|armv6*) \_gyp_arch=arm ;;
-i686*) \_gyp_arch=ia32 ;;
-\*) \_gyp_arch=x64 ;;
-esac
+do_build() {
+ # Determine target CPU arch for node-gyp
+ local _gyp_arch
+ case "$XBPS_TARGET_MACHINE" in
+ aarch64*) _gyp_arch=arm64 ;;
+ armv7*|armv6*) _gyp_arch=arm ;;
+ i686*) _gyp_arch=ia32 ;;
+ *) _gyp_arch=x64 ;;
+ esac
- # 1) Install all deps – skip scripts (no network in do_build, native modules
- # compiled separately below; better-sqlite3 is serverExternalPackage so
- # Next.js does not execute it during next build)
- NODE_ENV=development npm ci --ignore-scripts
+ # 1) Install all deps – skip scripts (no network in do_build, native modules
+ # compiled separately below; better-sqlite3 is serverExternalPackage so
+ # Next.js does not execute it during next build)
+ NODE_ENV=development npm ci --ignore-scripts
- # 2) Build the Next.js standalone bundle
- npm run build
+ # 2) Build the Next.js standalone bundle
+ npm run build
- # 3) Copy static assets into standalone
- cp -r .next/static .next/standalone/.next/static
- [ -d public ] && cp -r public .next/standalone/public || true
+ # 3) Copy static assets into standalone
+ cp -r .next/static .next/standalone/.next/static
+ [ -d public ] && cp -r public .next/standalone/public || true
- # 4) Compile better-sqlite3 native binding for the target architecture.
- # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
- # without npm altering them.
- local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
- (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
+ # 4) Compile better-sqlite3 native binding for the target architecture.
+ # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
+ # without npm altering them.
+ local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
+ (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
- # 5) Place the compiled binding into the standalone bundle
- local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
- mkdir -p "$_bs3_release"
- cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
+ # 5) Place the compiled binding into the standalone bundle
+ local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
+ mkdir -p "$_bs3_release"
+ cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
- # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
- # so sharp is not used at runtime; x64 .so files would break aarch64 strip
- rm -rf .next/standalone/node_modules/@img
-
- # 7) Copy pino runtime deps omitted by Next.js static analysis:
- # pino-abstract-transport – required by pino's worker thread
- # split2 – dep of pino-abstract-transport
- # process-warning – dep of pino itself
- for _mod in pino-abstract-transport split2 process-warning; do
- cp -r "node_modules/$_mod" .next/standalone/node_modules/
- done
+ # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
+ # so sharp is not used at runtime; x64 .so files would break aarch64 strip
+ rm -rf .next/standalone/node_modules/@img
+ # 7) Copy pino runtime deps omitted by Next.js static analysis:
+ # pino-abstract-transport – required by pino's worker thread
+ # split2 – dep of pino-abstract-transport
+ # process-warning – dep of pino itself
+ for _mod in pino-abstract-transport split2 process-warning; do
+ cp -r "node_modules/$_mod" .next/standalone/node_modules/
+ done
}
do_check() {
-npm run test:unit
+ npm run test:unit
}
do_install() {
-vmkdir usr/lib/omniroute/.next
+ vmkdir usr/lib/omniroute/.next
- vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
+ vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
- # Prevent removal of empty Next.js app router dirs by the post-install hook
- for _d in \
- .next/standalone/.next/server/app/dashboard \
- .next/standalone/.next/server/app/dashboard/settings \
- .next/standalone/.next/server/app/dashboard/providers; do
- touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
- done
-
- cat > "${WRKDIR}/omniroute" <<'EOF'
+ # Prevent removal of empty Next.js app router dirs by the post-install hook
+ for _d in \
+ .next/standalone/.next/server/app/dashboard \
+ .next/standalone/.next/server/app/dashboard/settings \
+ .next/standalone/.next/server/app/dashboard/providers; do
+ touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
+ done
+ cat > "${WRKDIR}/omniroute" <<'EOF'
#!/bin/sh
export PORT="${PORT:-20128}"
export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}"
@@ -869,10 +967,9 @@ EOF
}
post_install() {
-vlicense LICENSE
+ vlicense LICENSE
}
-
-````
+```
@@ -880,9 +977,11 @@ vlicense LICENSE
## 🐳 Docker
-OmniRoute er tilgængelig som et offentligt Docker-billede på [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
+OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
-**Hurtigt løb:**```bash
+**Quick run:**
+
+```bash
docker run -d \
--name omniroute \
--restart unless-stopped \
@@ -890,85 +989,96 @@ docker run -d \
-p 20128:20128 \
-v omniroute-data:/app/data \
diegosouzapw/omniroute:latest
-````
+```
-**Med miljøfil:**```bash
+**With environment file:**
+```bash
# Copy and edit .env first
-
cp .env.example .env
docker run -d \
- --name omniroute \
- --restart unless-stopped \
- --stop-timeout 40 \
- --env-file .env \
- -p 20128:20128 \
- -v omniroute-data:/app/data \
- diegosouzapw/omniroute:latest
+ --name omniroute \
+ --restart unless-stopped \
+ --stop-timeout 40 \
+ --env-file .env \
+ -p 20128:20128 \
+ -v omniroute-data:/app/data \
+ diegosouzapw/omniroute:latest
+```
-````
+**Using Docker Compose:**
-**Brug af Docker Compose:**```bash
+```bash
# Base profile (no CLI tools)
docker compose --profile base up -d
# CLI profile (Claude Code, Codex, OpenClaw built-in)
docker compose --profile cli up -d
-````
+```
-Dashboard-understøttelse til Docker-implementeringer inkluderer nu et enkelt-klik**Cloudflare Quick Tunnel**på `Dashboard → Endpoints`. Den første aktivering downloader kun `cloudflared`, når det er nødvendigt, starter en midlertidig tunnel til dit nuværende `/v1`-slutpunkt og viser den genererede `https://*.trycloudflare.com/v1`-URL direkte under din normale offentlige URL.
+Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL.
-Bemærkninger:
+Notes:
-- Hurtige tunnel-URL'er er midlertidige og ændres efter hver genstart.
-- Hurtige tunneler gendannes ikke automatisk efter en OmniRoute- eller containergenstart. Genaktiver dem fra dashboardet, når det er nødvendigt.
-- Administreret installation understøtter i øjeblikket Linux, macOS og Windows på `x64` / `arm64`.
-- Managed Quick Tunnels er som standard HTTP/2-transport for at undgå støjende QUIC UDP-bufferadvarsler i begrænsede containermiljøer. Indstil `CLOUDFLARED_PROTOCOL=quic` eller `auto`, hvis du ønsker en anden transport.
-- Docker-billeder samler systemets CA-rødder og sender dem til administreret `cloudflared`, hvilket undgår TLS-tillidsfejl, når tunnelen starter inde i containeren.
-- SQLite kører i WAL-tilstand. `docker stop` skal have lov til at afslutte, så OmniRoute kan kontrollere de seneste ændringer tilbage i `storage.sqlite`.
-- De medfølgende Compose-filer sætter allerede en 40'er-stop-periode. Hvis du kører billedet direkte, skal du beholde `--stop-timeout 40` (eller lignende), så manuelle stop ikke afbryder nedlukningsoprydning.
-- Indstil `CLOUDFLARED_BIN=/absolute/sti/to/cloudflared`, hvis du ønsker, at OmniRoute skal bruge en eksisterende binær i stedet for at downloade en.
+- Quick Tunnel URLs are temporary and change after every restart.
+- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed.
+- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`.
+- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport.
+- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container.
+- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`.
+- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup.
+- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one.
-**Brug af Docker Compose med Caddy (HTTPS Auto-TLS):**
+**Using Docker Compose with Caddy (HTTPS Auto-TLS):**
-OmniRoute kan eksponeres sikkert ved hjælp af Caddys automatiske SSL-klargøring. Sørg for, at dit domænes DNS A-record peger på din servers IP.```yaml
+OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP.
+
+```yaml
services:
-omniroute:
-image: diegosouzapw/omniroute:latest
-container_name: omniroute
-restart: unless-stopped
-volumes: - omniroute-data:/app/data
-environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com
+ omniroute:
+ image: diegosouzapw/omniroute:latest
+ container_name: omniroute
+ restart: unless-stopped
+ volumes:
+ - omniroute-data:/app/data
+ environment:
+ - PORT=20128
+ - NEXT_PUBLIC_BASE_URL=https://your-domain.com
-caddy:
-image: caddy:latest
-container_name: caddy
-restart: unless-stopped
-ports: - "80:80" - "443:443"
-command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
+ caddy:
+ image: caddy:latest
+ container_name: caddy
+ restart: unless-stopped
+ ports:
+ - "80:80"
+ - "443:443"
+ command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
volumes:
-omniroute-data:
+ omniroute-data:
+```
-````
+| Image | Tag | Size | Description |
+| ------------------------ | -------- | ------ | --------------------- |
+| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release |
+| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version |
-| Billede | Tag | Størrelse | Beskrivelse |
-| -------------------------- | -------- | ------ | ---------------------- |
-| `diegosouzapw/omniroute` | `nyeste` | ~250MB | Seneste stabile udgivelse |
-| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Nuværende version |---
+---
## 🖥️ Desktop App — Offline & Always-On
-> 🆕**NYT!**OmniRoute er nu tilgængelig som en**native desktop-applikation**til Windows, macOS og Linux.
+> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux.
-Kør OmniRoute som en selvstændig desktop-app - ingen terminal, ingen browser, intet internet påkrævet for lokale modeller. Den elektronbaserede app inkluderer:
+Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes:
-- 🖥️**Native Window**— Dedikeret appvindue med systembakkeintegration
-- 🔄**Auto-Start**— Start OmniRoute ved systemlogin
-- 🔔**Native notifikationer**— Få advarsler om kvoteopbrugt eller udbyderproblemer
-- ⚡**One-Click Install**— NSIS (Windows), DMG (macOS), AppImage (Linux)
-- 🌐**Offline-tilstand**— Fungerer fuldt ud offline med medfølgende server### Kom hurtigt i gang
+- 🖥️ **Native Window** — Dedicated app window with system tray integration
+- 🔄 **Auto-Start** — Launch OmniRoute on system login
+- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues
+- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux)
+- 🌐 **Offline Mode** — Works fully offline with bundled server
+
+### Kom hurtigt i gang
```bash
# Development mode
@@ -979,308 +1089,370 @@ npm run electron:build # Current platform
npm run electron:build:win # Windows (.exe)
npm run electron:build:mac # macOS (.dmg) — x64 & arm64
npm run electron:build:linux # Linux (.AppImage)
-````
+```
### System Tray
-Når den er minimeret, lever OmniRoute i din procesbakke med hurtige handlinger:
+When minimized, OmniRoute lives in your system tray with quick actions:
-- Åbn instrumentbrættet
-- Skift serverport
-- Afslut programmet
+- Open dashboard
+- Change server port
+- Quit application
-📖 Fuld dokumentation: [`electron/README.md`](electron/README.md)---
+📖 Full documentation: [`electron/README.md`](electron/README.md)
+
+---
## 💰 Pricing at a Glance
-| Tier | Udbyder | Omkostninger | Kvote nulstilling | Bedst til |
-| ----------------- | --------------------------- | ------------------------------- | ------------------ | ------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
-| **💳 ABONNEMENT** | Claude Code (Pro) | 20 USD/md. | 5 timer + ugentlig | Allerede abonneret |
-| | Codex (Plus/Pro) | $20-200/md. | 5 timer + ugentlig | OpenAI-brugere |
-| | Gemini CLI | **GRATIS** | 180K/md + 1K/dag | Alle sammen! |
-| | GitHub Copilot | $10-19/md. | Månedlig | GitHub-brugere |
-| **🔑 API NØGLE** | NVIDIA NIM | **GRATIS**(dev for evigt) | ~40 RPM | 70+ åbne modeller |
-| | Cerebras | **GRATIS**(1M tok/dag) | 60K TPM / 30 RPM | Verdens hurtigste |
-| | Groq | **GRATIS**(30 RPM) | 14,4K RPD | Ultrahurtig Lama/Gemma |
-| | DeepSeek V3.2 | 0,27 USD/1,10 USD pr. 1 mio. | Ingen | Bedste pris/kvalitet ræsonnement |
-| | xAI Grok-4 Hurtig | **$0,20/$0,50 pr. 1M**🆕 | Ingen | Hurtigste + værktøjsopkald, ultralav |
-| | xAI Grok-4 (standard) | 0,20 USD/1,50 USD pr. 1 mio. 🆕 | Ingen | Fornuft flagskib fra xAI |
-| | Mistral | Gratis prøveperiode + betalt | Sats begrænset | Europæisk AI |
-| | OpenRouter | Betal pr. brug | Ingen | 100+ modeller aggr. |
-| **💰 BILLIG** | GLM-5 (via Z.AI) 🆕 | 0,5 USD/1 mio. | Dagligt 10:00 | 128K output, nyeste flagskib |
-| | GLM-4.7 | 0,6 USD/1 mio. | Dagligt 10:00 | Budget backup |
-| | MiniMax M2.5 🆕 | $0,3/1 mio. input | 5-timers rullende | Begrundelse + agentopgaver |
-| | MiniMax M2.1 | $0,2/1 mio. | 5-timers rullende | Billigste mulighed |
-| | Kimi K2.5 (Moonshot API) 🆕 | Betal pr. brug | Ingen | Direkte Moonshot API-adgang |
-| | Kimi K2 | 9 USD/md. lejlighed | 10M tokens/md. | Forudsigelige omkostninger |
-| **🆓 GRATIS** | Qoder | **$0** | Ubegrænset | 5 modeller ubegrænset |
-| | Qwen | **$0** | Ubegrænset | 4 modeller ubegrænset |
-| | Kiro | **$0** | Ubegrænset | Claude Sonnet/Haiku (AWS Builder) |
-| | LongCat Flash-Lite 🆕 | **$0**(50 mio. tok/dag 🔥) | 1 RPS | Største gratis kvote på jorden |
-| | Bestøvninger AI 🆕 | **$0**(ingen nøgle nødvendig) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 |
-| | Cloudflare Workers AI 🆕 | **$0**(10.000 neuroner/dag) | ~150 hhv/dag | 50+ modeller, global kant |
-| | Scaleway AI 🆕 | **$0**(1 mio. tokens i alt) | Sats begrænset | EU/GDPR, Qwen3 235B, Lama 70B | > 🆕**Nye modeller tilføjet (mars 2026):**Grok-4 Fast-familie til $0,20/$0,50/M (benchmarked ved 1143ms — 30 % hurtigere end Gemini 2.5 Flash), GLM-5 via Z.AI med 128K output, MiniMax M2.5-begrundelse, KimSeidek pr. direkte API. |
+| Tier | Provider | Cost | Quota Reset | Best For |
+| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- |
+| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed |
+| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users |
+| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! |
+| | GitHub Copilot | $10-19/mo | Monthly | GitHub users |
+| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models |
+| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest |
+| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma |
+| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning |
+| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow |
+| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI |
+| | Mistral | Free trial + paid | Rate limited | European AI |
+| | OpenRouter | Pay-per-use | None | 100+ models aggr. |
+| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship |
+| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup |
+| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks |
+| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option |
+| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access |
+| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost |
+| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited |
+| | Qwen | **$0** | Unlimited | 4 models unlimited |
+| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) |
+| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth |
+| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 |
+| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge |
+| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B |
-**💡 $0 Combo Stack — Den komplette gratis opsætning:**```
+> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API.
+**💡 $0 Combo Stack — The Complete Free Setup:**
+
+```
# 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever
+Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
+Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
+Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
+Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
+NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+```
-Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
-Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
-LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
-Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
-Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
-Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
-Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
-Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
-Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
-NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
-Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever.
-````
-
-**Nul omkostninger. Stopper aldrig med at kode.**Konfigurer dette som én OmniRoute-kombination, og alle fallbacks sker automatisk - ingen manuel skift nogensinde.---
+---
---
## 🆓 Free Models — What You Actually Get
-> Alle modeller nedenfor er**100 % gratis uden kreditkort påkrævet**. OmniRoute dirigerer automatisk mellem dem, når én kvote løber ud - kombiner dem alle for en ubrydelig kombination af $0.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
+> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo.
-| Model | Præfiks | Grænse | Satsgrænse |
-| ------------------ | ------ | ------------- | ---------------------- |
-| `claude-sonnet-4.5` | `kr/` |**Ubegrænset**| Ingen rapporteret dagligt loft |
-| `claude-haiku-4.5` | `kr/` |**Ubegrænset**| Ingen rapporteret dagligt loft |
-| `claude-opus-4.6` | `kr/` |**Ubegrænset**| Seneste Opus via Kiro |### 🟢 QODER MODELS (Free PAT via qodercli)
+### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
-| Model | Præfiks | Grænse | Satsgrænse |
+| Model | Prefix | Limit | Rate Limit |
+| ------------------- | ------ | ------------- | --------------------- |
+| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro |
+
+### 🟢 QODER MODELS (Free PAT via qodercli)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------ | ------ | ------------- | --------------- |
-| `kimi-k2-tænkning` | `hvis/` |**Ubegrænset**| Ingen rapporteret loft |
-| `qwen3-coder-plus` | `hvis/` |**Ubegrænset**| Ingen rapporteret loft |
-| `deepseek-r1` | `hvis/` |**Ubegrænset**| Ingen rapporteret loft |
-| `minimax-m2.1` | `hvis/` |**Ubegrænset**| Ingen rapporteret loft |
-| `kimi-k2` | `hvis/` |**Ubegrænset**| Ingen rapporteret loft |
+| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap |
+| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap |
+| `deepseek-r1` | `if/` | **Unlimited** | No reported cap |
+| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap |
+| `kimi-k2` | `if/` | **Unlimited** | No reported cap |
-> Anbefalet forbindelsesmetode:**Personal Access Token + `qodercli`**. Browser OAuth er
-> eksperimentel og deaktiveret som standard, medmindre `QODER_OAUTH_*` miljøvariabler er konfigureret.### 🟡 QWEN MODELS (Device Code Auth)
+> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is
+> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured.
-| Model | Præfiks | Grænse | Satsgrænse |
-| ------------------ | ------ | ------------- | ------------------ |
-| `qwen3-coder-plus` | `qw/` |**Ubegrænset**| Ingen rapporteret loft |
-| `qwen3-coder-flash` | `qw/` |**Ubegrænset**| Ingen rapporteret loft |
-| `qwen3-coder-next` | `qw/` |**Ubegrænset**| Ingen rapporteret loft |
-| `vision-model` | `qw/` |**Ubegrænset**| Multimodal (billeder) |### 🟣 GEMINI CLI (Google OAuth)
+### 🟡 QWEN MODELS (Device Code Auth)
-| Model | Præfiks | Grænse | Satsgrænse |
-| -------------------------- | ------ | -------------------------- | ------------- |
-| `gemini-3-flash-preview` | `gc/` |**180K tok/måned**+ 1K/dag | Månedlig nulstilling |
-| `gemini-2.5-pro` | `gc/` | 180K/måned (delt pool) | Høj kvalitet |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------- | ------ | ------------- | ------------------- |
+| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap |
+| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) |
-| Tier | Daglig grænse | Satsgrænse | Noter |
-| ---------- | ------------ | ----------- | -------------------------------------------------------------- |
-| Gratis (Dev) | Ingen token cap |**~40 RPM**| 70+ modeller; overgang til rene satsgrænser medio 2025 |
+### 🟣 GEMINI CLI (Google OAuth)
-Populære gratis modeller: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`/, `deepseek-seek`/`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------------ | ------ | --------------------------- | ------------- |
+| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset |
+| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality |
-| Tier | Daglig grænse | Satsgrænse | Noter |
-| ---- | ------------------ | ---------------- | -------------------------------------------------- |
-| Gratis |**1 mio. tokens/dag**| 60K TPM / 30 RPM | Verdens hurtigste LLM-slutning; nulstilles dagligt |
+### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
-Tilgængelig gratis: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-destill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com)
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---------- | ------------ | ----------- | ------------------------------------------------------ |
+| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 |
-| Tier | Daglig grænse | Satsgrænse | Noter |
-| ---- | ------------- | ---------------- | ------------------------------------------ |
-| Gratis |**14,4K RPD**| 30 RPM pr. model | Intet kreditkort; 429 på grænse, ikke opkrævet |
+Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`
-Tilgængelig gratis: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
-| Model | Præfiks | Daglig gratis kvote | Noter |
-| ------------------------------ | ------ | ------------------ | ---------------------------- |
-| `LongCat-Flash-Lite` | `lc/` |**50 mio. tokens**💥 | Største gratis kvote nogensinde |
-| `LongCat-Flash-Chat` | `lc/` | 500.000 tokens | Multi-turn chat |
-| `LongCat-Flash-Thinking` | `lc/` | 500.000 tokens | Begrundelse / CoT |
-| `LongCat-Flash-Thinking-2601` | `lc/` | 500.000 tokens | Jan 2026 version |
-| `LongCat-Flash-Omni-2603` | `lc/` | 500.000 tokens | Multimodal |
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---- | ----------------- | ---------------- | ------------------------------------------- |
+| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily |
-> 100 % gratis, mens du er i offentlig beta. Tilmeld dig på [longcat.chat](https://longcat.chat) med e-mail eller telefon. Nulstiller dagligt 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`
-| Model | Præfiks | Satsgrænse | Udbyder bag |
+### 🔴 GROQ (Free API Key — console.groq.com)
+
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---- | ------------- | ---------------- | ----------------------------------------- |
+| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged |
+
+Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`
+
+### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+
+| Model | Prefix | Daily Free Quota | Notes |
+| ----------------------------- | ------ | ----------------- | ----------------------- |
+| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever |
+| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat |
+| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT |
+| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version |
+| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal |
+
+> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC.
+
+### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+
+| Model | Prefix | Rate Limit | Provider Behind |
| ---------- | ------ | ---------- | ------------------ |
-| `openai` | `pol/` | 1 req/15s | GPT-5 |
-| `claude` | `pol/` | 1 req/15s | Antropiske Claude |
-| `gemini` | `pol/` | 1 req/15s | Google Gemini |
-| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 |
-| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout |
-| `mistral` | `pol/` | 1 req/15s | Mistral AI |
+| `openai` | `pol/` | 1 req/15s | GPT-5 |
+| `claude` | `pol/` | 1 req/15s | Anthropic Claude |
+| `gemini` | `pol/` | 1 req/15s | Google Gemini |
+| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 |
+| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout |
+| `mistral` | `pol/` | 1 req/15s | Mistral AI |
-> ✨**Nul friktion:**Ingen tilmelding, ingen API-nøgle. Tilføj bestøvningsudbyderen med et tomt nøglefelt, og det virker med det samme.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately.
-| Tier | Daglige neuroner | Tilsvarende brug | Noter |
-| ---- | ------------- | ----------------------------------------------- | ---------------------------- |
-| Gratis |**10.000**| ~150 LLM resp. / 500s lyd / 15K indlejringer | Global kant, 50+ modeller |
+### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
-Populære gratis modeller: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (gratis lyd!), `@cf/qwen/qwen2.5-coder-15b-instruct`
+| Tier | Daily Neurons | Equivalent Usage | Notes |
+| ---- | ------------- | --------------------------------------- | ----------------------- |
+| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models |
-> Kræver API-token + konto-id fra [dash.cloudflare.com](https://dash.cloudflare.com). Gem konto-id i udbyderindstillinger.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct`
-| Tier | Gratis kvote | Beliggenhed | Noter |
+> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings.
+
+### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+
+| Tier | Free Quota | Location | Notes |
| ---- | ------------- | ------------ | ----------------------------------- |
-| Gratis |**1 mio. tokens**| 🇫🇷 Paris, EU | Intet kreditkort nødvendigt inden for grænserne |
+| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits |
-Tilgængelig gratis: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
+Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
-> EU/GDPR-kompatibel. Hent API-nøgle på [console.scaleway.com](https://console.scaleway.com).
+> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com).
->**💡 Den ultimative gratis stak (11 udbydere, $0 for evigt):**
+> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):**
>
> ```
-> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
-> Qoder (hvis/) → kimi-k2-tænkning, qwen3-coder-plus, deepseek-r1 UNLIMITED
-> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 mio. tokens/dag 🔥
-> Bestøvninger (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — ingen nøgle nødvendig
-> Qwen (qw/) → qwen3-koder modeller UBEGRÆNSET
-> Gemini (gemini/) → Gemini 2.5 Flash — 1.500 req/dag gratis
-> Cloudflare AI (jf/) → 50+ modeller — 10K neuroner/dag
-> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M gratis tokens (EU)
-> Groq (groq/) → Lama/Gemma — 14,4K req/dag ultrahurtig
-> NVIDIA NIM (nvidia/) → 70+ åbne modeller — 40 RPM for evigt
-> Cerebras (cerebras/) → Lama/Qwen verdenshurtigste — 1M tok/dag
-> ```## 🎙️ Free Transcription Combo
+> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+> Qwen (qw/) → qwen3-coder models UNLIMITED
+> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free
+> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day
+> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast
+> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+> ```
-> Transskriber enhver lyd/video for**$0**— Deepgram-emner med $200 gratis, AssemblyAI $50 fallback, Groq Whisper som ubegrænset nødbackup.
+## 🎙️ Free Transcription Combo
-| Udbyder | Gratis kreditter | Bedste model | Satsgrænse |
-| ------------------ | ---------------------- | -------------------------------------------- | ---------------------------- |
-|**Deepgram**|**$200 gratis**(tilmelding) | `nova-3` — bedste nøjagtighed, 30+ sprog | Ingen RPM-grænse på gratis kreditter |
-| 🔵**AssemblyAI**|**$50 gratis**(tilmelding) | `universal-3-pro` — kapitler, følelser, PII | Ingen RPM-grænse på gratis kreditter |
-| 🔴**Groq**|**Gratis for evigt**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (hastighedsbegrænset) |
+> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup.
-**Foreslået kombination i `/dashboard/combos`:**```
+| Provider | Free Credits | Best Model | Rate Limit |
+| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- |
+| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits |
+| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits |
+| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) |
+
+**Suggested combo in `/dashboard/combos`:**
+
+```
Name: free-transcription
Strategy: Priority
Nodes:
[1] deepgram/nova-3 → uses $200 free first
[2] assemblyai/universal-3-pro → fallback when Deepgram credits run out
[3] groq/whisper-large-v3 → free forever, emergency fallback
-````
+```
-Derefter i `/dashboard/media` → fanen**Transskription**: upload en lyd- eller videofil → vælg dit kombinationsslutpunkt → få transskription i understøttede formater.## 💡 Key Features
+Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats.
-OmniRoute v2.0 er bygget som en operationel platform, ikke kun en relæ-proxy.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
+## 💡 Key Features
-| Funktion | Hvad det gør |
-| ----------------------------------------- | ----------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ |
-| ⚡**Grok-4 Fast Family** | xAI-modeller til $0,20/$0,50/M — benchmarked 1143ms (30 % hurtigere end Gemini 2,5 Flash) |
-| 🧠**GLM-5 via Z.AI** | 128K output kontekst, $0,5/1M — nyeste flagskib fra GLM-familien |
-| 🔮**MiniMax M2.5** | Begrundelse + agentopgaver til $0,30/1M — betydelig opgradering fra M2.1 |
-| 🎯**værktøj Calling Flag per model** | "ToolCalling" pr. model: sand/falsk i registreringsdatabasen — AutoCombo springer ikke-værktøjskompatible modeller over |
-| 🌍**Flersproget hensigtsdetektion** | PT/ZH/ES/AR nøgleord i AutoCombo scoring — bedre modelvalg for ikke-engelsk indhold |
-| 📊**Benchmark-drevne fallbacks** | Ægte p95-forsinkelse fra live-anmodninger feeds combo scoring — AutoCombo lærer af faktiske data |
-| 🔁**Anmod om deduplikation** | Indholdshash-baseret dedup-vindue — multi-agent sikker, forhindrer duplikerede debiteringer |
-| 🔌**Strategi, der kan tilsluttes router** | Udvidelig `RouterStrategy`-grænseflade — tilføj brugerdefineret routinglogik som plugins | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP |
+OmniRoute v3.5 is built as an operational platform, not just a relay proxy.
-| Funktion | Hvad det gør |
-| ----------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |
-| 🎮**Model Legeplads** | Dashboard-side for at teste enhver model direkte — udbyder/model/slutpunktsvælgere, Monaco Editor, streaming, afbrydelse, timing |
-| 🔏**CLI Fingerprint Matching** | Bestilling af header/body pr. udbyder for at matche native CLI-signaturer — skift pr. udbyder i Indstillinger > Sikkerhed.**Din proxy-IP er bevaret** |
-| 🤝**ACP Support (Agent Client Protocol)** | CLI-agentopdagelse (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 mere), procesopstart, `/api/acp/agents` slutpunkt |
-| 🤖**ACP Agents Dashboard** | Fejlfinding › Agenter-side — gitter med 14 agenter med installationsstatus, version, brugerdefineret agentformular til ethvert CLI-værktøj.**OpenCode**-brugere får en "Download opencode.json"-knap, der automatisk genererer en klar-til-brug-konfiguration med alle tilgængelige modeller. |
-| 🔧**Brugerdefineret model `apiFormat` Routing** | Brugerdefinerede modeller med `apiFormat: "responses"` rutes nu korrekt til Responses API-oversætteren |
-| 🏢**Codex Workspace Isolation** | Flere Codex-arbejdsområder pr. e-mail — OAuth adskiller forbindelser korrekt efter arbejdsområde-id |
-| 🔄**Automatisk opdatering af elektroner** | Desktop-app søger efter opdateringer + automatisk installation ved genstart | ### 🤖 Agent & Protocol Operations (v2.0) |
+### 🆕 New — v3.5.5 Highlights (Apr 2026)
-| Funktion | Hvad det gør |
-| --------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- |
-| 🔧**MCP-server (25 værktøjer)** | IDE/agent værktøjer via 3 transporter: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 kerner + 3 hukommelse + 4 færdighedsværktøjer |
-| 🤝**A2A-server (JSON-RPC + SSE)** | Agent-til-agent opgaveudførelse med synkronisering og streaming flows |
-| 🧭**Konsoliderede slutpunkter-side** | Administrationsside med faner med Endpoint Proxy, MCP, A2A og API Endpoints faner |
-| 🎚️**Tjenesteaktiver/deaktiver skifter** | ON/OFF-kontakter til MCP og A2A med fastholdelse af indstillinger (standard: OFF) |
-| 🛰️**MCP Runtime Heartbeat** | Reel processtatus (pid, oppetid, hjerteslagsalder, transport, omfangstilstand) |
-| 📋**MCP Audit Trail** | Filtrerbare revisionslogfiler med succes/fejl og nøgletilskrivning |
-| 🔐**MCP Scope Enforcement** | 10 granulære omfangstilladelser til kontrolleret værktøjsadgang |
-| 📡**A2A Task Lifecycle Management** | Liste/filtrere opgaver, inspicere hændelser/artefakter, annullere kørende opgaver |
-| 📋**Agent Card Discovery** | `/.well-known/agent.json` til klient auto-discovery |
-| 🧪**Protokol E2E testsele** | Ægte MCP SDK + A2A klient flows i `test:protocols:e2e` |
-| ⚙️**Driftskontrol** | Switch combo, påfør elasticitetsprofiler, nulstil afbrydere fra én kontrolflade | ### 🧠 Routing & Intelligence |
+| Feature | What It Does |
+| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation |
+| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs |
+| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions |
+| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection |
+| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts |
-| Funktion | Hvad det gør |
-| ---------------------------------- | ------------------------------------------------------------------------------- | ----------------------- |
-| 🎯**Smart 4-lags fallback** | Auto-rute: Abonnement → API-nøgle → Billig → Gratis |
-| 📊**Kvotesporing i realtid** | Live token count + nulstil nedtælling pr. udbyder |
-| 🔄**Formatoversættelse** | OpenAI ↔ Claude ↔ Gemini ↔ Svar med skemasikre konverteringer |
-| 👥**Multi-Account Support** | Flere konti pr. udbyder med intelligent valg |
-| 🔄**Automatisk token-opdatering** | OAuth-tokens opdateres automatisk med genforsøg |
-| 🎨**Tilpassede kombinationer** | 9 balanceringsstrategier + fallback kædekontrol |
-| 🌐**Wildcard-router** | `udbyder/*` dynamisk routing |
-| 🧠**Tænker på budgetkontrol** | Grænser for gennemstrømning, automatisk, brugerdefineret og adaptiv ræsonnement |
-| 🔀**Modelaliaser** | Indbygget + brugerdefineret model aliasing og migration sikkerhed |
-| ⚡**Baggrundsforringelse** | Send baggrundsopgaver med lav prioritet til billigere modeller |
-| 🧪**Task-Aware Smart Routing** | Auto-vælg model efter indholdstype (kodning/vision/analyse/opsummering) |
-| 🔄**A2A Agent Workflows** | Deterministisk FSM-orkestrator til stateful multi-step agent henrettelser |
-| 🔀**Adaptiv Routing** | Dynamisk strategitilsidesættelse baseret på tokenvolumen og promptkompleksitet |
-| 🎲**Udbyderdiversitet** | Shannon entropi-scoring balancerer auto-combo-trafikfordeling |
-| 💬**System Prompt Injection** | Globale adfærdskontroller anvendes konsekvent |
-| 📄**Responses API-kompatibilitet** | Fuld `/v1/responses`-understøttelse af Codex og avancerede agent-arbejdsgange | ### 🎵 Multi-Modal APIs |
+### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
-| Funktion | Hvad det gør |
-| ----------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- |
-| 🖼️**Billedgenerering** | `/v1/images/generations` med sky og lokale backends |
-| 📐**Indlejringer** | `/v1/embeddings` til søgning og RAG-rørledninger |
-| 🎤**Lydtransskription** | `/v1/audio/transcriptions` — 7 udbydere (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-sprogdetektion, MP4/MP3/WAV-understøttelse |
-| 🔊**Tekst-til-tale** | `/v1/audio/speech` — 10 udbydere (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) med korrekte fejlmeddelelser |
-| 🎬**Videogenerering** | `/v1/videos/generations` (ComfyUI + SD WebUI-arbejdsgange) |
-| 🎵**Music Generation** | `/v1/music/generations` (ComfyUI-arbejdsgange) |
-| 🛡️**Moderationer** | `/v1/moderations` sikkerhedstjek |
-| 🔀**Omrangering** | `/v1/rerank` for relevansscoring |
-| 🔍**Websøgning**🆕 | `/v1/search` — 5 udbydere (Serper, Brave, Perplexity, Exa, Tavily), 6.500+ gratis/måned, auto-failover, cache | ### 🛡️ Resilience, Security & Governance |
+| Feature | What It Does |
+| ------------------------------------ | ------------------------------------------------------------------------------------------- |
+| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) |
+| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family |
+| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 |
+| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models |
+| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content |
+| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data |
+| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges |
+| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins |
-| Funktion | Hvad det gør |
-| ----------------------------------- | ------------------------------------------------------------------------------------------- | -------------------------------- |
-| 🔌**Maksimalafbrydere** | Pr. model tur/restitution med tærskelkontrol |
-| 🎯**Endpoint-Aware-modeller** | Brugerdefinerede modeller erklærer understøttede slutpunkter + API-format |
-| 🛡️**Anti-tordenbesætning** | Mutex + semaforbeskyttelse ved genforsøg/rate hændelser |
-| 🧠**Semantisk + signaturcache** | Reduktion af omkostninger/latens med to cachelag |
-| ⚡**Anmod om idempotens** | Dobbelt beskyttelsesvindue |
-| 🔒**TLS Fingerprint Spoofing** | Browserlignende TLS-fingeraftryk —**reducerer botgenkendelse og kontoflaggning** |
-| 🔏**CLI Fingerprint Matching** | Matcher native CLI-anmodningssignaturer —**reducerer forbudsrisiko, mens proxy-IP bevares** |
-| 🌐**IP-filtrering** | Tilladelsesliste/blokeringslistekontrol for udsatte implementeringer |
-| 📊**Redigerbare satsgrænser** | Konfigurerbare grænser på globalt niveau/udbyderniveau med persistens |
-| 📉**Graceful Nedbrydning** | Muligheder med flere lag, der beskytter kerne-gateway-operationer |
-| 📜**Config Audit Trail** | Diff-baseret ændringssporing forhindrer driftsafdrift med simple rollbacks |
-| ⏳**Provider Health Sync** | Proaktiv overvågning af tokens udløb, der udløser advarsler før godkendelsesfejl |
-| 🚪**Auto-deaktiver forbudte konti** | Driftsafbryder forsegling permanent blokerede token-konti automatisk |
-| 🔑**API Key Management + Scoping** | Sikker nøgleudstedelse/rotation og model-/leverandørkontrol |
-| 👁️**Scoped API Key Reveal**🆕 | Opt-in gendannelse af API-nøgler via `ALLOW_API_KEY_REVEAL` |
-| 🛡️**Beskyttet `/modeller`** | Valgfri godkendelse og udbyderskjul til modelkatalog | ### 📊 Observability & Analytics |
+### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP
-| Funktion | Hvad det gør |
-| -------------------------------------- | ---------------------------------------------------------------- | ---------------------------- |
-| 📝**Forespørgsel + Proxylogning** | Fuld anmodning/svar og proxy-logning |
-| 📉**Streamede detaljerede logfiler**🆕 | Rekonstruerer SSE-nyttelaststrømme rent ind i brugergrænsefladen |
-| 📋**Unified Logs Dashboard** | Anmodning, proxy, revision og konsolvisning på én side |
-| 🔍**Anmod om telemetri** | p50/p95/p99 latens og anmodningssporing |
-| 🏥**Sundhedskontrolpanel** | Oppetid, breaker-tilstande, lockouts, cache-statistik |
-| 💰**Omkostningssporing** | Budgetkontrol og prisfastsættelse pr. model |
-| 📈**Analytiske visualiseringer** | Model-/udbyderbrugsindsigt og trendvisninger |
-| 🧪**Evalueringsramme** | Gyldne sæt-test med konfigurerbare matchstrategier |
-| 📡**Live Diagnostics**🆕 | Semantisk cache-bypass for nøjagtig combo live-test | ### ☁️ Deployment & Platform |
+| Feature | What It Does |
+| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing |
+| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** |
+| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint |
+| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. |
+| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator |
+| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID |
+| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart |
-| Funktion | Hvad det gør |
-| ------------------------------------- | ----------------------------------------------------------------- | --------------------- |
-| 🌐**Deploy hvor som helst** | Localhost, VPS, Docker, Cloud-miljøer |
-| 🚇**Cloudflare Tunnel**🆕 | Hurtig tunnel-integration med et enkelt klik fra dashboardet |
-| 🔑**API-nøglemodelfiltrering** | Native /v1/models-svar filtreret via tildelte bærerkontekstroller |
-| ⚡**Smart Cache Bypass** | Konfigurerbar TTL-heuristik og tvungen genhentningskontroller |
-| 🔄**Sikkerhedskopiering/gendannelse** | Eksport/import og gendannelsesstrømme |
-| 🧙**Onboarding Wizard** | Første kørsel guidet opsætning |
-| 🔧**CLI Tools Dashboard** | Et-klik opsætning til populære kodningsværktøjer |
-| 🎮**Model Legeplads** | Test enhver udbyder/model/slutpunkt fra dashboardet |
-| 🔏**CLI Fingerprint Toggle** | Fingeraftryksmatchning pr. udbyder i Indstillinger > Sikkerhed |
-| 🌐**i18n (30 sprog)** | Fuldt dashboard + understøttelse af docs-sprog med RTL-dækning |
-| 🧹**Ryd alle modeller** | Rydning af modelliste med ét klik i udbyderoplysninger |
-| 👁️**Sidebjælkekontrol**🆕 | Skjul komponenter og integrationer fra Udseendeindstillinger |
-| 📋**Udgaveskabeloner** | Standardiserede GitHub-skabeloner til fejl og funktioner |
-| 📂**Tilpasset datakatalog** | `DATA_DIR` tilsidesættelse for lagerplacering | ### Feature Deep Dive |
+### 🤖 Agent & Protocol Operations (v2.0)
+
+| Feature | What It Does |
+| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
+| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools |
+| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows |
+| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs |
+| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) |
+| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) |
+| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution |
+| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access |
+| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks |
+| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery |
+| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` |
+| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface |
+
+### 🧠 Routing & Intelligence
+
+| Feature | What It Does |
+| ---------------------------------- | ------------------------------------------------------------------------ |
+| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free |
+| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider |
+| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions |
+| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection |
+| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry |
+| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control |
+| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session |
+| 🌐 **Wildcard Router** | `provider/*` dynamic routing |
+| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits |
+| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety |
+| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models |
+| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) |
+| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions |
+| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity |
+| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution |
+| 💬 **System Prompt Injection** | Global behavior controls applied consistently |
+| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows |
+
+### 🎵 Multi-Modal APIs
+
+| Feature | What It Does |
+| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends |
+| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines |
+| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support |
+| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages |
+| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) |
+| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) |
+| 🛡️ **Moderations** | `/v1/moderations` safety checks |
+| 🔀 **Reranking** | `/v1/rerank` for relevance scoring |
+| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache |
+
+### 🛡️ Resilience, Security & Governance
+
+| Feature | What It Does |
+| ----------------------------------- | -------------------------------------------------------------------------------------- |
+| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls |
+| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format |
+| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events |
+| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers |
+| ⚡ **Request Idempotency** | Duplicate protection window |
+| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** |
+| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** |
+| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments |
+| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence |
+| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations |
+| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks |
+| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures |
+| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically |
+| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls |
+| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` |
+| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog |
+
+### 📊 Observability & Analytics
+
+| Feature | What It Does |
+| -------------------------------- | ----------------------------------------------------- |
+| 📝 **Request + Proxy Logging** | Full request/response and proxy logging |
+| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI |
+| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page |
+| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing |
+| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats |
+| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility |
+| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views |
+| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies |
+| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing |
+
+### ☁️ Deployment & Platform
+
+| Feature | What It Does |
+| ------------------------------ | --------------------------------------------------------------------- |
+| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments |
+| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard |
+| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles |
+| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls |
+| 🔄 **Backup/Restore** | Export/import and disaster recovery flows |
+| 🧙 **Onboarding Wizard** | First-run guided setup |
+| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools |
+| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard |
+| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security |
+| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage |
+| 🧹 **Clear All Models** | One-click model list clearing in provider details |
+| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings |
+| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features |
+| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location |
+
+### Feature Deep Dive
#### Smart fallback with practical cost control
@@ -1292,103 +1464,132 @@ Combo: "my-coding-stack"
4. if/kimi-k2-thinking
```
-Når kvote, sats eller sundhed svigter, flytter OmniRoute automatisk til den næste kandidat uden manuel skift.#### Protocol management that is visible and operable
+When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching.
-- MCP + A2A kan findes i brugergrænsefladen og dokumenter (ikke skjult)
-- Protokolstatus API'er afslører live operationelle data (`/api/mcp/*`, `/api/a2a/*`)
-- Dashboards inkluderer handlinger for dag-2 operationer (kombinationsskift, nulstilling af breaker, annullering af opgave)#### Translator + validation workflow
+#### Protocol management that is visible and operable
-Oversætterområdet omfatter:
+- MCP + A2A are discoverable in UI and docs (not hidden)
+- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`)
+- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation)
--**Legeplads**: anmod om transformationstjek -**Chattester**: fuld anmodning/svar tur/retur -**Testbænk**: flere sager på én gang -**Live Monitor**: trafikvisning i realtid
+#### Translator + validation workflow
-Plus protokolvalidering med rigtige klienter via `npm run test:protocols:e2e`.
+The Translator area includes:
-> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Værktøjsreference, IDE-konfigurationer og klienteksempler
+- **Playground**: request transformation checks
+- **Chat Tester**: full request/response round-trip
+- **Test Bench**: multiple cases in one run
+- **Live Monitor**: real-time traffic view
+
+Plus protocol validation with real clients via `npm run test:protocols:e2e`.
+
+> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples
>
-> 📖**[A2A Server README](src/lib/a2a/README.md)**— Færdigheder, JSON-RPC-metoder, streaming og opgavelivscyklus## 🧪 Evaluations (Evals)
+> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle
-OmniRoute inkluderer en indbygget evalueringsramme til at teste LLM-svarkvaliteten mod et gyldent sæt. Få adgang til det via**Analytics → Evals**i dashboardet.### Built-in Golden Set
+## 🧪 Evaluations (Evals)
-Det forudindlæste "OmniRoute Golden Set" indeholder testcases til:
+OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard.
-- Hilsen, matematik, geografi, kodegenerering
-- JSON format compliance, oversættelse, markdown generation
-- Sikkerhedsafvisning (skadeligt indhold), optælling, boolsk logik### Evaluation Strategies
+### Built-in Golden Set
-| Strategi | Beskrivelse | Eksempel |
-| ----------------- | ----------------------------------------------------------------------- | -------------------------------- | --- |
-| 'præcis' | Output skal matche nøjagtigt | `"4"` |
-| `indeholder` | Output skal indeholde understreng (uafhængig af store og små bogstaver) | `"Paris"` |
-| "regex" | Output skal matche regex-mønster | `"1.*2.*3"` |
-| `brugerdefineret` | Brugerdefineret JS-funktion returnerer sand/falsk | `(output) => output.længde > 10` | --- |
+The pre-loaded "OmniRoute Golden Set" contains test cases for:
+
+- Greetings, math, geography, code generation
+- JSON format compliance, translation, markdown generation
+- Safety refusal (harmful content), counting, boolean logic
+
+### Evaluation Strategies
+
+| Strategy | Description | Example |
+| ---------- | ------------------------------------------------ | -------------------------------- |
+| `exact` | Output must match exactly | `"4"` |
+| `contains` | Output must contain substring (case-insensitive) | `"Paris"` |
+| `regex` | Output must match regex pattern | `"1.*2.*3"` |
+| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` |
+
+---
## 📖 Setup Guide
### Protocol Setup (MCP + A2A)
-
-🧩 MCP-opsætning (modelkontekstprotokol)
+
+🧩 MCP Setup (Model Context Protocol)
-Start MCP-transport i stdio-tilstand:```bash
+Start MCP transport in stdio mode:
+
+```bash
omniroute --mcp
+```
-````
+Recommended validation flow:
-Anbefalet valideringsflow:
+1. Connect your MCP client over stdio.
+2. Run `omniroute_get_health`.
+3. Run `omniroute_list_combos`.
+4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit.
-1. Tilslut din MCP-klient via stdio.
-2. Kør `omniroute_get_health`.
-3. Kør `omniroute_list_combos`.
-4. Åbn `/dashboard/mcp` for at bekræfte hjerteslag, aktivitet og audit.
-
-Nyttige API'er til automatisering:
+Useful APIs for automation:
- `GET /api/mcp/status`
- `GET /api/mcp/tools`
- `GET /api/mcp/audit`
-- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats`
-
-🤝 A2A-opsætning (Agent2Agent)
+
-Opdag agenten:```bash
+
+🤝 A2A Setup (Agent2Agent)
+
+Discover the agent:
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
-Send en opgave:```bash
+Send a task:
+
+```bash
curl -X POST http://localhost:20128/a2a \
- -H 'content-type: application/json' \
- -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+ -H 'content-type: application/json' \
+ -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+```
-````
-
-Administrer livscyklus:
+Manage lifecycle:
- `GET /api/a2a/status`
-- `GET /api/a2a/opgaver`
+- `GET /api/a2a/tasks`
- `GET /api/a2a/tasks/:id`
- `POST /api/a2a/tasks/:id/cancel`
-Operationel UI:
+Operational UI:
-- `/dashboard/a2a` til observerbarhed for opgave/tilstand/strøm og røghandlinger
+- `/dashboard/a2a` for task/state/stream observability and smoke actions
-
-🧪 End-to-end protokolvalidering
+
-Valider begge protokoller med rigtige klienter:```bash
+
+🧪 End-to-end protocol validation
+
+Validate both protocols with real clients:
+
+```bash
npm run test:protocols:e2e
-````
+```
-Dette verificerer:
+This verifies:
-- MCP SDK-klient forbinde/liste/opkald
-- A2A opdagelse/send/stream/hent/annuller
-- Krydstjek data i MCP-audit og A2A opgavestyring API'er
+- MCP SDK client connect/list/call
+- A2A discovery/send/stream/get/cancel
+- Cross-check data in MCP audit and A2A task management APIs
-
-💳 Abonnementsudbydere### Claude Code (Pro/Max)
+
+
+
+💳 Subscription Providers
+
+### Claude Code (Pro/Max)
```bash
Dashboard → Providers → Connect Claude Code
@@ -1401,7 +1602,9 @@ Models:
cc/claude-haiku-4-5-20251001
```
-**Prof tip:**Brug Opus til komplekse opgaver, Sonnet for hurtighed. OmniRoute sporer kvote pr. model!### OpenAI Codex (Plus/Pro)
+**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model!
+
+### OpenAI Codex (Plus/Pro)
```bash
Dashboard → Providers → Connect Codex
@@ -1415,20 +1618,22 @@ Models:
#### Codex Account Limit Management (5h + Weekly)
-Hver Codex-konto har nu politikskift i `Dashboard -> Udbydere`:
+Each Codex account now has policy toggles in `Dashboard -> Providers`:
-- `5h` (ON/OFF): håndhæv politikken for 5-timers vinduestærskel.
-- `Ugentligt` (TIL/FRA): håndhæv politikken for tærskelværdi for ugentlige vinduer.
-- Tærskeladfærd: Når et aktiveret vindue når >=90 % brug, springes den konto over.
-- Rotationsadfærd: OmniRoute ruter automatisk til den næste kvalificerede Codex-konto.
-- Nulstil adfærd: Når udbyderens 'resetAt'-tid går, bliver kontoen automatisk kvalificeret igen.
+- `5h` (ON/OFF): enforce the 5-hour window threshold policy.
+- `Weekly` (ON/OFF): enforce the weekly window threshold policy.
+- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped.
+- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically.
+- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically.
-Scenarier:
+Scenarios:
-- `5h ON` + ` Weekly ON`: Konto springes over, når et af vinduerne når tærsklen.
-- `5h OFF` + ` Weekly ON`: kun ugentlig brug kan blokere kontoen.
-- `5h ON` + `Ugentlig OFF`: kun 5-timers brug kan blokere kontoen.
-- `resetAt` bestået: Kontoen går automatisk i rotation igen (ingen manuel genaktivering).### Gemini CLI (FREE 180K/month!)
+- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold.
+- `5h OFF` + `Weekly ON`: only weekly usage can block the account.
+- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account.
+- `resetAt` passed: account re-enters rotation automatically (no manual re-enable).
+
+### Gemini CLI (FREE 180K/month!)
```bash
Dashboard → Providers → Connect Gemini CLI
@@ -1440,7 +1645,9 @@ Models:
gc/gemini-2.5-pro
```
-**Bedste værdi:**Kæmpe gratis niveau! Brug dette før betalte niveauer.### GitHub Copilot
+**Best Value:** Huge free tier! Use this before paid tiers.
+
+### GitHub Copilot
```bash
Dashboard → Providers → Connect GitHub
@@ -1455,71 +1662,91 @@ Models:
-
-🔑 API-nøgleudbydere### NVIDIA NIM (FREE developer access — 70+ models)
+
+🔑 API Key Providers
-1. Tilmeld dig: [build.nvidia.com](https://build.nvidia.com)
-2. Få gratis API-nøgle (1000 slutningskreditter inkluderet)
-3. Dashboard → Tilføj udbyder → NVIDIA NIM:
- - API-nøgle: `nvapi-din-nøgle`
+### NVIDIA NIM (FREE developer access — 70+ models)
-**Modeller:**"nvidia/llama-3.3-70b-instruct", "nvidia/mistral-7b-instruct" og 50+ flere
+1. Sign up: [build.nvidia.com](https://build.nvidia.com)
+2. Get free API key (1000 inference credits included)
+3. Dashboard → Add Provider → NVIDIA NIM:
+ - API Key: `nvapi-your-key`
-**Prof tip:**OpenAI-kompatibel API — fungerer problemfrit med OmniRoutes formatoversættelse!### DeepSeek
+**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more
-1. Tilmeld dig: [platform.deepseek.com](https://platform.deepseek.com)
-2. Hent API-nøgle
-3. Dashboard → Tilføj udbyder → DeepSeek
+**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation!
-**Modeller:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!)
+### DeepSeek
-1. Tilmeld dig: [console.groq.com](https://console.groq.com)
-2. Få API-nøgle (gratis niveau inkluderet)
-3. Dashboard → Tilføj udbyder → Groq
+1. Sign up: [platform.deepseek.com](https://platform.deepseek.com)
+2. Get API key
+3. Dashboard → Add Provider → DeepSeek
-**Modeller:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b`
+**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder`
-**Prof tip:**Ultrahurtig inferens — bedst til realtidskodning!### OpenRouter (100+ Models)
+### Groq (Free Tier Available!)
-1. Tilmeld dig: [openrouter.ai](https://openrouter.ai)
-2. Hent API-nøgle
-3. Dashboard → Tilføj udbyder → OpenRouter
+1. Sign up: [console.groq.com](https://console.groq.com)
+2. Get API key (free tier included)
+3. Dashboard → Add Provider → Groq
-**Modeller:**Få adgang til mere end 100 modeller fra alle større udbydere via en enkelt API-nøgle.
+**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b`
-**Dashboard-adfærd:**OpenRouter-modeller administreres fra**Tilgængelige modeller**. Manuel tilføjelse, import og automatisk synkronisering opdaterer alle den samme liste.
+**Pro Tip:** Ultra-fast inference — best for real-time coding!
-
-💰 Billige udbydere (backup)### GLM-4.7 (Daily reset, $0.6/1M)
+### OpenRouter (100+ Models)
-1. Tilmeld dig: [Zhipu AI](https://open.bigmodel.cn/)
-2. Hent API-nøgle fra Coding Plan
-3. Dashboard → Tilføj API-nøgle:
- - Udbyder: `glm`
- - API-nøgle: `din-nøgle`
+1. Sign up: [openrouter.ai](https://openrouter.ai)
+2. Get API key
+3. Dashboard → Add Provider → OpenRouter
-**Brug:**`glm/glm-4.7`
+**Models:** Access 100+ models from all major providers through a single API key.
-**Pro-tip:**Coding Plan tilbyder 3× kvote til 1/7 pris! Nulstil dagligt 10:00.### MiniMax M2.1 (5h reset, $0.20/1M)
+**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list.
-1. Tilmeld dig: [MiniMax](https://www.minimax.io/)
-2. Hent API-nøgle
-3. Dashboard → Tilføj API-nøgle
+
-**Brug:**`minimax/MiniMax-M2.1`
+
+💰 Cheap Providers (Backup)
-**Prof tip:**Billigste mulighed for lang sammenhæng (1M tokens)!### Kimi K2 ($9/month flat)
+### GLM-4.7 (Daily reset, $0.6/1M)
-1. Abonner: [Moonshot AI](https://platform.moonshot.ai/)
-2. Hent API-nøgle
-3. Dashboard → Tilføj API-nøgle
+1. Sign up: [Zhipu AI](https://open.bigmodel.cn/)
+2. Get API key from Coding Plan
+3. Dashboard → Add API Key:
+ - Provider: `glm`
+ - API Key: `your-key`
-**Brug:**`kimi/kimi-latest`
+**Use:** `glm/glm-4.7`
-**Prof tip:**Fast $9/måned for 10M tokens = $0,90/1M effektive omkostninger!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM.
-
-🆓 GRATIS udbydere (nødbackup)### Qoder (5 FREE models via OAuth)
+### MiniMax M2.1 (5h reset, $0.20/1M)
+
+1. Sign up: [MiniMax](https://www.minimax.io/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `minimax/MiniMax-M2.1`
+
+**Pro Tip:** Cheapest option for long context (1M tokens)!
+
+### Kimi K2 ($9/month flat)
+
+1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `kimi/kimi-latest`
+
+**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost!
+
+
+
+
+🆓 FREE Providers (Emergency Backup)
+
+### Qoder (5 FREE models via OAuth)
```bash
Dashboard → Connect Qoder
@@ -1560,8 +1787,10 @@ Models:
-
-🎨 Opret kombinationer### Example 1: Maximize Subscription → Cheap Backup
+
+🎨 Create Combos
+
+### Example 1: Maximize Subscription → Cheap Backup
```
Dashboard → Combos → Create New
@@ -1589,8 +1818,10 @@ Cost: $0 forever!
-
-🔧 CLI-integration### Cursor IDE
+
+🔧 CLI Integration
+
+### Cursor IDE
```
Settings → Models → Advanced:
@@ -1601,7 +1832,9 @@ Settings → Models → Advanced:
### Claude Code
-Brug siden**CLI Tools**i dashboardet til konfiguration med et enkelt klik, eller rediger `~/.claude/settings.json` manuelt.### Codex CLI
+Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually.
+
+### Codex CLI
```bash
export OPENAI_BASE_URL="http://localhost:20128"
@@ -1612,12 +1845,15 @@ codex "your prompt"
### OpenClaw
-**Mulighed 1 — Dashboard (anbefalet):**```
+**Option 1 — Dashboard (recommended):**
+
+```
Dashboard → CLI Tools → OpenClaw → Select Model → Apply
+```
-````
+**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`:
-**Mulighed 2 — Manuel:**Rediger `~/.openclaw/openclaw.json`:```json
+```json
{
"models": {
"providers": {
@@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply
}
}
}
-````
+```
-> **Bemærk:**OpenClaw fungerer kun med lokale OmniRoute. Brug `127.0.0.1` i stedet for `localhost` for at undgå problemer med IPv6-opløsning.### Cline / Continue / RooCode
+> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues.
+
+### Cline / Continue / RooCode
```
Settings → API Configuration:
@@ -1643,15 +1881,17 @@ Settings → API Configuration:
### OpenCode
-**Trin 1:**Tilføj OmniRoute som en tilpasset udbyder:```bash
+**Step 1:** Add OmniRoute as a custom provider:
+
+```bash
opencode
/connect
-
# Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key
+```
-````
+**Step 2:** Create/edit `opencode.json` in your project root:
-**Trin 2:**Opret/rediger `opencode.json` i dit projektrod:```json
+```json
{
"$schema": "https://opencode.ai/config.json",
"provider": {
@@ -1669,117 +1909,130 @@ opencode
}
}
}
-````
+```
-**Trin 3:**Vælg modellen i OpenCode:```bash
+**Step 3:** Select the model in OpenCode:
+
+```bash
/models
-
# Select any OmniRoute model from the list
+```
-````
+> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard.
->**Tip:**Tilføj en hvilken som helst model, der er tilgængelig i dit OmniRoute `/v1/models` slutpunkt til sektionen `modeller`. Brug formatet `provider/model-id` fra dit OmniRoute-dashboard.
+
---
## Fejlfinding
-
-Klik for at udvide fejlfindingsvejledningen
+
+Click to expand troubleshooting guide
-**"Sprogmodellen leverede ikke beskeder"**
+**"Language model did not provide messages"**
-- Udbyderkvote opbrugt → Tjek dashboardkvotesporing
-- Løsning: Brug combo fallback eller skift til et billigere niveau
+- Provider quota exhausted → Check dashboard quota tracker
+- Solution: Use combo fallback or switch to cheaper tier
-**Satsbegrænsende**
+**Rate limiting**
-- Abonnementskontingent ude → Fallback til GLM/MiniMax
-- Tilføj combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Subscription quota out → Fallback to GLM/MiniMax
+- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-**OAuth-token er udløbet**
+**OAuth token expired**
-- Automatisk genopfrisket af OmniRoute
-- Hvis problemerne fortsætter: Dashboard → Udbyder → Genopret forbindelse
+- Auto-refreshed by OmniRoute
+- If issues persist: Dashboard → Provider → Reconnect
-**Høje omkostninger**
+**High costs**
-- Tjek brugsstatistik i Dashboard → Omkostninger
-- Skift primær model til GLM/MiniMax
-- Brug gratis niveau (Gemini CLI, Qoder) til ikke-kritiske opgaver
+- Check usage stats in Dashboard → Costs
+- Switch primary model to GLM/MiniMax
+- Use free tier (Gemini CLI, Qoder) for non-critical tasks
-**Dashboard/API-porte er forkerte**
+**Dashboard/API ports are wrong**
-- `PORT` er den kanoniske basisport (og API-port som standard)
-- `API_PORT` tilsidesætter kun OpenAI-kompatibel API-lytter
-- `DASHBOARD_PORT` tilsidesætter kun dashboard/Next.js-lytter
-- Indstil `NEXT_PUBLIC_BASE_URL` til dit dashboard/offentlige URL (til OAuth-tilbagekald)
+- `PORT` is the canonical base port (and API port by default)
+- `API_PORT` overrides only OpenAI-compatible API listener
+- `DASHBOARD_PORT` overrides only dashboard/Next.js listener
+- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks)
-**Skysynkroniseringsfejl**
+**Cloud sync errors**
-- Bekræft, at `BASE_URL` peger på din kørende instans
-- Bekræft `CLOUD_URL` peger på dit forventede cloud-slutpunkt
-- Hold `NEXT_PUBLIC_*`-værdier på linje med værdier på serversiden
+- Verify `BASE_URL` points to your running instance
+- Verify `CLOUD_URL` points to your expected cloud endpoint
+- Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**Første login virker ikke**
+**First login not working**
-- Tjek `INITIAL_PASSWORD` i `.env`
-- Hvis den ikke er indstillet, er reserveadgangskoden "123456".
+- Check `INITIAL_PASSWORD` in `.env`
+- If unset, fallback password is `123456`
-**Ingen anmodningslogfiler**
+**No request logs**
-- Anmodningsartefakter skrives til `DATA_DIR/call_logs/` som én JSON-fil pr. anmodning
-- Aktiver pipeline capture fra Dashboard → Logs → Request Logs, hvis du har brug for detaljerede per-stage payloads
-- Indstil `APP_LOG_TO_FILE=true`, hvis du også vil have applikationskonsollogfiler i `logs/application/app.log`
-- Juster `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` og `CALL_LOG_MAX_ENTRIES` efter behov
+- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request
+- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads
+- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log`
+- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed
-**Forbindelsestest viser "Ugyldig" for OpenAI-kompatible udbydere**
+**Connection test shows "Invalid" for OpenAI-compatible providers**
-- Mange udbydere afslører ikke et `/models` slutpunkt
-- OmniRoute v1.0.6+ inkluderer fallback-validering via chatafslutninger
-- Sørg for, at basis-URL'en indeholder `/v1`-suffiks### 🔐 OAuth on a Remote Server
+- Many providers don't expose a `/models` endpoint
+- OmniRoute v1.0.6+ includes fallback validation via chat completions
+- Ensure base URL includes `/v1` suffix
+
+### 🔐 OAuth on a Remote Server
-
+
->**⚠️ Vigtigt for brugere, der kører OmniRoute på en VPS, Docker eller enhver ekstern server**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
+> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server**
-**Antigravity**og**Gemini CLI**-udbyderne bruger**Google OAuth 2.0**. Google kræver, at `redirect_uri` i OAuth-flowet nøjagtigt matcher en af de forudregistrerede URI'er i appens Google Cloud Console.
+#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
-OAuth-legitimationsoplysningerne, der er bundtet i OmniRoute, er kun registreret**for 'localhost'**. Når du får adgang til OmniRoute på en ekstern server (f.eks. `https://omniroute.myserver.com`), afviser Google godkendelsen med:```
+The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console.
+
+The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solution: Configure your own OAuth credentials
-Du skal oprette et**OAuth 2.0 Client ID**i Google Cloud Console med din servers URI.#### Step-by-step
+You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI.
-**1. Åbn Google Cloud Console**
+#### Step-by-step
-Gå til: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
+**1. Open Google Cloud Console**
-**2. Opret et nyt OAuth 2.0-klient-id**
+Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-- Klik på**"+ Opret legitimationsoplysninger"**→**"OAuth-klient-id"**
-- Ansøgningstype:**"Webapplikation"**
-- Navn: alt, hvad du kan lide (f.eks. `OmniRoute Remote`)
+**2. Create a new OAuth 2.0 Client ID**
-**3. Tilføj autoriserede omdirigerings-URI'er**
+- Click **"+ Create Credentials"** → **"OAuth client ID"**
+- Application type: **"Web application"**
+- Name: anything you like (e.g. `OmniRoute Remote`)
-I feltet**"Autoriserede omdirigerings-URI'er"**skal du tilføje:```
+**3. Add Authorized Redirect URIs**
+
+In the **"Authorized redirect URIs"** field, add:
+
+```
https://your-server.com/callback
+```
-````
+> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`).
-> Erstat `din-server.com` med din servers domæne eller IP (medtag porten, hvis det er nødvendigt, f.eks. `http://45.33.32.156:20128/callback`).
+**4. Save and copy the credentials**
-**4. Gem og kopier legitimationsoplysningerne**
+After creating, Google will show the **Client ID** and **Client Secret**.
-Efter oprettelse vil Google vise**klient-id**og**klienthemmelighed**.
+**5. Set environment variables**
-**5. Indstil miljøvariabler**
+In your `.env` (or Docker environment variables):
-I dine `.env` (eller Docker-miljøvariabler):```bash
+```bash
# For Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
@@ -1788,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
-````
+```
-**6. Genstart OmniRoute**```bash
+**6. Restart OmniRoute**
+```bash
# npm:
-
npm run dev
# Docker:
-
docker restart omniroute
+```
-````
+**7. Try connecting again**
-**7. Prøv at oprette forbindelse igen**
+Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth
-Dashboard → Udbydere → Antigravity (eller Gemini CLI) → OAuth
+Google will now redirect correctly to `https://your-server.com/callback`.
-Google vil nu omdirigere korrekt til `https://din-server.com/callback`.---
+---
#### Temporary workaround (without custom credentials)
-Hvis du ikke vil konfigurere dine egne legitimationsoplysninger lige nu, kan du stadig bruge det**manuelle URL-flow**:
+If you don't want to set up your own credentials right now, you can still use the **manual URL flow**:
-1. OmniRoute åbner Googles autorisations-URL
-2. Efter godkendelse forsøger Google at omdirigere til `localhost` (som fejler på fjernserveren)
-3.**Kopiér den fulde URL**fra din browsers adresselinje (også selvom siden ikke indlæses)
-4. Indsæt denne URL i feltet vist i OmniRoute-forbindelsesmodal
-5. Klik på**"Forbind"**
+1. OmniRoute opens the Google authorization URL
+2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server)
+3. **Copy the full URL** from your browser's address bar (even if the page doesn't load)
+4. Paste that URL into the field shown in the OmniRoute connection modal
+5. Click **"Connect"**
-> Dette virker, fordi autorisationskoden i URL'en er gyldig, uanset om omdirigeringssiden er indlæst.---
+> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded.
-
-🇧🇷 Versão em Português
#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+---
-Os testedores**Antigravity**og**Gemini CLI**usam**Google OAuth 2.0**for autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja**exatamente**uma das URIs pré-cadastradas no Google Cloud Console do aplicativo.
+
+🇧🇷 Versão em Português
-Som credenciais OAuth embutidas no OmniRoute estão cadastradas**apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google afviser en autenticação com:```
+#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+
+Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo.
+
+As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solução: Configure suas próprias credenciais OAuth
-Você precisa criar um**OAuth 2.0 Client ID**ingen Google Cloud Console med en URI, der udfører denne service.#### Passo a passo
+Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor.
-**1. Adgang til Google Cloud Console**
+#### Passo a passo
+
+**1. Acesse o Google Cloud Console**
Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
**2. Crie um novo OAuth 2.0 Client ID**
-- Klik på dem**"+ Opret legitimationsoplysninger"**→**"OAuth-klient-id"**
-- Tipo de aplicativo:**"Webapplikation"**
-- Navn: escolha qualquer nome (eks.: `OmniRoute Remote`)
+- Clique em **"+ Create Credentials"** → **"OAuth client ID"**
+- Tipo de aplicativo: **"Web application"**
+- Nome: escolha qualquer nome (ex: `OmniRoute Remote`)
-**3. Adicione som autoriseret omdirigerings-URI**
+**3. Adicione as Authorized Redirect URIs**
-Ingen campo**"Autoriseret omdirigerings-URI'er"**, adicione:```
+No campo **"Authorized redirect URIs"**, adicione:
+
+```
https://seu-servidor.com/callback
+```
-````
+> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`).
-> Substitua `seu-servidor.com` pelo domínio eller IP do seu servidor (inklusive en porta se necessário, f.eks: `http://45.33.32.156:20128/callback`).
+**4. Salve e copie as credenciais**
-**4. Salve e copy as credenciais**
+Após criar, o Google mostrará o **Client ID** e o **Client Secret**.
-Após criar, o Google mostrará o**Client ID**e o**Client Secret**.
+**5. Configure as variáveis de ambiente**
-**5. Konfigurer som variáveis de ambiente**
+No seu `.env` (ou nas variáveis de ambiente do Docker):
-No seu `.env` (ou nas variáveis de ambiente do Docker):```bash
+```bash
# Para Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
@@ -1867,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
-````
+```
-**6. Reinicie o OmniRoute**```bash
+**6. Reinicie o OmniRoute**
+```bash
# Se usando npm:
-
npm run dev
# Se usando Docker:
-
docker restart omniroute
-
-````
+```
**7. Tente conectar novamente**
-Dashboard → Udbydere → Antigravity (ou Gemini CLI) → OAuth
+Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth
-Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` og autenticação funcionará.---
+Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará.
+
+---
#### Workaround temporário (sem configurar credenciais próprias)
-Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo**manual de URL**:
+Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**:
-1. O OmniRoute abrirá en URL de autorização til Google
-2. Após você autorizar, o Google tentará redirecionar for `localhost` (que falha no servidor remoto)
-3.**Kopier en URL komplet**da barra de endereço do sin browser (mesmo que a página não carregue)
+1. O OmniRoute abrirá a URL de autorização do Google
+2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto)
+3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue)
4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute
-5. Klik på**"Forbind"**
+5. Clique em **"Connect"**
-> Este workaround funciona porque or código de autorização na URL é válido independente do redirect ter carregado or não.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+
+
---
@@ -1905,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux
## 🛠️ Tech Stack
-
-Klik for at udvide tekniske stakdetaljer
+
+Click to expand tech stack details
--**Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ er**ikke understøttet**— "better-sqlite3" native binære filer er inkompatible)
--**Sprog**: TypeScript 5.9 —**100 % TypeScript**på tværs af `src/` og `open-sse/` (nul `enhver` i kernemoduler siden v2.0)
--**Framework**: Next.js 16 + React 19 + Tailwind CSS 4
--**Database**: LowDB (JSON) + SQLite (domænetilstand + proxylogfiler + MCP-revision + routingbeslutninger)
--**Skemaer**: Zod (MCP-værktøj I/O-validering, API-kontrakter)
--**Protokoller**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
--**Streaming**: Server-sendte hændelser (SSE)
--**Auth**: OAuth 2.0 (PKCE) + JWT + API-nøgler + MCP Scoped Authorization
--**Test**: Node.js testløber + Vitest (900+ tests inklusive enhed, integration, E2E)
--**CI/CD**: GitHub-handlinger (automatisk npm-udgivelse + Docker Hub ved udgivelse)
--**Websted**: [omniroute.online](https://omniroute.online)
--**Pakke**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
--**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
--**Resiliens**: Circuit breaker, eksponentiel backoff, anti-tordenbesætning, TLS spoofing, auto-combo selvhelbredelse
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible)
+- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0)
+- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4
+- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions)
+- **Schemas**: Zod (MCP tool I/O validation, API contracts)
+- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
+- **Streaming**: Server-Sent Events (SSE)
+- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization
+- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E)
+- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release)
+- **Website**: [omniroute.online](https://omniroute.online)
+- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
+- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
+- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing
+
+
---
## Dokumentation
-| Dokument | Beskrivelse |
-| ------------------------------------------------------ | ---------------------------------------------------------- |
-| [Brugervejledning](docs/USER_GUIDE.md) | Udbydere, kombinationer, CLI-integration, implementering |
-| [API-reference](docs/API_REFERENCE.md) | Alle endepunkter med eksempler |
-| [MCP-server](open-sse/mcp-server/README.md) | 16 MCP-værktøjer, IDE-konfigurationer, Python/TS/Go-klienter |
-| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protokol, færdigheder, streaming, opgavestyring |
-| [Auto-Combo Engine](docs/auto-combo.md) | 6-faktor scoring, tilstandspakker, selvhelbredende |
-| [Fejlfinding](docs/TROUBLESHOOTING.md) | Almindelige problemer og løsninger |
-| [Arkitektur](docs/ARCHITECTURE.md) | Systemarkitektur og indre |
-| [Bidrager](BIDRØRENDE.md) | Udviklingsopsætning og retningslinjer |
-| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0-specifikation |
-| [Sikkerhedspolitik](SECURITY.md) | Sårbarhedsrapportering og sikkerhedspraksis |
-| [VM-implementering](docs/VM_DEPLOYMENT_GUIDE.md) | Komplet guide: VM + nginx + Cloudflare opsætning |
-| [Feature Gallery](docs/FEATURES.md) | Visuel dashboard-rundvisning med skærmbilleder |
-| [Udgivelsestjekliste](docs/RELEASE_CHECKLIST.md) | Pre-release valideringstrin |---
+| Document | Description |
+| ---------------------------------------------- | --------------------------------------------------- |
+| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment |
+| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples |
+| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients |
+| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt |
+| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing |
+| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation |
+| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions |
+| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals |
+| [Contributing](CONTRIBUTING.md) | Development setup and guidelines |
+| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification |
+| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices |
+| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup |
+| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots |
+| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps |
+
+---
## 🗺️ Roadmap
-OmniRoute har**210+ funktioner planlagt**på tværs af flere udviklingsfaser. Her er nøgleområderne:
+OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas:
-| Kategori | Planlagte funktioner | Højdepunkter |
-| ------------------------------ | ---------------- | ---------------------------------------------------------------------------------------------- |
-| 🧠**Routing & intelligens**| 25+ | Routing med laveste latens, tag-baseret routing, kvote preflight, valg af P2C-konto |
-| 🔒**Sikkerhed og overholdelse**| 20+ | SSRF-hærdning, tilsløring af legitimationsoplysninger, hastighedsgrænse pr. slutpunkt, styringsnøgleomfang |
-| 📊**Observabilitet**| 15+ | OpenTelemetry-integration, kvoteovervågning i realtid, omkostningssporing pr. model |
-| 🔄**Udbyderintegrationer**| 20+ | Dynamisk modelregistrering, udbydernedkøling, multi-konto Codex, Copilot-kvoteparsing |
-| ⚡**Ydeevne**| 15+ | Dobbelt cachelag, promptcache, svarcache, streaming keepalive, batch API |
-| 🌐**Økosystem**| 10+ | WebSocket API, config hot-reload, distribueret config butik, kommerciel tilstand |### 🔜 Coming Soon
+| Category | Planned Features | Highlights |
+| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- |
+| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection |
+| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping |
+| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model |
+| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing |
+| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API |
+| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode |
-- 🔗**OpenCode-integration**— Native udbyderunderstøttelse af OpenCode AI-kodnings-IDE
-- 🔗**TRAE-integration**— Fuld understøttelse af TRAE AI-udviklingsrammen
-- 📦**Batch API**— Asynkron batchbehandling til masseanmodninger
-- 🎯**Tag-baseret Routing**— Ruteanmodninger baseret på tilpassede tags og metadata
-- 💰**Laveste omkostningsstrategi**— Vælg automatisk den billigste tilgængelige udbyder
+### 🔜 Coming Soon
-> 📝 Fuld funktionsspecifikationer tilgængelige i [`docs/new-features/`](docs/new-features/) (217 detaljerede specifikationer)---
+- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE
+- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework
+- 📦 **Batch API** — Asynchronous batch processing for bulk requests
+- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata
+- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider
+
+> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs)
+
+---
## 👥 Contributors
@@ -1970,18 +2245,20 @@ OmniRoute har**210+ funktioner planlagt**på tværs af flere udviklingsfaser. He
### How to Contribute
-1. Fork depotet
-2. Opret din feature-gren (`git checkout -b feature/amazing-feature`)
-3. Bekræft dine ændringer (`git commit -m 'Tilføj fantastisk funktion'`)
-4. Skub til grenen ("git push origin feature/amazing-feature")
-5. Åbn en pull-anmodning
+1. Fork the repository
+2. Create your feature branch (`git checkout -b feature/amazing-feature`)
+3. Commit your changes (`git commit -m 'Add amazing feature'`)
+4. Push to the branch (`git push origin feature/amazing-feature`)
+5. Open a Pull Request
-Se [CONTRIBUTING.md](CONTRIBUTING.md) for detaljerede retningslinjer.### Releasing a New Version
+See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines.
+
+### Releasing a New Version
```bash
# Create a release — npm publish happens automatically
gh release create v2.0.0 --title "v2.0.0" --generate-notes
-````
+```
---
@@ -1993,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes
## 🙏 Acknowledgments
-Særlig tak til**[9router](https://github.com/decolua/9router)**af**[decolua](https://github.com/decolua)**— det originale projekt, der inspirerede denne gaffel. OmniRoute bygger på det utrolige fundament med yderligere funktioner, multimodale API'er og en fuld TypeScript-omskrivning.
+Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite.
-Særlig tak til**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— den originale Go-implementering, der inspirerede denne JavaScript-port.---
+Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port.
+
+---
## Licens
-MIT-licens - se [LICENS](LICENS) for detaljer.---
+MIT License - see [LICENSE](LICENSE) for details.
+
+---
Built with ❤️ for developers who code 24/7
diff --git a/docs/i18n/da/docs/ARCHITECTURE.md b/docs/i18n/da/docs/ARCHITECTURE.md
index 223d04e314..5713fd9cd6 100644
--- a/docs/i18n/da/docs/ARCHITECTURE.md
+++ b/docs/i18n/da/docs/ARCHITECTURE.md
@@ -4,80 +4,93 @@
---
-_Sidst opdateret: 2026-03-28_## Executive Summary
-OmniRoute er en lokal AI-routinggateway og dashboard bygget på Next.js.
-Det giver et enkelt OpenAI-kompatibelt slutpunkt (`/v1/*`) og dirigerer trafik på tværs af flere upstream-udbydere med oversættelse, fallback, token-opdatering og brugssporing.
-Kerneegenskaber:
+_Last updated: 2026-03-28_
-- OpenAI-kompatibel API-overflade til CLI/værktøjer (28 udbydere)
-- Anmodning/svar oversættelse på tværs af udbyderformater
-- Model combo fallback (multi-model sekvens)
-- Fallback på kontoniveau (multi-konto pr. udbyder)
-- Administration af forbindelse til OAuth + API-nøgleudbyder
-- Indlejringsgenerering via `/v1/embeddings` (6 udbydere, 9 modeller)
-- Billedgenerering via `/v1/images/generations` (4 udbydere, 9 modeller)
-- Tænk tag-parsing (`
...`) for ræsonneringsmodeller
-- Response sanitization for streng OpenAI SDK-kompatibilitet
-- Rollenormalisering (udvikler→system, system→bruger) for kompatibilitet på tværs af udbydere
-- Struktureret outputkonvertering (json_schema → Gemini responseSchema)
-- Lokal persistens for udbydere, nøgler, aliaser, kombinationer, indstillinger, priser
-- Brug/omkostningssporing og anmodningslogning
-- Valgfri skysynkronisering til synkronisering af flere enheder/tilstande
-- IP-tilladelsesliste/blokeringsliste til API-adgangskontrol
-- Tænkende budgetstyring (passthrough/auto/custom/adaptive)
-- Global system prompt injektion
-- Sessionssporing og fingeraftryk
-- Forbedret prisbegrænsning pr. konto med udbyderspecifikke profiler
-- Circuit breaker mønster for udbyderens modstandsdygtighed
-- Anti-tordenbeskyttelse med mutex-låsning
-- Signaturbaseret anmodnings deduplikeringscache
-- Domænelag: modeltilgængelighed, omkostningsregler, fallback-politik, lockout-politik
-- Vedvarende domænetilstand (SQLite-gennemskrivningscache til fallbacks, budgetter, lockouts, strømafbrydere)
-- Politikmotor til centraliseret anmodningsevaluering (lockout → budget → fallback)
-- Anmod om telemetri med p50/p95/p99 latency aggregering
-- Korrelations-ID (X-Request-Id) til ende-til-ende-sporing
-- Overholdelsesrevisionslogning med opt-out pr. API-nøgle
-- Evalueringsramme for LLM kvalitetssikring
-- Resilience UI-dashboard med strømafbryderstatus i realtid
-- Modulære OAuth-udbydere (12 individuelle moduler under `src/lib/oauth/providers/`)
+## Executive Summary
-Primær runtime model:
+OmniRoute is a local AI routing gateway and dashboard built on Next.js.
+It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking.
-- Next.js app-ruter under `src/app/api/*` implementerer både dashboard-API'er og kompatibilitets-API'er
-- En delt SSE/routingkerne i `src/sse/*` + `open-sse/*` håndterer udbyderens udførelse, oversættelse, streaming, fallback og brug## Scope and Boundaries
+Core capabilities:
+
+- OpenAI-compatible API surface for CLI/tools (28 providers)
+- Request/response translation across provider formats
+- Model combo fallback (multi-model sequence)
+- Account-level fallback (multi-account per provider)
+- OAuth + API-key provider connection management
+- Embedding generation via `/v1/embeddings` (6 providers, 9 models)
+- Image generation via `/v1/images/generations` (4 providers, 9 models)
+- Think tag parsing (`
...`) for reasoning models
+- Response sanitization for strict OpenAI SDK compatibility
+- Role normalization (developer→system, system→user) for cross-provider compatibility
+- Structured output conversion (json_schema → Gemini responseSchema)
+- Local persistence for providers, keys, aliases, combos, settings, pricing
+- Usage/cost tracking and request logging
+- Optional cloud sync for multi-device/state sync
+- IP allowlist/blocklist for API access control
+- Thinking budget management (passthrough/auto/custom/adaptive)
+- Global system prompt injection
+- Session tracking and fingerprinting
+- Per-account enhanced rate limiting with provider-specific profiles
+- Circuit breaker pattern for provider resilience
+- Anti-thundering herd protection with mutex locking
+- Signature-based request deduplication cache
+- Domain layer: model availability, cost rules, fallback policy, lockout policy
+- Context Relay: session handoff summaries for account rotation continuity
+- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers)
+- Policy engine for centralized request evaluation (lockout → budget → fallback)
+- Request telemetry with p50/p95/p99 latency aggregation
+- Correlation ID (X-Request-Id) for end-to-end tracing
+- Compliance audit logging with opt-out per API key
+- Eval framework for LLM quality assurance
+- Resilience UI dashboard with real-time circuit breaker status
+- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`)
+
+Primary runtime model:
+
+- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs
+- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage
+
+## Scope and Boundaries
### In Scope
-- Lokal gateway køretid
-- Dashboard management API'er
-- Udbydergodkendelse og tokenopdatering
-- Anmod om oversættelse og SSE-streaming
-- Lokal stat + vedvarende brug
-- Valgfri skysynkroniseringsorkestrering### Out of Scope
+- Local gateway runtime
+- Dashboard management APIs
+- Provider authentication and token refresh
+- Request translation and SSE streaming
+- Local state + usage persistence
+- Optional cloud sync orchestration
-- Implementering af skytjenester bag `NEXT_PUBLIC_CLOUD_URL`
-- Udbyder SLA/kontrolplan uden for lokal proces
-- Eksterne CLI-binære filer selv (Claude CLI, Codex CLI osv.)## Dashboard Surface (Current)
+### Out of Scope
-Hovedsider under `src/app/(dashboard)/dashboard/`:
+- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL`
+- Provider SLA/control plane outside local process
+- External CLI binaries themselves (Claude CLI, Codex CLI, etc.)
-- `/dashboard` — hurtig start + udbyderoversigt
-- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint faner
-- `/dashboard/providers` — udbyderforbindelser og legitimationsoplysninger
-- `/dashboard/combos` — kombinationsstrategier, skabeloner, modelrutingsregler
-- `/dashboard/costs` — prissammenlægning og prissynlighed
-- `/dashboard/analytics` — brugsanalyse og -evalueringer
-- `/dashboard/limits` — kvote-/satskontrol
+## Dashboard Surface (Current)
+
+Main pages under `src/app/(dashboard)/dashboard/`:
+
+- `/dashboard` — quick start + provider overview
+- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs
+- `/dashboard/providers` — provider connections and credentials
+- `/dashboard/combos` — combo strategies, templates, model routing rules
+- `/dashboard/costs` — cost aggregation and pricing visibility
+- `/dashboard/analytics` — usage analytics and evaluations
+- `/dashboard/limits` — quota/rate controls
- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation
-- `/dashboard/agents` — opdagede ACP-agenter + tilpasset agentregistrering
-- `/dashboard/media` — billed-/video-/musiklegeplads
-- `/dashboard/search-tools` — test af søgeudbydere og historik
-- `/dashboard/health` — oppetid, strømafbrydere, hastighedsgrænser
+- `/dashboard/agents` — detected ACP agents + custom agent registration
+- `/dashboard/media` — image/video/music playground
+- `/dashboard/search-tools` — search provider testing and history
+- `/dashboard/health` — uptime, circuit breakers, rate limits
- `/dashboard/logs` — request/proxy/audit/console logs
-- `/dashboard/indstillinger` — systemindstillinger faner (generelt, routing, kombinationsstandarder osv.)
-- `/dashboard/api-manager` — API-nøglelivscyklus og modeltilladelser## High-Level System Context
+- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.)
+- `/dashboard/api-manager` — API key lifecycle and model permissions
+
+## High-Level System Context
```mermaid
flowchart LR
@@ -129,139 +142,151 @@ flowchart LR
## 1) API and Routing Layer (Next.js App Routes)
-Hovedmapper:
+Main directories:
-- `src/app/api/v1/*` og `src/app/api/v1beta/*` til kompatibilitets-API'er
-- `src/app/api/*` til administrations-/konfigurations-API'er
-- Næste omskrivninger i `next.config.mjs` map `/v1/*` til `/api/v1/*`
+- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs
+- `src/app/api/*` for management/configuration APIs
+- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*`
-Vigtige kompatibilitetsruter:
+Important compatibility routes:
- `src/app/api/v1/chat/completions/route.ts`
- `src/app/api/v1/messages/route.ts`
- `src/app/api/v1/responses/route.ts`
-- `src/app/api/v1/models/route.ts` – inkluderer brugerdefinerede modeller med `custom: true`
-- `src/app/api/v1/embeddings/route.ts` — indlejringsgenerering (6 udbydere)
-- `src/app/api/v1/images/generations/route.ts` — billedgenerering (4+ udbydere inkl. Antigravity/Nebius)
+- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true`
+- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers)
+- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius)
- `src/app/api/v1/messages/count_tokens/route.ts`
-- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedikeret chat pr. udbyder
-- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedikerede indlejringer pr. udbyder
-- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedikerede billeder pr. udbyder
+- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat
+- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings
+- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images
- `src/app/api/v1beta/models/route.ts`
- `src/app/api/v1beta/models/[...path]/route.ts`
-Ledelsesdomæner:
+Management domains:
-- Godkendelse/indstillinger: `src/app/api/auth/*`, `src/app/api/settings/*`
-- Udbydere/forbindelser: `src/app/api/providers*`
-- Provider noder: `src/app/api/provider-nodes*`
-- Brugerdefinerede modeller: `src/app/api/provider-models` (GET/POST/DELETE)
-- Modelkatalog: `src/app/api/models/route.ts` (GET)
+- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*`
+- Providers/connections: `src/app/api/providers*`
+- Provider nodes: `src/app/api/provider-nodes*`
+- Custom models: `src/app/api/provider-models` (GET/POST/DELETE)
+- Model catalog: `src/app/api/models/route.ts` (GET)
- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
- OAuth: `src/app/api/oauth/*`
- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing`
-- Brug: `src/app/api/usage/*`
+- Usage: `src/app/api/usage/*`
- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*`
-- CLI-værktøjshjælpere: `src/app/api/cli-tools/*`
-- IP-filter: `src/app/api/settings/ip-filter` (GET/PUT)
-- Tænkebudget: `src/app/api/settings/thinking-budget` (GET/PUT)
-- Systemprompt: `src/app/api/settings/system-prompt` (GET/PUT)
-- Sessioner: `src/app/api/sessions` (GET)
-- Satsgrænser: `src/app/api/rate-limits` (GET)
-- Resiliens: `src/app/api/resilience` (GET/PATCH) — udbyderprofiler, afbryder, hastighedsgrænsetilstand
-- Resilience reset: `src/app/api/resilience/reset` (POST) — nulstil breakers + cooldowns
-- Cachestatistik: `src/app/api/cache/stats` (GET/DELETE)
-- Modeltilgængelighed: `src/app/api/models/availability` (GET/POST)
-- Telemetri: `src/app/api/telemetry/summary` (GET)
+- CLI tooling helpers: `src/app/api/cli-tools/*`
+- IP filter: `src/app/api/settings/ip-filter` (GET/PUT)
+- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT)
+- System prompt: `src/app/api/settings/system-prompt` (GET/PUT)
+- Sessions: `src/app/api/sessions` (GET)
+- Rate limits: `src/app/api/rate-limits` (GET)
+- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state
+- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns
+- Cache stats: `src/app/api/cache/stats` (GET/DELETE)
+- Model availability: `src/app/api/models/availability` (GET/POST)
+- Telemetry: `src/app/api/telemetry/summary` (GET)
- Budget: `src/app/api/usage/budget` (GET/POST)
-- Fallback-kæder: `src/app/api/fallback/chains` (GET/POST/DELETE)
-- Overholdelsesrevision: `src/app/api/compliance/audit-log` (GET)
+- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE)
+- Compliance audit: `src/app/api/compliance/audit-log` (GET)
- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET)
-- Politikker: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core
+- Policies: `src/app/api/policies` (GET/POST)
-Hovedflowmoduler:
+## 2) SSE + Translation Core
-- Indtastning: `src/sse/handlers/chat.ts`
-- Kerneorkestrering: `open-sse/handlers/chatCore.ts`
-- Udbyder eksekveringsadaptere: `open-sse/executors/*`
-- Formatdetektion/udbyderkonfiguration: `open-sse/services/provider.ts`
+Main flow modules:
+
+- Entry: `src/sse/handlers/chat.ts`
+- Core orchestration: `open-sse/handlers/chatCore.ts`
+- Provider execution adapters: `open-sse/executors/*`
+- Format detection/provider config: `open-sse/services/provider.ts`
- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts`
-- Konto fallback logik: `open-sse/services/accountFallback.ts`
-- Oversættelsesregister: `open-sse/translator/index.ts`
-- Stream transformationer: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
-- Brugsudtrækning/normalisering: `open-sse/utils/usageTracking.ts`
-- Tænk tag-parser: `open-sse/utils/thinkTagParser.ts`
-- Indlejringshåndtering: `open-sse/handlers/embeddings.ts`
-- Indlejringsudbyderregistrering: `open-sse/config/embeddingRegistry.ts`
-- Billedgenereringshåndtering: `open-sse/handlers/imageGeneration.ts`
-- Billedudbyderregistrering: `open-sse/config/imageRegistry.ts`
-- Reaktionssanering: `open-sse/handlers/responseSanitizer.ts`
-- Rollenormalisering: `open-sse/services/roleNormalizer.ts`
+- Account fallback logic: `open-sse/services/accountFallback.ts`
+- Translation registry: `open-sse/translator/index.ts`
+- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
+- Usage extraction/normalization: `open-sse/utils/usageTracking.ts`
+- Think tag parser: `open-sse/utils/thinkTagParser.ts`
+- Embedding handler: `open-sse/handlers/embeddings.ts`
+- Embedding provider registry: `open-sse/config/embeddingRegistry.ts`
+- Image generation handler: `open-sse/handlers/imageGeneration.ts`
+- Image provider registry: `open-sse/config/imageRegistry.ts`
+- Response sanitization: `open-sse/handlers/responseSanitizer.ts`
+- Role normalization: `open-sse/services/roleNormalizer.ts`
-Tjenester (forretningslogik):
+Services (business logic):
-- Kontovalg/scoring: `open-sse/services/accountSelector.ts`
+- Account selection/scoring: `open-sse/services/accountSelector.ts`
- Context lifecycle management: `open-sse/services/contextManager.ts`
-- Håndhævelse af IP-filter: `open-sse/services/ipFilter.ts`
-- Sessionssporing: `open-sse/services/sessionManager.ts`
-- Anmod om deduplikering: `open-sse/services/signatureCache.ts`
-- Injektion af systemprompt: `open-sse/services/systemPrompt.ts`
-- Tænkende budgetstyring: `open-sse/services/thinkingBudget.ts`
+- IP filter enforcement: `open-sse/services/ipFilter.ts`
+- Session tracking: `open-sse/services/sessionManager.ts`
+- Request deduplication: `open-sse/services/signatureCache.ts`
+- System prompt injection: `open-sse/services/systemPrompt.ts`
+- Thinking budget management: `open-sse/services/thinkingBudget.ts`
- Wildcard model routing: `open-sse/services/wildcardRouter.ts`
-- Satsgrænsestyring: `open-sse/services/rateLimitManager.ts`
+- Rate limit management: `open-sse/services/rateLimitManager.ts`
- Circuit breaker: `open-sse/services/circuitBreaker.ts`
+- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy
+- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions
-Domænelagsmoduler:
+Domain layer modules:
-- Modeltilgængelighed: `src/lib/domain/modelAvailability.ts`
-- Omkostningsregler/budgetter: `src/lib/domain/costRules.ts`
-- Fallback-politik: `src/lib/domain/fallbackPolicy.ts`
+- Model availability: `src/lib/domain/modelAvailability.ts`
+- Cost rules/budgets: `src/lib/domain/costRules.ts`
+- Fallback policy: `src/lib/domain/fallbackPolicy.ts`
- Combo resolver: `src/lib/domain/comboResolver.ts`
-- Lockout-politik: `src/lib/domain/lockoutPolicy.ts`
-- Politikmotor: `src/domain/policyEngine.ts` — centraliseret lockout → budget → fallback-evaluering
-- Fejlkodekatalog: `src/lib/domain/errorCodes.ts`
-- Anmodnings-id: `src/lib/domain/requestId.ts`
-- Hente timeout: `src/lib/domain/fetchTimeout.ts`
-- Anmod om telemetri: `src/lib/domain/requestTelemetry.ts`
-- Overholdelse/revision: `src/lib/domain/compliance/index.ts`
+- Lockout policy: `src/lib/domain/lockoutPolicy.ts`
+- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation
+- Error codes catalog: `src/lib/domain/errorCodes.ts`
+- Request ID: `src/lib/domain/requestId.ts`
+- Fetch timeout: `src/lib/domain/fetchTimeout.ts`
+- Request telemetry: `src/lib/domain/requestTelemetry.ts`
+- Compliance/audit: `src/lib/domain/compliance/index.ts`
- Eval runner: `src/lib/domain/evalRunner.ts`
-- Vedvarende domænetilstand: `src/lib/db/domainState.ts` — SQLite CRUD til reservekæder, budgetter, omkostningshistorik, lockouttilstand, strømafbrydere
+- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers
-OAuth-udbydermoduler (12 individuelle filer under `src/lib/oauth/providers/`):
+OAuth provider modules (12 individual files under `src/lib/oauth/providers/`):
-- Registerindeks: `src/lib/oauth/providers/index.ts`
-- Individuelle udbydere: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts.line`, `t.`s.`s.`s.`
-- Tyndt omslag: `src/lib/oauth/providers.ts` — reeksporterer fra individuelle moduler## 3) Persistence Layer
+- Registry index: `src/lib/oauth/providers/index.ts`
+- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
+- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules
-Primær tilstand DB (SQLite):
+## 3) Persistence Layer
-- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrationer, WAL)
-- Re-eksport facade: `src/lib/localDb.ts` (tyndt kompatibilitetslag for opkaldere)
-- fil: `${DATA_DIR}/storage.sqlite` (eller `$XDG_CONFIG_HOME/omniroute/storage.sqlite` når indstillet, ellers `~/.omniroute/storage.sqlite`)
-- enheder (tabeller + KV-navnerum): providerConnections, providerNodes, modelAliaser, combos, apiKeys, indstillinger, prissætning,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt**
+Primary state DB (SQLite):
-Brugsvedholdenhed:
+- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL)
+- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers)
+- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`)
+- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt**
-- facade: `src/lib/usageDb.ts` (dekomponerede moduler i `src/lib/usage/*`)
-- SQLite-tabeller i `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs`
-- valgfrie filartefakter forbliver for kompatibilitet/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `
/logs/...`)
-- Ældre JSON-filer migreres til SQLite ved opstartsmigreringer, når de er til stede
+Usage persistence:
+
+- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`)
+- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs`
+- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`)
+- legacy JSON files are migrated to SQLite by startup migrations when present
Domain State DB (SQLite):
-- `src/lib/db/domainState.ts` — CRUD-operationer for domænetilstand
-- Tabeller (oprettet i `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
-- Gennemskrivningscachemønster: Kort i hukommelsen er autoritative under kørsel; mutationer skrives synkront til SQLite; tilstand gendannes fra DB ved koldstart## 4) Auth + Security Surfaces
+- `src/lib/db/domainState.ts` — CRUD operations for domain state
+- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
+- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start
+
+## 4) Auth + Security Surfaces
- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts`
-- Generering/bekræftelse af API-nøgler: `src/shared/utils/apiKey.ts`
-- Udbyderhemmeligheder vedblev i 'providerConnections'-poster
-- Udgående proxy-understøttelse via `open-sse/utils/proxyFetch.ts` (env vars) og `open-sse/utils/networkProxy.ts` (konfigurerbar pr. udbyder eller global)## 5) Cloud Sync
+- API key generation/verification: `src/shared/utils/apiKey.ts`
+- Provider secrets persisted in `providerConnections` entries
+- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global)
+
+## 5) Cloud Sync
- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
-- Periodisk opgave: `src/shared/services/cloudSyncScheduler.ts`
-- Periodisk opgave: `src/shared/services/modelSyncScheduler.ts`
-- Styr rute: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`)
+- Periodic task: `src/shared/services/cloudSyncScheduler.ts`
+- Periodic task: `src/shared/services/modelSyncScheduler.ts`
+- Control route: `src/app/api/sync/cloud/route.ts`
+
+## Request Lifecycle (`/v1/chat/completions`)
```mermaid
sequenceDiagram
@@ -338,7 +363,9 @@ flowchart TD
Q -- No --> R[Return all unavailable]
```
-Fallback-beslutninger er drevet af `open-sse/services/accountFallback.ts` ved hjælp af statuskoder og fejlmeddelelsesheuristik. Combo-routing tilføjer en ekstra beskyttelse: udbyder-omfattede 400'er, såsom upstream-indholdsblokering og rollevalideringsfejl, behandles som model-lokale fejl, så senere combo-mål kan stadig køre.## OAuth Onboarding and Token Refresh Lifecycle
+Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run.
+
+## OAuth Onboarding and Token Refresh Lifecycle
```mermaid
sequenceDiagram
@@ -368,7 +395,9 @@ sequenceDiagram
Test-->>UI: validation result
```
-Opdatering under live trafik udføres inde i `open-sse/handlers/chatCore.ts` via eksekveren `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable)
+Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`.
+
+## Cloud Sync Lifecycle (Enable / Sync / Disable)
```mermaid
sequenceDiagram
@@ -400,7 +429,9 @@ sequenceDiagram
Sync-->>UI: disabled
```
-Periodisk synkronisering udløses af "CloudSyncScheduler", når skyen er aktiveret.## Data Model and Storage Map
+Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled.
+
+## Data Model and Storage Map
```mermaid
erDiagram
@@ -501,12 +532,14 @@ erDiagram
}
```
-Fysiske lagerfiler:
+Physical storage files:
-- primær runtime DB: `${DATA_DIR}/storage.sqlite`
-- anmode om log linjer: `${DATA_DIR}/log.txt` (compat/debug artefakt)
-- strukturerede opkaldsdataarkiver: `${DATA_DIR}/call_logs/`
-- valgfri oversætter/anmodningsfejlfindingssessioner: `/logs/...`## Deployment Topology
+- primary runtime DB: `${DATA_DIR}/storage.sqlite`
+- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact)
+- structured call payload archives: `${DATA_DIR}/call_logs/`
+- optional translator/request debug sessions: `/logs/...`
+
+## Deployment Topology
```mermaid
flowchart LR
@@ -541,205 +574,249 @@ flowchart LR
### Route and API Modules
-- `src/app/api/v1/*`, `src/app/api/v1beta/*`: kompatibilitets-API'er
-- `src/app/api/v1/providers/[provider]/*`: dedikerede ruter pr. udbyder (chat, indlejringer, billeder)
-- `src/app/api/providers*`: udbyder CRUD, validering, test
-- `src/app/api/provider-nodes*`: tilpasset kompatibel nodestyring
-- `src/app/api/provider-models`: Custom model management (CRUD)
-- `src/app/api/models/route.ts`: modelkatalog API (aliaser + tilpassede modeller)
-- `src/app/api/oauth/*`: OAuth/enhedskode-flow
-- `src/app/api/keys*`: lokal API-nøglelivscyklus
-- `src/app/api/models/alias`: aliashåndtering
+- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs
+- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images)
+- `src/app/api/providers*`: provider CRUD, validation, testing
+- `src/app/api/provider-nodes*`: custom compatible node management
+- `src/app/api/provider-models`: custom model management (CRUD)
+- `src/app/api/models/route.ts`: model catalog API (aliases + custom models)
+- `src/app/api/oauth/*`: OAuth/device-code flows
+- `src/app/api/keys*`: local API key lifecycle
+- `src/app/api/models/alias`: alias management
- `src/app/api/combos*`: fallback combo management
-- `src/app/api/pricing`: pristilsidesættelser til omkostningsberegning
-- `src/app/api/settings/proxy`: proxy-konfiguration (GET/PUT/DELETE)
-- `src/app/api/settings/proxy/test`: test af udgående proxyforbindelse (POST)
-- `src/app/api/usage/*`: brugs- og log-API'er
-- `src/app/api/sync/*` + `src/app/api/cloud/*`: skysynkronisering og skyvendte hjælpere
-- `src/app/api/cli-tools/*`: lokale CLI-konfigurationsskrivere/checkers
-- `src/app/api/settings/ip-filter`: IP-tilladelsesliste/blokeringsliste (GET/PUT)
-- `src/app/api/settings/thinking-budget`: Tænketoken-budgetkonfiguration (GET/PUT)
-- `src/app/api/settings/system-prompt`: global systemprompt (GET/PUT)
-- `src/app/api/sessions`: aktiv sessionsfortegnelse (GET)
-- `src/app/api/rate-limits`: rategrænsestatus pr. konto (GET)### Routing and Execution Core
+- `src/app/api/pricing`: pricing overrides for cost calculation
+- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE)
+- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST)
+- `src/app/api/usage/*`: usage and logs APIs
+- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers
+- `src/app/api/cli-tools/*`: local CLI config writers/checkers
+- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT)
+- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT)
+- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT)
+- `src/app/api/sessions`: active session listing (GET)
+- `src/app/api/rate-limits`: per-account rate limit status (GET)
-- `src/sse/handlers/chat.ts`: anmodning om parse, kombinationshåndtering, kontovalgsløkke
-- `open-sse/handlers/chatCore.ts`: oversættelse, eksekutorafsendelse, genforsøg/opdateringshåndtering, stream-opsætning
-- `open-sse/executors/*`: udbyderspecifik netværks- og formatadfærd### Translation Registry and Format Converters
+### Routing and Execution Core
-- `open-sse/translator/index.ts`: oversætterregister og orkestrering
-- Anmod om oversættere: `open-sse/translator/request/*`
-- Svaroversættere: `open-sse/translator/response/*`
-- Formatkonstanter: `open-sse/translator/formats.ts`### Persistence
+- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop
+- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup
+- `open-sse/executors/*`: provider-specific network and format behavior
-- `src/lib/db/*`: persistent config/state og domæne persistens på SQLite
-- `src/lib/localDb.ts`: re-eksport af kompatibilitet til DB-moduler
-- `src/lib/usageDb.ts`: brugshistorik/opkaldslogs facade oven på SQLite-tabeller## Provider Executor Coverage (Strategy Pattern)
+### Translation Registry and Format Converters
-Hver udbyder har en specialiseret executor, der udvider `BaseExecutor` (i `open-sse/executors/base.ts`), som giver URL-opbygning, header-konstruktion, genforsøg med eksponentiel backoff, credential refresh hooks og `execute()`-orkestreringsmetoden.
+- `open-sse/translator/index.ts`: translator registry and orchestration
+- Request translators: `open-sse/translator/request/*`
+- Response translators: `open-sse/translator/response/*`
+- Format constants: `open-sse/translator/formats.ts`
-| Eksekutør | Udbyder(e) | Særlig håndtering |
-| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------- |
-| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fyrværkeri, Cerebras, Cohere, NVIDIA | Dynamisk URL/header-konfiguration pr. udbyder |
-| `AntigravityExecutor` | Google Antigravity | Brugerdefinerede projekt-/sessions-id'er, forsøg igen - efter parsing |
-| `CodexExecutor` | OpenAI Codex | Injicerer systeminstruktioner, fremtvinger ræsonnement indsats |
-| `CursorExecutor` | Markør IDE | ConnectRPC-protokol, Protobuf-kodning, anmodningssignering via checksum |
-| `GithubExecutor` | GitHub Copilot | Copilot token opdatering, VSCode-mimicing headers |
-| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binært format → SSE-konvertering |
-| `GeminiCLIEexecutor` | Gemini CLI | Opdateringscyklus for Google OAuth-token |
+### Persistence
-Alle andre udbydere (inklusive brugerdefinerede kompatible noder) bruger `DefaultExecutor`.## Provider Compatibility Matrix
+- `src/lib/db/*`: persistent config/state and domain persistence on SQLite
+- `src/lib/localDb.ts`: compatibility re-export for DB modules
+- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables
-| Udbyder | Format | Auth | Stream | Ikke-stream | Token Opdater | Brug API |
-| ---------------- | --------------- | --------------------- | ---------------- | ----------- | ------------- | -------------------- | ------------------------------ |
-| Claude | claude | API-nøgle / OAuth | ✅ | ✅ | ✅ | ⚠️ Kun administrator |
-| Tvillingerne | gemini | API-nøgle / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
-| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
-| Antigravitation | antityngdekraft | OAuth | ✅ | ✅ | ✅ | ✅ Fuld kvote API |
-| OpenAI | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ |
-| Codex | openai-svar | OAuth | ✅ tvunget | ❌ | ✅ | ✅ Satsgrænser |
-| GitHub Copilot | åbne | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Kvote snapshots |
-| Markør | markør | Tilpasset kontrolsum | ✅ | ✅ | ❌ | ❌ |
-| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Brugsgrænser |
-| Qwen | åbne | OAuth | ✅ | ✅ | ✅ | ⚠️ Efter anmodning |
-| Qoder | åbne | OAuth (Grundlæggende) | ✅ | ✅ | ✅ | ⚠️ Efter anmodning |
-| OpenRouter | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ |
-| GLM/Kimi/MiniMax | claude | API-nøgle | ✅ | ✅ | ❌ | ❌ |
-| DeepSeek | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ |
-| Groq | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ |
-| xAI (Grok) | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ |
-| Mistral | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ |
-| Forvirring | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ |
-| Sammen AI | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ |
-| Fyrværkeri AI | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ |
-| Cerebras | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ |
-| Sammenhæng | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ |
-| NVIDIA NIM | åbne | API-nøgle | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage |
+## Provider Executor Coverage (Strategy Pattern)
-Detekterede kildeformater omfatter:
+Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method.
-- 'openai'
-- `openai-svar`
-- `Claude`
-- 'tvilling'
+| Executor | Provider(s) | Special Handling |
+| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- |
+| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider |
+| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing |
+| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort |
+| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum |
+| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers |
+| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion |
+| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle |
-Målformater omfatter:
+All other providers (including custom compatible nodes) use the `DefaultExecutor`.
-- OpenAI chat/svar
+## Provider Compatibility Matrix
+
+| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API |
+| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ |
+| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only |
+| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API |
+| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits |
+| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots |
+| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ |
+| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits |
+| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request |
+| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request |
+| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ |
+| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+
+## Format Translation Coverage
+
+Detected source formats include:
+
+- `openai`
+- `openai-responses`
+- `claude`
+- `gemini`
+
+Target formats include:
+
+- OpenAI chat/Responses
- Claude
-- Gemini/Gemini-CLI/Antigravity kuvert
+- Gemini/Gemini-CLI/Antigravity envelope
- Kiro
-- Markør
+- Cursor
-Oversættelser bruger**OpenAI som hub-format**- alle konverteringer går gennem OpenAI som mellemliggende:```
+Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate:
+
+```
Source Format → OpenAI (hub) → Target Format
+```
-````
+Translations are selected dynamically based on source payload shape and provider target format.
-Oversættelser vælges dynamisk baseret på kildens nyttelastform og udbyderens målformat.
+Additional processing layers in the translation pipeline:
-Yderligere behandlingslag i oversættelsespipelinen:
+- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance
+- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE)
+- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field
+- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema`
--**Responssanering**— Fjerner ikke-standardfelter fra OpenAI-formatsvar (både streaming og ikke-streaming) for at sikre streng SDK-overholdelse
--**Rollenormalisering**— Konverterer `udvikler` → `system` til ikke-OpenAI-mål; fletter `system` → `bruger` for modeller, der afviser systemrollen (GLM, ERNIE)
--**Tænk tag-udtrækning**— Parser "..."-blokke fra indhold til feltet "reasoning_content"
--**Structured output**— Konverterer OpenAI `response_format.json_schema` til Geminis `responseMimeType` + `responseSchema`## Supported API Endpoints
+## Supported API Endpoints
-| Slutpunkt | Format | Behandler |
-| -------------------------------------------------- | ------------------ | -------------------------------------------------------------------------- |
-| `POST /v1/chat/afslutninger` | OpenAI Chat | `src/sse/handlers/chat.ts` |
-| `POST /v1/meddelelser` | Claude Beskeder | Samme handler (auto-detekteret) |
-| `POST /v1/svar` | OpenAI-svar | `open-sse/handlers/responsesHandler.ts` |
-| `POST /v1/indlejringer` | OpenAI-indlejringer | `open-sse/handlers/embeddings.ts` |
-| `GET /v1/indlejringer` | Modelliste | API-rute |
-| `POST /v1/billeder/generationer` | OpenAI Billeder | `open-sse/handlers/imageGeneration.ts` |
-| `GET /v1/billeder/generationer` | Modelliste | API-rute |
-| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedikeret per udbyder med modelvalidering |
-| `POST /v1/providers/{provider}/embeddings` | OpenAI-indlejringer | Dedikeret per udbyder med modelvalidering |
-| `POST /v1/providers/{provider}/images/generations` | OpenAI Billeder | Dedikeret per udbyder med modelvalidering |
-| `POST /v1/messages/count_tokens` | Claude Token Count | API-rute |
-| `GET /v1/modeller` | OpenAI Models liste | API-rute (chat + indlejring + billede + brugerdefinerede modeller) |
-| `GET /api/models/catalog` | Katalog | Alle modeller grupperet efter udbyder + type |
-| `POST /v1beta/models/*:streamGenerateContent` | Tvilling hjemmehørende | API-rute |
-| `GET/PUT/DELETE /api/indstillinger/proxy` | Proxy-konfiguration | Netværk proxy-konfiguration |
-| `POST /api/settings/proxy/test` | Proxy-forbindelse | Proxy-sundheds-/forbindelsestestslutpunkt |
-| `GET/POST/DELETE /api/provider-models` | Udbyder modeller | Udbydermodelmetadata understøtter tilpassede og administrerede tilgængelige modeller |## Bypass Handler
+| Endpoint | Format | Handler |
+| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- |
+| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` |
+| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) |
+| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` |
+| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` |
+| `GET /v1/embeddings` | Model listing | API route |
+| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` |
+| `GET /v1/images/generations` | Model listing | API route |
+| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation |
+| `POST /v1/messages/count_tokens` | Claude Token Count | API route |
+| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) |
+| `GET /api/models/catalog` | Catalog | All models grouped by provider + type |
+| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route |
+| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration |
+| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint |
+| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models |
-Bypass-handleren (`open-sse/utils/bypassHandler.ts`) opsnapper kendte "throwaway"-anmodninger fra Claude CLI - opvarmningsping, titeludtræk og tokentællinger - og returnerer et**falsk svar**uden at forbruge upstream-udbydertokens. Dette udløses kun, når `User-Agent` indeholder `claude-cli`.## Request Logger Pipeline
+## Bypass Handler
-Anmodningsloggeren (`open-sse/utils/requestLogger.ts`) giver en 7-trins debug-logningspipeline, deaktiveret som standard, aktiveret via `ENABLE_REQUEST_LOGS=true`:```
+The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`.
+
+## Request Logger Pipeline
+
+The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`:
+
+```
1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json
→ 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt
-````
+```
-Filer skrives til `/logs//` for hver anmodningssession.## Failure Modes and Resilience
+Files are written to `/logs//` for each request session.
+
+## Failure Modes and Resilience
## 1) Account/Provider Availability
-- Nedkøling af udbyderkonto på forbigående/rate/godkendelsesfejl
-- konto fallback før mislykket anmodning
-- combo model fallback, når den nuværende model/udbydersti er udtømt## 2) Token Expiry
+- provider account cooldown on transient/rate/auth errors
+- account fallback before failing request
+- combo model fallback when current model/provider path is exhausted
-- Forhåndstjek og opdater med genforsøg for udbydere, der kan opdateres
-- 401/403 forsøg igen efter opdateringsforsøg i kernestien## 3) Stream Safety
+## 2) Token Expiry
-- afbrydelsesbevidst streamcontroller
-- oversættelsesstrøm med slut-af-stream-skyl og `[DONE]`-håndtering
-- forbrugsestimeret fallback, når udbyderens brugsmetadata mangler## 4) Cloud Sync Degradation
+- pre-check and refresh with retry for refreshable providers
+- 401/403 retry after refresh attempt in core path
-- Synkroniseringsfejl dukker op, men lokal kørsel fortsætter
-- Scheduler har logik, der kan genforsøge, men periodisk udførelse kalder i øjeblikket enkelt-forsøgssynkronisering som standard## 5) Data Integrity
+## 3) Stream Safety
-- SQLite-skemamigreringer og auto-opgraderingshooks ved opstart
-- ældre JSON → SQLite-migreringskompatibilitetssti## Observability and Operational Signals
+- disconnect-aware stream controller
+- translation stream with end-of-stream flush and `[DONE]` handling
+- usage estimation fallback when provider usage metadata is missing
-Kilder til synlighed ved kørsel:
+## 4) Cloud Sync Degradation
-- konsollogfiler fra `src/sse/utils/logger.ts`
-- brugsaggregater pr. anmodning i SQLite (`usage_history`, `call_logs`, `proxy_logs`)
-- fire-trins detaljeret nyttelastfangst i SQLite (`request_detail_logs`), når `settings.detailed_logs_enabled=true`
-- statuslog for tekstanmodning i `log.txt` (valgfrit/kompatibelt)
-- valgfri dybe anmodnings-/oversættelseslogfiler under `logs/` når `ENABLE_REQUEST_LOGS=true`
-- dashboardbrugsslutpunkter (`/api/usage/*`) for brugergrænsefladeforbrug
+- sync errors are surfaced but local runtime continues
+- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default
-Detaljeret anmodning om nyttelastfangst gemmer op til fire JSON-nyttelasttrin pr. dirigeret opkald:
+## 5) Data Integrity
-- rå anmodning modtaget fra klienten
-- oversat anmodning faktisk sendt opstrøms
-- Providersvar rekonstrueret som JSON; streamede svar komprimeres til den endelige oversigt plus stream metadata
-- endelig kundesvar returneret af OmniRoute; streamede svar gemmes i den samme kompakte oversigtsform## Security-Sensitive Boundaries
+- SQLite schema migrations and auto-upgrade hooks at startup
+- legacy JSON → SQLite migration compatibility path
-- JWT-hemmelighed (`JWT_SECRET`) sikrer bekræftelse/signering af dashboard-sessionscookie
-- Oprindelig adgangskode-bootstrap ('INITIAL_PASSWORD') skal eksplicit konfigureres til førstegangs-klargøring
-- API-nøgle HMAC-hemmelighed (`API_KEY_SECRET`) sikrer genereret lokalt API-nøgleformat
-- Udbyderhemmeligheder (API-nøgler/tokens) bevares i lokal DB og bør beskyttes på filsystemniveau
-- Slutpunkter for skysynkronisering er afhængige af API-nøglegodkendelse + maskin-id-semantik## Environment and Runtime Matrix
+## Observability and Operational Signals
-Miljøvariabler aktivt brugt af kode:
+Runtime visibility sources:
-- App/godkendelse: `JWT_SECRET`, `INITIAL_PASSWORD`
-- Lager: `DATA_DIR`
-- Kompatibel nodeadfærd: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
-- Valgfri lagerbasetilsidesættelse (Linux/macOS, når `DATA_DIR` er deaktiveret): `XDG_CONFIG_HOME`
-- Sikkerhedshashing: `API_KEY_SECRET`, `MACHINE_ID_SALT`
-- Logning: `ENABLE_REQUEST_LOGS`
-- Synkronisering/sky-URL: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
-- Udgående proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` og varianter med små bogstaver
-- SOCKS5-funktionsflag: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
-- Platform/runtime-hjælpere (ikke app-specifik konfiguration): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes
+- console logs from `src/sse/utils/logger.ts`
+- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`)
+- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true`
+- textual request status log in `log.txt` (optional/compat)
+- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true`
+- dashboard usage endpoints (`/api/usage/*`) for UI consumption
-1. `usageDb` og `localDb` deler den samme basismappepolitik (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) med ældre filmigrering.
-2. `/api/v1/route.ts` uddelegerer til den samme forenede katalogbygger, der bruges af `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) for at undgå semantisk drift.
-3. Anmodningslogger skriver hele headers/body, når den er aktiveret; behandle logbiblioteket som følsomt.
-4. Cloudadfærd afhænger af korrekt `NEXT_PUBLIC_BASE_URL` og cloud-endepunkts tilgængelighed.
-5. `open-sse/` biblioteket udgives som `@omniroute/open-sse`**npm workspace-pakken**. Kildekoden importerer den via `@omniroute/open-sse/...` (løst af Next.js `transpilePackages`). Filstier i dette dokument bruger stadig mappenavnet `open-sse/` for at opnå konsistens.
-6. Diagrammer i dashboardet bruger**Recharts**(SVG-baseret) til tilgængelige, interaktive analysevisualiseringer (søjlediagrammer for modelbrug, udbyderopdelingstabeller med succesrater).
-7. E2E-tests bruger**Playwright**(`tests/e2e/`), køres via `npm run test:e2e`. Enhedstests bruger**Node.js test runner**(`tests/unit/`), køres via `npm run test:unit`. Kildekoden under `src/` er**TypeScript**(`.ts`/`.tsx`); `open-sse/`-arbejdsområdet forbliver JavaScript (`.js`).
-8. Siden Indstillinger er organiseret i 5 faner: Sikkerhed, Routing (6 globale strategier: fill-first, round-robin, p2c, random, mindst brugt, omkostningsoptimeret), Resiliens (redigerbare hastighedsgrænser, strømafbryder, politikker), AI (tænkebudget, systemprompt, promptcache), Avanceret (proxy).## Operational Verification Checklist
+Detailed request payload capture stores up to four JSON payload stages per routed call:
-- Byg fra kilde: `npm run build`
-- Byg Docker-billede: `docker build -t omniroute .`
-- Start service og bekræft:
-- `GET /api/indstillinger`
-- `GET /api/v1/modeller`
-- CLI-målbasis-URL skal være "http://:20128/v1", når "PORT=20128"
+- raw request received from the client
+- translated request actually sent upstream
+- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata
+- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form
+
+## Security-Sensitive Boundaries
+
+- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing
+- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning
+- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format
+- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level
+- Cloud sync endpoints rely on API key auth + machine id semantics
+
+## Environment and Runtime Matrix
+
+Environment variables actively used by code:
+
+- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD`
+- Storage: `DATA_DIR`
+- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
+- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME`
+- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT`
+- Logging: `ENABLE_REQUEST_LOGS`
+- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
+- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants
+- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
+- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`
+
+## Known Architectural Notes
+
+1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration.
+2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift.
+3. Request logger writes full headers/body when enabled; treat log directory as sensitive.
+4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability.
+5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency.
+6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates).
+7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`).
+8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy).
+9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed.
+10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22.
+11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible.
+
+## Operational Verification Checklist
+
+- Build from source: `npm run build`
+- Build Docker image: `docker build -t omniroute .`
+- Start service and verify:
+- `GET /api/settings`
+- `GET /api/v1/models`
+- CLI target base URL should be `http://:20128/v1` when `PORT=20128`
diff --git a/docs/i18n/da/docs/FEATURES.md b/docs/i18n/da/docs/FEATURES.md
index 19884147f8..7c798a15dd 100644
--- a/docs/i18n/da/docs/FEATURES.md
+++ b/docs/i18n/da/docs/FEATURES.md
@@ -4,102 +4,168 @@
---
-Visuel guide til hver sektion af OmniRoute-dashboardet.---
+
+
+Visual guide to every section of the OmniRoute dashboard.
+
+---
## 🔌 Providers
-Administrer AI-udbyderforbindelser: OAuth-udbydere (Claude Code, Codex, Gemini CLI), API-nøgleudbydere (Groq, DeepSeek, OpenRouter) og gratis udbydere (Qoder, Qwen, Kiro). Kiro-konti inkluderer sporing af kreditsaldo - resterende kreditter, samlet godtgørelse og fornyelsesdato synlig i Dashboard → Brug.
+Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage.
+
+
---
## 🎨 Combos
-Opret modelrouting-kombinationer med 6 strategier: prioritet, vægtet, round-robin, tilfældig, mindst brugt og omkostningsoptimeret. Hver combo kæder flere modeller med automatisk fallback og inkluderer hurtige skabeloner og klarhedstjek.
+Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks.
+
+
---
## 📊 Analytics
-Omfattende brugsanalyse med token-forbrug, omkostningsestimater, aktivitetsvarmekort, ugentlige distributionsdiagrammer og opdelinger pr. udbyder.
+Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns.
+
+
---
## 🏥 System Health
-Overvågning i realtid: oppetid, hukommelse, version, latency percentiler (p50/p95/p99), cache-statistik og udbyderens afbrydertilstande.
+Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states.
+
+
---
## 🔧 Translator Playground
-Fire tilstande til fejlfinding af API-oversættelser:**Playground**(formatkonverter),**Chat Tester**(live-anmodninger),**Test Bench**(batchtest) og**Live Monitor**(streaming i realtid).
+Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream).
+
+
---
## 🎮 Model Playground _(v2.0.9+)_
-Test enhver model direkte fra instrumentbrættet. Vælg udbyder, model og slutpunkt, skriv prompts med Monaco Editor, stream svar i realtid, afbryd midt-stream, og se timing-metrics.---
+Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics.
+
+---
## 🎨 Themes _(v2.0.5+)_
-Brugerdefinerbare farvetemaer til hele dashboardet. Vælg mellem 7 forudindstillede farver (koral, blå, rød, grøn, violet, orange, cyan) eller opret et brugerdefineret tema ved at vælge en hex-farve. Understøtter lys, mørk og systemtilstand.---
+Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode.
+
+---
## ⚙️ Settings
-Omfattende indstillingspanel med faner:
+Comprehensive settings panel with tabs:
--**Generelt**— Systemlagring, backupstyring (eksport/importdatabase) -**Udseende**— Temavælger (mørke/lys/system), forudindstillinger af farvetema og brugerdefinerede farver, synlighed i sundhedslog, synlighedskontrol for sidebjælkeelementer -**Sikkerhed**— API-endepunktsbeskyttelse, tilpasset udbyderblokering, IP-filtrering, sessionsoplysninger -**Routing**— Modelaliaser, forringelse af baggrundsopgaver -**Resiliens**— Frekvensgrænsevedholdenhed, tuning af strømafbryder, automatisk deaktivering af forbudte konti, overvågning af udbyderens udløb -**Avanceret**— Konfigurationstilsidesættelser, konfigurationsrevisionsspor, fallback-forringelsestilstand
+- **General** — System storage, backup management (export/import database)
+- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls
+- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info
+- **Routing** — Model aliases, background task degradation
+- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration
+- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode
+
+
---
## 🔧 CLI Tools
-Et-klik-konfiguration til AI-kodningsværktøjer: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor og Factory Droid. Indeholder automatiseret konfigurationsanvendelse/nulstilling, forbindelsesprofiler og modelkortlægning.
+One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping.
+
+
---
## 🤖 CLI Agents _(v2.0.11+)_
-Dashboard til at opdage og administrere CLI-agenter. Viser et gitter med 14 indbyggede agenter (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) med:
+Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with:
--**Installationsstatus**— Installeret / Ikke fundet med versionsregistrering -**Protokolmærker**— stdio, HTTP osv. -**Tilpassede agenter**- Registrer ethvert CLI-værktøj via formular (navn, binær, versionskommando, spawn args) -**CLI Fingerprint Matching**— Skift pr. udbyder for at matche native CLI-anmodningssignaturer, hvilket reducerer risikoen for forbud, mens proxy-IP bevares---
+- **Installation status** — Installed / Not Found with version detection
+- **Protocol badges** — stdio, HTTP, etc.
+- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args)
+- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP
+
+---
+
+## 🔗 Context Relay _(v3.5.5+)_
+
+A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context.
+
+Configurable via combo-level or global settings:
+- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%)
+- **Max Messages For Summary** — How much recent history to condense
+- **Summary Model** — Optional override model for generating the handoff summary
+
+Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md).
+
+---
+
+## 🛡️ Proxy Hardening _(v3.5.5+)_
+
+Comprehensive proxy configuration enforcement across the entire request pipeline:
+
+- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments
+- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings
+- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22
+- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS
+
+---
## 🖼️ Media _(v2.0.3+)_
-Generer billeder, videoer og musik fra dashboardet. Understøtter OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open og MusicGen.---
+Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen.
+
+---
## 📝 Request Logs
-Anmodningslogning i realtid med filtrering efter udbyder, model, konto og API-nøgle. Viser statuskoder, tokenbrug, latenstid og svardetaljer.
+Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details.
+
+
---
## 🌐 API Endpoint
-Dit forenede API-slutpunkt med kapacitetsopdeling: Chatfuldførelser, Responses API, indlejringer, billedgenerering, omrangering, lydtransskription, tekst-til-tale, modereringer og registrerede API-nøgler. Cloudflare Quick Tunnel integration og cloud proxy support til fjernadgang.
+Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access.
+
+
---
## 🔑 API Key Management
-Opret, omfang og tilbagekald API-nøgler. Hver nøgle kan begrænses til specifikke modeller/udbydere med fuld adgang eller skrivebeskyttet tilladelse. Visuel nøglestyring med brugssporing.---
+Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking.
+
+---
## 📋 Audit Log
-Administrativ handlingssporing med filtrering efter handlingstype, aktør, mål, IP-adresse og tidsstempel. Fuld historik for sikkerhedshændelser.---
+Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history.
+
+---
## 🖥️ Desktop Application
-Native Electron desktop-app til Windows, macOS og Linux. Kør OmniRoute som et selvstændigt program med systembakkeintegration, offline support, automatisk opdatering og installation med ét klik.
+Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install.
-Nøglefunktioner:
+Key features:
-- Afstemning af serverberedskab (ingen tom skærm ved koldstart)
-- Systembakke med portstyring
-- Indholdssikkerhedspolitik
-- Engangslås
-- Automatisk opdatering ved genstart
-- Platform-betinget UI (macOS trafiklys, Windows/Linux standard titellinje)
-- Hærdet Electron build-emballage — symlinkede 'node_modules' i den selvstændige bundt detekteres og afvises før pakning, hvilket forhindrer runtime-afhængighed af build-maskinen (v2.5.5+)
+- Server readiness polling (no blank screen on cold start)
+- System tray with port management
+- Content Security Policy
+- Single-instance lock
+- Auto-update on restart
+- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar)
+- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+)
-📖 Se [`electron/README.md`](../electron/README.md) for fuld dokumentation.
+📖 See [`electron/README.md`](../electron/README.md) for full documentation.
diff --git a/docs/i18n/da/docs/TROUBLESHOOTING.md b/docs/i18n/da/docs/TROUBLESHOOTING.md
index 67794ed8eb..063677c844 100644
--- a/docs/i18n/da/docs/TROUBLESHOOTING.md
+++ b/docs/i18n/da/docs/TROUBLESHOOTING.md
@@ -4,68 +4,142 @@
---
-Almindelige problemer og løsninger til OmniRoute.---
+
+
+Common problems and solutions for OmniRoute.
+
+---
## Quick Fixes
-| Problem | Løsning |
-| -------------------------------------- | --------------------------------------------------------------------------- | --- |
-| Første login virker ikke | Indstil `INITIAL_PASSWORD` i `.env` (ingen hardcoded standard) |
-| Dashboard åbner ved forkert port | Indstil `PORT=20128` og `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
-| Ingen anmodningslogfiler under `logs/` | Indstil `ENABLE_REQUEST_LOGS=true` |
-| EACCES: tilladelse nægtet | Indstil `DATA_DIR=/path/to/writable/dir` for at tilsidesætte `~/.omniroute` |
-| Routingstrategi gemmer ikke | Opdatering til v1.4.11+ (Zod-skemafix for indstillinger persistens) | --- |
+| Problem | Solution |
+| ----------------------------- | ------------------------------------------------------------------ |
+| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) |
+| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
+| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` |
+| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` |
+| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) |
+| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below |
+| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below |
+
+---
+
+## Node.js Compatibility
+
+
+
+### Login page crashes or shows "Module self-registration" error
+
+**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database.
+
+**Symptoms:**
+- Login page shows a blank screen or a server error
+- Console shows `Error: Module did not self-register` or similar native binding errors
+- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected
+
+**Fix:**
+
+1. Install Node.js 22 LTS (recommended):
+ ```bash
+ nvm install 22
+ nvm use 22
+ ```
+2. Verify your version: `node --version` should show `v22.x.x`
+3. Reinstall OmniRoute: `npm install -g omniroute`
+4. Restart: `omniroute`
+
+> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**.
+
+---
+
+## Proxy Issues
+
+
+
+### Provider validation shows "fetch failed"
+
+**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing.
+
+**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically.
+
+### Token health check fails with "fetch failed"
+
+**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection.
+
+**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+.
+
+### SOCKS5 proxy returns "invalid onRequestStart method"
+
+**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation.
+
+**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+.
+
+---
## Provider Issues
### "Language model did not provide messages"
-**Årsag:**Udbyderkvoten er opbrugt.
+**Cause:** Provider quota exhausted.
-**Ret:**
+**Fix:**
-1. Tjek dashboard-kvotesporing
-2. Brug en kombination med reserveniveauer
-3. Skift til billigere/gratis niveau### Rate Limiting
+1. Check dashboard quota tracker
+2. Use a combo with fallback tiers
+3. Switch to cheaper/free tier
-**Årsag:**Abonnementskvoten er opbrugt.
+### Rate Limiting
-**Ret:**
+**Cause:** Subscription quota exhausted.
-- Tilføj reserve: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-- Brug GLM/MiniMax som billig backup### OAuth Token Expired
+**Fix:**
-OmniRoute opdaterer automatisk tokens. Hvis problemerne fortsætter:
+- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Use GLM/MiniMax as cheap backup
-1. Dashboard → Udbyder → Genopret forbindelse
-2. Slet og tilføj udbyderforbindelsen igen---
+### OAuth Token Expired
+
+OmniRoute auto-refreshes tokens. If issues persist:
+
+1. Dashboard → Provider → Reconnect
+2. Delete and re-add the provider connection
+
+---
## Cloud Issues
### Cloud Sync Errors
-1. Bekræft, at `BASE_URL` peger på din kørende forekomst (f.eks. `http://localhost:20128`)
-2. Bekræft "CLOUD_URL" peger på dit cloud-slutpunkt (f.eks. "https://omniroute.dev")
-3. Hold `NEXT_PUBLIC_*`-værdier på linje med værdier på serversiden### Cloud `stream=false` Returns 500
+1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`)
+2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`)
+3. Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**Symptom:**`Uventet token 'd'...` på cloud-slutpunktet for ikke-streaming-opkald.
+### Cloud `stream=false` Returns 500
-**Årsag:**Upstream returnerer SSE-nyttelast, mens klienten forventer JSON.
+**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls.
-**Løsning:**Brug 'stream=true' til direkte skyopkald. Lokal kørselstid inkluderer SSE→JSON fallback.### Cloud Says Connected but "Invalid API key"
+**Cause:** Upstream returns SSE payload while client expects JSON.
-1. Opret en ny nøgle fra det lokale dashboard (`/api/keys`)
-2. Kør skysynkronisering: Aktiver sky → Synkroniser nu
-3. Gamle/ikke-synkroniserede nøgler kan stadig returnere '401' på skyen---
+**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback.
+
+### Cloud Says Connected but "Invalid API key"
+
+1. Create a fresh key from local dashboard (`/api/keys`)
+2. Run cloud sync: Enable Cloud → Sync Now
+3. Old/non-synced keys can still return `401` on cloud
+
+---
## Docker Issues
### CLI Tool Shows Not Installed
-1. Tjek runtime-felter: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
-2. For bærbar tilstand: brug billedmål "runner-cli" (bundtet CLI'er)
-3. For værtsmonteringstilstand: indstil `CLI_EXTRA_PATHS` og monter host bin-mappe som skrivebeskyttet
-4. Hvis `installed=true` og `runnable=false`: binær blev fundet, men sundhedstjekket mislykkedes### Quick Runtime Validation
+1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
+2. For portable mode: use image target `runner-cli` (bundled CLIs)
+3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only
+4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck
+
+### Quick Runtime Validation
```bash
curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
@@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,
### High Costs
-1. Tjek brugsstatistik i Dashboard → Brug
-2. Skift primær model til GLM/MiniMax
-3. Brug gratis niveau (Gemini CLI, Qoder) til ikke-kritiske opgaver
-4. Indstil omkostningsbudgetter pr. API-nøgle: Dashboard → API-nøgler → Budget---
+1. Check usage stats in Dashboard → Usage
+2. Switch primary model to GLM/MiniMax
+3. Use free tier (Gemini CLI, Qoder) for non-critical tasks
+4. Set cost budgets per API key: Dashboard → API Keys → Budget
+
+---
## Debugging
### Enable Request Logs
-Indstil `ENABLE_REQUEST_LOGS=true` i din `.env`-fil. Logs vises under mappen `logs/`.### Check Provider Health
+Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory.
+
+### Check Provider Health
```bash
# Health dashboard
@@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health
### Runtime Storage
-- Hovedtilstand: `${DATA_DIR}/storage.sqlite` (udbydere, kombinationer, aliaser, nøgler, indstillinger)
-- Anvendelse: SQLite-tabeller i `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + valgfri `${DATA_DIR}/log.txt` og `${DATA_DIR}/call_logs/`
-- Anmodningslogfiler: `/logs/...` (når `ENABLE_REQUEST_LOGS=true`)---
+- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings)
+- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/`
+- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`)
+
+---
## Circuit Breaker Issues
### Provider stuck in OPEN state
-Når en udbyders afbryder er ÅBEN, blokeres anmodninger, indtil nedkølingen udløber.
+When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires.
-**Ret:**
+**Fix:**
-1. Gå til**Dashboard → Indstillinger → Resiliens**
-2. Tjek afbryderkortet for den berørte udbyder
-3. Klik på**Nulstil alle**for at rydde alle afbrydere, eller vent på, at nedkølingen udløber
-4. Bekræft, at udbyderen faktisk er tilgængelig, før du nulstiller### Provider keeps tripping the circuit breaker
+1. Go to **Dashboard → Settings → Resilience**
+2. Check the circuit breaker card for the affected provider
+3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire
+4. Verify the provider is actually available before resetting
-Hvis en udbyder gentagne gange går i ÅBEN tilstand:
+### Provider keeps tripping the circuit breaker
-1. Tjek**Dashboard → Health → Provider Health**for fejlmønsteret
-2. Gå til**Indstillinger → Resiliens → Udbyderprofiler**og øg fejltærsklen
-3. Tjek, om udbyderen har ændret API-grænser eller kræver gengodkendelse
-4. Gennemgå latency-telemetri — høj latenstid kan forårsage timeout-baserede fejl---
+If a provider repeatedly enters OPEN state:
+
+1. Check **Dashboard → Health → Provider Health** for the failure pattern
+2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold
+3. Check if the provider has changed API limits or requires re-authentication
+4. Review latency telemetry — high latency may cause timeout-based failures
+
+---
## Audio Transcription Issues
### "Unsupported model" error
-- Sørg for, at du bruger det korrekte præfiks: `deepgram/nova-3` eller `assemblyai/best`
-- Bekræft, at udbyderen er tilsluttet i**Dashboard → Udbydere**### Transcription returns empty or fails
+- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best`
+- Verify the provider is connected in **Dashboard → Providers**
-- Tjek understøttede lydformater: "mp3", "wav", "m4a", "flac", "ogg", "webm"
-- Bekræft filstørrelsen er inden for udbyderens grænser (typisk < 25 MB)
-- Tjek gyldigheden af udbyderens API-nøgle på udbyderkortet---
+### Transcription returns empty or fails
+
+- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm`
+- Verify file size is within provider limits (typically < 25MB)
+- Check provider API key validity in the provider card
+
+---
## Translator Debugging
-Brug**Dashboard → Oversætter**til at fejlfinde problemer med formatoversættelse:
+Use **Dashboard → Translator** to debug format translation issues:
-| Tilstand | Hvornår skal man bruge |
-| ---------------- | --------------------------------------------------------------------------------------------------------------- | ------------------------ |
-| **Legeplads** | Sammenlign input/output-formater side om side — indsæt en mislykket anmodning for at se, hvordan den oversættes |
-| **Chattester** | Send livebeskeder og inspicer den fulde anmodnings-/svarnyttelast inklusive overskrifter |
-| **Testbænk** | Kør batchtest på tværs af formatkombinationer for at finde ud af, hvilke oversættelser der er brudte |
-| **Live Monitor** | Se anmodningsflow i realtid for at fange periodiske oversættelsesproblemer | ### Common format issues |
+| Mode | When to Use |
+| ---------------- | -------------------------------------------------------------------------------------------- |
+| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates |
+| **Chat Tester** | Send live messages and inspect the full request/response payload including headers |
+| **Test Bench** | Run batch tests across format combinations to find which translations are broken |
+| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues |
--**Tænke-tags vises ikke**— Tjek, om måludbyderen understøtter tænkning og indstilling af tænkebudget -**Værktøjsopkald falder**— Nogle formatoversættelser kan fjerne ikke-understøttede felter; verificere i Playground-tilstand -**Systemprompt mangler**— Claude og Gemini håndterer systemprompts forskelligt; kontrollere oversættelsesoutput -**SDK returnerer rå streng i stedet for objekt**— Rettet i v1.1.0: svar sanitizer fjerner nu ikke-standard felter (`x_groq`, `usage_breakdown` osv.), der forårsager OpenAI SDK Pydantic valideringsfejl -**GLM/ERNIE afviser 'system'-rolle**— Rettet i v1.1.0: Rollenormalisering flettes automatisk systemmeddelelser ind i brugermeddelelser for inkompatible modeller -**"udvikler"-rolle ikke genkendt**- Rettet i v1.1.0: automatisk konverteret til "system" for ikke-OpenAI-udbydere -**`json_schema` virker ikke med Gemini**- Rettet i v1.1.0: `response_format` er nu konverteret til Gemini's `responseMimeType` + `responseSchema`---
+### Common format issues
+
+- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting
+- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode
+- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output
+- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures
+- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models
+- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers
+- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema`
+
+---
## Resilience Settings
### Auto rate-limit not triggering
-- Automatisk hastighedsgrænse gælder kun for API-nøgleudbydere (ikke OAuth/abonnement)
-- Bekræft, at**Indstillinger → Modstandsdygtighed → Udbyderprofiler**har aktiveret automatisk satsgrænse
-- Tjek, om udbyderen returnerer '429'-statuskoder eller 'Retry-After'-overskrifter### Tuning exponential backoff
+- Auto rate-limit only applies to API key providers (not OAuth/subscription)
+- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled
+- Check if the provider returns `429` status codes or `Retry-After` headers
-Udbyderprofiler understøtter disse indstillinger:
+### Tuning exponential backoff
--**Base delay**— Indledende ventetid efter første fejl (standard: 1s) -**Maksimal forsinkelse**— Maksimal ventetid (standard: 30s) -**Multiplikator**— Hvor meget skal forsinkelsen øges pr. på hinanden følgende fejl (standard: 2x)### Anti-thundering herd
+Provider profiles support these settings:
-Når mange samtidige anmodninger rammer en hastighedsbegrænset udbyder, bruger OmniRoute mutex + automatisk hastighedsbegrænsning til at serialisere anmodninger og forhindre kaskadefejl. Dette er automatisk for API-nøgleudbydere.---
+- **Base delay** — Initial wait time after first failure (default: 1s)
+- **Max delay** — Maximum wait time cap (default: 30s)
+- **Multiplier** — How much to increase delay per consecutive failure (default: 2x)
+
+### Anti-thundering herd
+
+When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers.
+
+---
## Optional RAG / LLM failure taxonomy (16 problems)
-Nogle OmniRoute-brugere placerer gatewayen foran RAG- eller agentstakke. I disse opsætninger er det almindeligt at se et mærkeligt mønster: OmniRoute ser sund ud (udbydere op, routing profiler ok, ingen hastighedsgrænse advarsler), men det endelige svar er stadig forkert.
+Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong.
-I praksis kommer disse hændelser normalt fra RAG-rørledningen nedstrøms, ikke fra selve gatewayen.
+In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself.
-Hvis du ønsker et fælles ordforråd til at beskrive disse fejl, kan du bruge WFGY ProblemMap, en ekstern MIT-licenstekstressource, der definerer seksten tilbagevendende RAG/LLM-fejlmønstre. På et højt niveau dækker det over:
+If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers:
-- genfindingsdrift og brudte kontekstgrænser
-- tomme eller uaktuelle indekser og vektorlagre
-- indlejring versus semantisk mismatch
-- problemer med hurtig montering og kontekstvindue
-- logisk sammenbrud og oversikre svar
-- lang kæde og agentkoordinationsfejl
-- multiagent hukommelse og rolledrift
-- problemer med implementering og bootstrap-bestilling
+- retrieval drift and broken context boundaries
+- empty or stale indexes and vector stores
+- embedding versus semantic mismatch
+- prompt assembly and context window issues
+- logic collapse and overconfident answers
+- long chain and agent coordination failures
+- multi agent memory and role drift
+- deployment and bootstrap ordering problems
-Ideen er enkel:
+The idea is simple:
-1. Når du undersøger et dårligt svar, skal du fange:
- - brugeropgave og anmodning
- - rute eller udbyderkombination i OmniRoute
- - enhver RAG-kontekst, der bruges downstream (hentede dokumenter, værktøjsopkald osv.)
-2. Kortlæg hændelsen til et eller to WFGY ProblemMap-numre (`No.1` … `No.16`).
-3. Gem nummeret i dit eget dashboard, runbook eller hændelsessporing ved siden af OmniRoute-logfilerne.
-4. Brug den tilsvarende WFGY-side til at beslutte, om du skal ændre din RAG-stack, retriever eller routingstrategi.
+1. When you investigate a bad response, capture:
+ - user task and request
+ - route or provider combo in OmniRoute
+ - any RAG context used downstream (retrieved documents, tool calls, etc)
+2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`).
+3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs.
+4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy.
-Fuld tekst og konkrete opskrifter live her (MIT-licens, kun tekst):
+Full text and concrete recipes live here (MIT license, text only):
[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md)
-Du kan ignorere dette afsnit, hvis du ikke kører RAG eller agentpipelines bag OmniRoute.---
+You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute.
+
+---
## Still Stuck?
--**GitHub-problemer**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architecture**: Se [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for interne detaljer -**API-reference**: Se [`docs/API_REFERENCE.md`](API_REFERENCE.md) for alle endepunkter -**Health Dashboard**: Tjek**Dashboard → Health**for systemstatus i realtid -**Oversætter**: Brug**Dashboard → Oversætter**til at fejlsøge formatproblemer
+- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details
+- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints
+- **Health Dashboard**: Check **Dashboard → Health** for real-time system status
+- **Translator**: Use **Dashboard → Translator** to debug format issues
diff --git a/docs/i18n/da/llm.txt b/docs/i18n/da/llm.txt
new file mode 100644
index 0000000000..e245fa1d2d
--- /dev/null
+++ b/docs/i18n/da/llm.txt
@@ -0,0 +1,476 @@
+# OmniRoute (Dansk)
+
+🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt)
+
+---
+
+
+> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app.
+
+## Overblik
+
+OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free.
+
+**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost.
+
+**Current version:** 3.5.5
+
+## Tech Stack
+
+- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`)
+- **Framework:** Next.js 16 (App Router) with TypeScript 5.9
+- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations)
+- **State management:** Zustand (client), SQLite (server persistence)
+- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons
+- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth
+- **Schemas:** Zod v4 for all API / MCP input validation
+- **Background jobs:** Custom token health check scheduler, 24h model auto-sync
+- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses
+- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine
+- **i18n:** next-intl with 30 languages
+- **Desktop:** Electron (cross-platform: Windows, macOS, Linux)
+- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`)
+
+## Project Structure
+
+```
+/
+├── src/ # Main application source
+│ ├── app/ # Next.js App Router pages and API routes
+│ │ ├── (dashboard)/ # Dashboard UI pages
+│ │ │ └── dashboard/
+│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents)
+│ │ │ ├── analytics/ # Usage analytics and charts
+│ │ │ ├── api-manager/ # API key management
+│ │ │ ├── audit/ # Audit logs
+│ │ │ ├── auto-combo/ # Auto-combo engine dashboard
+│ │ │ ├── cache/ # Cache dashboard (semantic cache stats)
+│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.)
+│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates)
+│ │ │ ├── costs/ # Cost tracking per provider/model
+│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs
+│ │ │ ├── health/ # System health (uptime, circuit breakers, latency)
+│ │ │ ├── limits/ # Rate limits dashboard
+│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed)
+│ │ │ ├── media/ # Image/video/music generation + transcription
+│ │ │ ├── memory/ # Memory system dashboard
+│ │ │ ├── onboarding/ # Onboarding wizard
+│ │ │ ├── playground/ # Model playground (Monaco editor, streaming)
+│ │ │ ├── providers/ # Provider management (OAuth + API key + free)
+│ │ │ ├── search-tools/ # Search tools configuration
+│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced)
+│ │ │ ├── skills/ # Skills system dashboard
+│ │ │ ├── translator/ # Format translator + debug tools
+│ │ │ └── usage/ # Usage history
+│ │ ├── api/ # REST API endpoints (51 route directories)
+│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings,
+│ │ │ │ # images, audio, videos, music, moderations, rerank, search,
+│ │ │ │ # responses, messages, registered-keys, quotas, accounts)
+│ │ │ ├── v1beta/ # Gemini-compatible API
+│ │ │ ├── a2a/ # A2A agent management API
+│ │ │ ├── acp/ # ACP agent management API
+│ │ │ ├── oauth/ # OAuth flows per provider
+│ │ │ ├── providers/ # Provider CRUD and batch testing
+│ │ │ ├── models/ # Dashboard model listing and aliases
+│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains)
+│ │ │ ├── memory/ # Memory system API
+│ │ │ ├── skills/ # Skills system API
+│ │ │ ├── evals/ # Eval runner API
+│ │ │ ├── mcp/ # MCP HTTP transport API
+│ │ │ ├── search/ # Search provider API
+│ │ │ ├── webhooks/ # Webhook management
+│ │ │ ├── tunnels/ # Cloudflare tunnel management
+│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.)
+│ │ ├── landing/ # Landing page
+│ │ ├── login/ # Login page
+│ │ ├── forgot-password/ # Password recovery
+│ │ ├── status/ # Status page
+│ │ └── docs/ # In-app documentation
+│ ├── domain/ # Domain types and policy engine
+│ │ ├── policyEngine.ts # Central policy engine
+│ │ ├── comboResolver.ts # Combo resolution logic
+│ │ ├── costRules.ts # Cost calculation rules
+│ │ ├── degradation.ts # Graceful degradation
+│ │ ├── fallbackPolicy.ts # Fallback behavior
+│ │ ├── lockoutPolicy.ts # Account lockout logic
+│ │ ├── modelAvailability.ts # Model availability checks
+│ │ ├── providerExpiration.ts # Provider credential expiration
+│ │ ├── quotaCache.ts # Quota caching layer
+│ │ ├── configAudit.ts # Configuration auditing
+│ │ └── responses.ts # Domain response types
+│ ├── i18n/ # Internationalization
+│ │ └── messages/ # 30 language JSON files
+│ ├── lib/ # Core libraries
+│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server
+│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting)
+│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup
+│ │ │ └── streaming.ts # SSE streaming for A2A
+│ │ ├── acp/ # Agent Communication Protocol registry and manager
+│ │ ├── compliance/ # Compliance policy engine
+│ │ ├── db/ # SQLite database layer (21 modules + migrations)
+│ │ │ ├── core.ts # Database initialization, connection, schema
+│ │ │ ├── providers.ts # Provider connection CRUD
+│ │ │ ├── models.ts # Model catalog management
+│ │ │ ├── combos.ts # Combo configuration
+│ │ │ ├── apiKeys.ts # API key management
+│ │ │ ├── settings.ts # Settings persistence
+│ │ │ ├── backup.ts # Database backup/restore
+│ │ │ ├── proxies.ts # Proxy registry
+│ │ │ ├── prompts.ts # Prompt templates
+│ │ │ ├── webhooks.ts # Webhook subscriptions
+│ │ │ ├── detailedLogs.ts # Detailed request logging
+│ │ │ ├── domainState.ts # Domain state persistence
+│ │ │ ├── registeredKeys.ts # Registered API keys with quotas
+│ │ │ ├── quotaSnapshots.ts # Quota snapshot history
+│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings
+│ │ │ ├── cliToolState.ts # CLI tool state tracking
+│ │ │ ├── encryption.ts # Data encryption
+│ │ │ ├── readCache.ts # Read-through cache layer
+│ │ │ ├── secrets.ts # Secrets management
+│ │ │ ├── stateReset.ts # State reset utilities
+│ │ │ ├── migrationRunner.ts # Schema migration runner
+│ │ │ └── migrations/ # 16 SQL migration files
+│ │ ├── evals/ # Eval runner and scheduler
+│ │ ├── memory/ # Persistent conversational memory
+│ │ │ ├── extraction.ts # Memory extraction from conversations
+│ │ │ ├── injection.ts # Memory injection into context
+│ │ │ ├── retrieval.ts # Memory retrieval/search
+│ │ │ ├── store.ts # Memory persistence layer
+│ │ │ └── summarization.ts # Memory summarization
+│ │ ├── oauth/ # OAuth providers, services, and utilities
+│ │ │ ├── constants/ # Default OAuth credentials (overridable via env)
+│ │ │ ├── providers/ # Provider-specific OAuth configs
+│ │ │ ├── services/ # Provider-specific token exchange logic
+│ │ │ └── utils/ # PKCE, callback server, token helpers
+│ │ ├── plugins/ # Plugin system
+│ │ ├── skills/ # Extensible skill framework
+│ │ │ ├── registry.ts # Skill registration
+│ │ │ ├── executor.ts # Skill execution engine
+│ │ │ ├── sandbox.ts # Skill sandbox environment
+│ │ │ ├── builtin/ # Built-in skills
+│ │ │ ├── interception.ts # Skill request interception
+│ │ │ └── injection.ts # Skill context injection
+│ │ ├── usage/ # Usage tracking system
+│ │ │ ├── callLogs.ts # Call log persistence
+│ │ │ ├── costCalculator.ts # Cost calculation engine
+│ │ │ └── usageHistory.ts # Usage history queries
+│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers
+│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management
+│ │ ├── pricingSync.ts # LiteLLM pricing data sync
+│ │ ├── semanticCache.ts # Semantic caching layer
+│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler
+│ │ ├── webhookDispatcher.ts # Webhook event dispatcher
+│ │ └── localDb.ts # Unified re-export layer for all DB modules
+│ ├── middleware/ # Request middleware
+│ │ └── promptInjectionGuard.ts # Prompt injection detection
+│ ├── mitm/ # MITM proxy capability
+│ │ ├── cert/ # Certificate management
+│ │ ├── dns/ # DNS handling
+│ │ ├── targets/ # Target routing
+│ │ └── manager.ts # MITM proxy manager
+│ ├── shared/ # Shared utilities, components, and constants
+│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.)
+│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes
+│ │ ├── contracts/ # Shared API contracts
+│ │ ├── hooks/ # React hooks
+│ │ ├── middleware/ # Shared middleware utilities
+│ │ ├── schemas/ # Shared Zod schemas
+│ │ ├── services/ # Shared services
+│ │ ├── types/ # Shared TypeScript types
+│ │ ├── validation/ # Zod schemas (settings, providers, routes)
+│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID)
+│ ├── sse/ # SSE proxy pipeline
+│ │ ├── services/ # Auth resolution, format translation, response handling
+│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency
+│ ├── store/ # Zustand client-side stores (theme, providers, etc.)
+│ └── types/ # TypeScript type definitions
+├── open-sse/ # Standalone SSE server (npm workspace)
+│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video,
+│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models)
+│ ├── executors/ # Provider-specific request executors (14 executors)
+│ │ ├── base.ts # Base executor with shared logic
+│ │ ├── default.ts # Default OpenAI-compatible executor
+│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum)
+│ │ ├── codex.ts # OpenAI Codex CLI
+│ │ ├── antigravity.ts # Antigravity IDE
+│ │ ├── github.ts # GitHub Copilot
+│ │ ├── gemini-cli.ts # Gemini CLI
+│ │ ├── kiro.ts # Kiro AI
+│ │ ├── qoder.ts # Qoder AI
+│ │ ├── vertex.ts # Vertex AI (Service Account JSON)
+│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI
+│ │ ├── opencode.ts # OpenCode Zen/Go
+│ │ ├── pollinations.ts # Pollinations AI
+│ │ └── puter.ts # Puter AI
+│ ├── handlers/ # Request handlers per API type (11 handlers)
+│ │ ├── chatCore.ts # Main chat completions handler
+│ │ ├── responsesHandler.ts # OpenAI Responses API handler
+│ │ ├── embeddings.ts # Embedding generation
+│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.)
+│ │ ├── videoGeneration.ts # Video generation
+│ │ ├── musicGeneration.ts # Music generation
+│ │ ├── audioSpeech.ts # Text-to-speech
+│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI)
+│ │ ├── moderations.ts # Content moderation
+│ │ ├── rerank.ts # Reranking API
+│ │ └── search.ts # Web search API
+│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP)
+│ │ ├── server.ts # MCP server core (tool registration, scope enforcement)
+│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools)
+│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a)
+│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes)
+│ │ ├── audit.ts # Tool call audit logging
+│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat
+│ │ └── httpTransport.ts # HTTP transport handler
+│ ├── services/ # 36+ service modules
+│ │ ├── combo.ts # Core routing engine
+│ │ ├── usage.ts # Usage tracking
+│ │ ├── tokenRefresh.ts # OAuth token refresh
+│ │ ├── rateLimitManager.ts # Rate limit management
+│ │ ├── accountFallback.ts # Multi-account fallback
+│ │ ├── sessionManager.ts # Session management
+│ │ ├── wildcardRouter.ts # Wildcard model routing
+│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration)
+│ │ ├── intentClassifier.ts # Request intent classification
+│ │ ├── taskAwareRouter.ts # Task-aware routing
+│ │ ├── thinkingBudget.ts # Thinking budget management
+│ │ ├── contextManager.ts # Context window management
+│ │ ├── modelDeprecation.ts # Model deprecation handling
+│ │ ├── modelFamilyFallback.ts # Intra-family model fallback
+│ │ ├── emergencyFallback.ts # Emergency fallback
+│ │ ├── workflowFSM.ts # Workflow state machine
+│ │ ├── backgroundTaskDetector.ts # Background task detection
+│ │ ├── ipFilter.ts # IP-based access control
+│ │ ├── signatureCache.ts # CLI signature caching
+│ │ ├── volumeDetector.ts # Request volume detection
+│ │ ├── contextHandoff.ts # Context relay handoff generation and injection
+│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay
+│ │ └── ... # Additional services (14 more modules)
+│ ├── transformer/ # Responses API transformer
+│ │ └── responsesTransformer.ts
+│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek)
+│ │ ├── request/ # Request translators per provider
+│ │ ├── response/ # Response translators per provider
+│ │ ├── helpers/ # Translation helpers
+│ │ └── image/ # Image format translation
+│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.)
+├── electron/ # Electron desktop app (cross-platform)
+│ ├── main.js # Electron main process
+│ ├── preload.js # Preload script (IPC bridge)
+│ └── assets/ # App icons and assets
+├── tests/ # Test suites
+│ ├── unit/ # 122 unit test files
+│ ├── integration/ # Integration tests
+│ ├── e2e/ # Playwright E2E tests
+│ ├── security/ # Security tests
+│ ├── translator/ # Translator-specific tests
+│ └── load/ # Load tests
+├── docs/ # Documentation
+│ ├── i18n/ # 30-language translated docs
+│ ├── ARCHITECTURE.md # Full architecture documentation
+│ ├── API_REFERENCE.md # API reference
+│ ├── USER_GUIDE.md # User guide
+│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview
+│ ├── CLI-TOOLS.md # CLI tools integration guide
+│ ├── A2A-SERVER.md # A2A agent protocol documentation
+│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring)
+│ ├── MCP-SERVER.md # MCP server (25 tools)
+│ ├── TROUBLESHOOTING.md # Troubleshooting guide
+│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide
+│ ├── openapi.yaml # OpenAPI specification
+│ └── screenshots/ # Dashboard screenshots
+├── bin/ # CLI entry points (omniroute, reset-password)
+├── scripts/ # Build and utility scripts
+└── .env.example # Environment variable template
+```
+
+## Key Features (v3.5.5)
+
+### Core Proxy
+- **60+ AI providers** with automatic format translation
+- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible)
+- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay
+- **4-tier fallback**: Subscription → API Key → Cheap → Free
+- **Context Relay strategy**: Session handoff summaries on account rotation for continuity
+- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown
+- **Semantic caching** with cache hit/miss headers
+- **Idempotency** with configurable dedup window
+- **Circuit breaker** per provider with configurable thresholds
+- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback
+- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers
+- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement
+- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization
+- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills
+- **Prompt Injection Guard**: Middleware-level prompt injection detection
+- **MITM Proxy**: Certificate management, DNS handling, and target routing
+- **Cloudflare Tunnels**: Managed tunnel creation for remote access
+- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches)
+
+### Sikkerhed
+- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs)
+- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()`
+- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()`
+- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection
+- **CLI Fingerprint Matching** — Per-provider request signature matching
+- **Prompt injection guard** — Request middleware detection
+- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`)
+- **PII sanitizer** — Sensitive data scrubbing in logs
+
+### Dashboard Pages (23 sections)
+- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons
+- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies
+- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics
+- **Analytics** — Token consumption, cost, heatmaps, distributions
+- **Health** — Uptime, memory, latency percentiles, circuit breakers
+- **Logs** — Request, Proxy, Audit, Console (tabbed)
+- **Audit** — Audit trail and compliance logging
+- **Costs** — Cost tracking per provider/model
+- **Limits** — Rate limit monitoring
+- **Cache** — Semantic cache statistics and management
+- **CLI Tools** — One-click configuration for 10+ AI CLI tools
+- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration
+- **Playground** — Test any model with Monaco editor, streaming responses
+- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files)
+- **Search Tools** — Search provider configuration and testing
+- **Memory** — Memory system management and visualization
+- **Skills** — Skills framework management and execution
+- **Translator** — Format debugging: playground, chat tester, test bench, live monitor
+- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced
+- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed)
+- **Onboarding** — Setup wizard for new users
+- **Usage** — Usage history and analytics
+- **API Manager** — API key management with scoped permissions
+
+### Protocol Support
+- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations`
+- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens`
+- **OpenAI Responses** — `/v1/responses`
+- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}`
+- **Ollama** — `/v1/api/chat`, `/api/tags`
+- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily)
+- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP)
+- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills)
+- **ACP** — Agent Communication Protocol registry and manager
+
+### MCP Server (25 Tools)
+| Category | Tools |
+|-----------|-------|
+| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` |
+| Memory (3) | `memory_search`, `memory_add`, `memory_clear` |
+| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` |
+
+**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience`
+
+### Provider Categories
+
+**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI
+
+**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline
+
+**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan
+
+**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs
+
+### Internationalization
+- 30 languages for UI (all dashboard pages)
+- 30 translated documentation sets in docs/i18n/
+- Language switcher in documentation
+
+## Key Architectural Decisions
+
+1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints.
+
+2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently.
+
+3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation.
+
+4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity.
+
+5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client.
+
+6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes.
+
+7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`).
+
+8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages.
+
+9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations.
+
+10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`.
+
+11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation.
+
+12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline.
+
+13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast.
+
+## Main Flows
+
+### Proxy Request Flow
+1. Client sends OpenAI-format request to `/v1/chat/completions`
+2. API key validation
+3. Model resolution: direct model or combo lookup
+4. For combos: iterate through models with selected strategy
+5. Auth resolution: get credentials for the target provider
+6. Format translation: OpenAI → provider native format
+7. CLI fingerprint matching (if enabled for provider)
+8. Upstream request with circuit breaker and rate limiting
+9. Response translation: provider → OpenAI format
+10. omniModel tag sanitization (strip internal tags)
+11. SSE streaming back to client
+12. Memory extraction (if memory system enabled)
+13. Usage logging and cost calculation
+
+### OAuth Flow
+1. Dashboard initiates `/api/oauth/[provider]/authorize`
+2. User completes OAuth login in browser
+3. Callback hits `/api/oauth/[provider]/exchange`
+4. Tokens stored as a provider connection in SQLite
+5. Background job refreshes tokens before expiry
+
+## Important Notes for LLMs
+
+1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only).
+
+2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`).
+
+3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services.
+
+4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`.
+
+5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module.
+
+6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`).
+
+7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes.
+
+8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB.
+
+9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown.
+
+10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130).
+
+11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux).
+
+12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint.
+
+13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API.
+
+14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API.
+
+15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time.
+
+16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`.
+
+17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22.
+
+18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false.
+
+## Links
+
+- Repository: https://github.com/diegosouzapw/OmniRoute
+- Website: https://omniroute.online
+- npm: https://www.npmjs.com/package/omniroute
+- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute
+- Documentation: See `/docs/` directory
diff --git a/docs/i18n/de/README.md b/docs/i18n/de/README.md
index 8a56a17378..3019d440b4 100644
--- a/docs/i18n/de/README.md
+++ b/docs/i18n/de/README.md
@@ -4,11 +4,14 @@
---
+
### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback.
-_Ihr universeller API-Proxy – ein Endpunkt, über 60 Anbieter, keine Ausfallzeiten. Jetzt mit**MCP Server (25 Tools)**,**A2A-Protokoll**,**Speicher-/Skills-Systeme**und**Electron Desktop App**._
+_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._
-**Chat-Abschlüsse • Einbettungen • Bildgenerierung • Video • Musik • Audio • Reranking •**Websuche**• MCP-Server • A2A-Protokoll • 100 % TypeScript**---
+**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript**
+
+---
@@ -39,9 +42,13 @@ _Ihr universeller API-Proxy – ein Endpunkt, über 60 Anbieter, keine Ausfallze
[](https://omniroute.online)
[](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-[🌐 Website](https://omniroute.online) • [🚀 Schnellstart](#-quick-start) • [💡 Funktionen](#-key-features) • [📖 Dokumente](#-documentation) • [💰 Preise](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-🌐**Verfügbar in:**🇺🇸 [Englisch](README.md) | 🇧🇷 [Português (Brasilien)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italienisch](docs/i18n/it/README.md) | 🇷🇺 [Russisch](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dänisch](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Niederlande](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polnisch](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)---
+
+
+🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)
+
+---
## 🖼️ Main Dashboard
@@ -53,28 +60,30 @@ _Ihr universeller API-Proxy – ein Endpunkt, über 60 Anbieter, keine Ausfallze
## 📸 Dashboard Preview
-
-Klicken Sie hier, um Dashboard-Screenshots anzuzeigen
+
+Click to see dashboard screenshots
-| Seite | Screenshot |
-| ---------------------- | -------------------------------------------------- | ---------- |
-| **Anbieter** |  |
-| **Kombinationen** |  |
-| **Analytik** |  |
-| **Gesundheit** |  |
-| **Übersetzer** |  |
-| **Einstellungen** |  |
-| **CLI-Tools** |  |
-| **Nutzungsprotokolle** |  |
-| **Endpunkte** |  | |
+| Page | Screenshot |
+| -------------- | ------------------------------------------------- |
+| **Providers** |  |
+| **Combos** |  |
+| **Analytics** |  |
+| **Health** |  |
+| **Translator** |  |
+| **Settings** |  |
+| **CLI Tools** |  |
+| **Usage Logs** |  |
+| **Endpoints** |  |
+
+
---
### 🤖 Free AI Provider for your favorite coding agents
-_Verbinden Sie jedes KI-gestützte IDE- oder CLI-Tool über OmniRoute – kostenloses API-Gateway für unbegrenzte Codierung._
+_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._
-
+
@@ -88,28 +97,28 @@ _Verbinden Sie jedes KI-gestützte IDE- oder CLI-Tool über OmniRoute – kosten

NanoBot
- ⭐ 20,9K
+ ⭐ 20.9K
|

PicoClaw
- ⭐ 14,6K
+ ⭐ 14.6K
|

ZeroClaw
- ⭐ 9,9K
+ ⭐ 9.9K
|

- Eisenklaue
+ IronClaw
- ⭐ 2,1K
+ ⭐ 2.1K
|
@@ -123,483 +132,557 @@ _Verbinden Sie jedes KI-gestützte IDE- oder CLI-Tool über OmniRoute – kosten

- Codex-CLI
+ Codex CLI
- ⭐ 60,8K
+ ⭐ 60.8K
|

Claude Code
- ⭐ 67,3K
+ ⭐ 67.3K
|

- Gemini-CLI
+ Gemini CLI
- ⭐ 94,7K
+ ⭐ 94.7K
|

- Kilo-Code
+ Kilo Code
- ⭐ 15,5K
+ ⭐ 15.5K
|
-📡 Alle Agenten verbinden sich über http://localhost:20128/v1 oder http://cloud.omniroute.online/v1 – eine Konfiguration, unbegrenzte Modelle und Kontingent---
+📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota
+
+---
## 🤔 Why OmniRoute?
-**Hören Sie auf, Geld zu verschwenden und an Grenzen zu stoßen:**
+**Stop wasting money and hitting limits:**
--
Das Abonnementkontingent läuft jeden Monat ungenutzt ab
--
Ratenbeschränkungen stoppen Sie mitten beim Codieren
- –
Teure APIs (20–50 $/Monat pro Anbieter)
--
Manueller Wechsel zwischen Anbietern
+-
Subscription quota expires unused every month
+-
Rate limits stop you mid-coding
+-
Expensive APIs ($20-50/month per provider)
+-
Manual switching between providers
-**OmniRoute löst dieses Problem:**
+**OmniRoute solves this:**
-- ✅**Abonnements maximieren**- Verfolgen Sie das Kontingent, nutzen Sie jedes Bit vor dem Zurücksetzen
-- ✅**Auto-Fallback**– Abonnement → API-Schlüssel → Günstig → Kostenlos, keine Ausfallzeiten
-- ✅**Mehrere Konten**– Round-Robin zwischen Konten pro Anbieter
-- ✅**Universell**– Funktioniert mit Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw und jedem CLI-Tool---
+- ✅ **Maximize subscriptions** - Track quota, use every bit before reset
+- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime
+- ✅ **Multi-account** - Round-robin between accounts per provider
+- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool
+
+---
## 📧 Support
-> 💬**Treten Sie unserer Community bei!**[WhatsApp-Gruppe](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) – Holen Sie sich Hilfe, tauschen Sie Tipps aus und bleiben Sie auf dem Laufenden.
+> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated.
--**Website**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Probleme**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Community-Gruppe](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Mitwirken**: Siehe [CONTRIBUTING.md](CONTRIBUTING.md), öffnen Sie eine PR oder wählen Sie eine „gute erste Ausgabe“ aus -**Originalprojekt**: [9router von decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug?
+- **Website**: [omniroute.online](https://omniroute.online)
+- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute)
+- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue`
+- **Original Project**: [9router by decolua](https://github.com/decolua/9router)
-Wenn Sie ein Problem öffnen, führen Sie bitte den Befehl „system-info“ aus und hängen Sie die generierte Datei an:```bash
+### 🐛 Reporting a Bug?
+
+When opening an issue, please run the system-info command and attach the generated file:
+
+```bash
npm run system-info
-
```
-Dadurch wird eine „system-info.txt“ mit Ihrer Node.js-Version, OmniRoute-Version, Betriebssystemdetails, installierten CLI-Tools (Qoder, Gemini, Claude, Codex, Antigravity, Droid usw.), Docker/PM2-Status und Systempaketen generiert – alles, was wir brauchen, um Ihr Problem schnell zu reproduzieren. Hängen Sie die Datei direkt an Ihr GitHub-Problem an.---
+This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue.
+
+---
## 🔄 How It Works
```
-
┌─────────────┐
-│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
-│ Tool │
+│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
+│ Tool │
└──────┬──────┘
-│ http://localhost:20128/v1
-↓
+ │ http://localhost:20128/v1
+ ↓
┌─────────────────────────────────────────┐
-│ OmniRoute (Smart Router) │
-│ • Format translation (OpenAI ↔ Claude) │
-│ • Quota tracking + Embeddings + Images │
-│ • Auto token refresh │
+│ OmniRoute (Smart Router) │
+│ • Format translation (OpenAI ↔ Claude) │
+│ • Quota tracking + Embeddings + Images │
+│ • Auto token refresh │
└──────┬──────────────────────────────────┘
-│
-├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
-│ ↓ quota exhausted
-├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
-│ ↓ budget limit
-├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
-│ ↓ budget limit
-└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
+ │
+ ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
+ │ ↓ quota exhausted
+ ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
+ │ ↓ budget limit
+ ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
+ │ ↓ budget limit
+ └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
Result: Never stop coding, minimal cost
-
-````
+```
---
## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases
->**Jeder Entwickler, der KI-Tools verwendet, ist täglich mit diesen Problemen konfrontiert.**OmniRoute wurde entwickelt, um sie alle zu lösen – von Kostenüberschreitungen bis hin zu regionalen Blockaden, von unterbrochenen OAuth-Flüssen bis hin zu Protokollvorgängen und Unternehmensbeobachtbarkeit.
+> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability.
-
-💸 1. „Ich bezahle ein teures Abonnement, werde aber trotzdem durch Limits unterbrochen“
+
+💸 1. "I pay for an expensive subscription but still get interrupted by limits"
-Entwickler zahlen 20–200 US-Dollar/Monat für Claude Pro, Codex Pro oder GitHub Copilot. Auch wenn das Kontingent bezahlt wird, gibt es eine Obergrenze – 5 Stunden Nutzung, wöchentliche Limits oder Tariflimits pro Minute. Während der Codierungssitzung reagiert der Anbieter nicht mehr und der Entwickler verliert an Fluss und Produktivität.
+Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity.
-**So löst OmniRoute das Problem:**
+**How OmniRoute solves it:**
--**Intelligenter 4-Stufen-Fallback**– Wenn das Abonnementkontingent aufgebraucht ist, wird automatisch zu API Key → Günstig → Kostenlos weitergeleitet, ohne dass ein manueller Eingriff erforderlich ist
--**Verfolgung von Anbieterlimits**– Zwischengespeicherte Kontingent-Snapshots werden nach einem serverseitigen Zeitplan aktualisiert (Standard „PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70“), wobei eine manuelle Aktualisierung in der Benutzeroberfläche verfügbar ist
--**Unterstützung mehrerer Konten**– Mehrere Konten pro Anbieter mit automatischem Round-Robin – wenn eines aufgebraucht ist, wird zum nächsten gewechselt
--**Benutzerdefinierte Kombinationen**– Anpassbare Fallback-Ketten mit 9 Ausgleichsstrategien (Priorität, gewichtet, Fill-First, Round-Robin, P2C, zufällig, am wenigsten genutzt, kostenoptimiert, strikt zufällig)
--**Codex Business Quotas**– Überwachung der Geschäfts-/Team-Arbeitsbereichskontingente direkt im Dashboard
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention
+- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI
+- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next
+- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**)
+- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard
-
-🔌 2. „Ich muss mehrere Anbieter nutzen, aber jeder hat eine andere API“
+
-OpenAI verwendet ein Format, Claude (Anthropic) verwendet ein anderes, Gemini noch ein anderes. Wenn ein Entwickler Modelle verschiedener Anbieter testen oder zwischen ihnen wechseln möchte, muss er SDKs neu konfigurieren, Endpunkte ändern und mit inkompatiblen Formaten umgehen. Benutzerdefinierte Anbieter (FriendLI, NIM) verfügen über nicht standardmäßige Modellendpunkte.
+
+🔌 2. "I need to use multiple providers but each has a different API"
-**So löst OmniRoute das Problem:**
+OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints.
--**Unified Endpoint**– Ein einzelner „http://localhost:20128/v1“ dient als Proxy für alle über 60 Anbieter
--**Formatübersetzung**– Automatisch und transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
--**Antwortbereinigung**– Entfernt nicht standardmäßige Felder („x_groq“, „usage_breakdown“, „service_tier“), die OpenAI SDK v1.83+ beschädigen
--**Rollennormalisierung**– Konvertiert „Entwickler“ → „System“ für Nicht-OpenAI-Anbieter; „System“ → „Benutzer“ für GLM/ERNIE
--**Think Tag Extraction**– Extrahiert „“-Blöcke aus Modellen wie DeepSeek R1 in standardisierten „reasoning_content“.
--**Strukturierte Ausgabe für Gemini**– automatische Konvertierung von „json_schema“ → „responseMimeType“/„responseSchema“.
--**`stream` ist standardmäßig auf `false`**— Entspricht der OpenAI-Spezifikation und vermeidet unerwartetes SSE in Python/Rust/Go-SDKs
+**How OmniRoute solves it:**
-
-🌐 3. „Mein KI-Anbieter blockiert meine Region/mein Land“
+- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers
+- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
+- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+
+- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE
+- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content`
+- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion
+- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs
-Anbieter wie OpenAI/Codex blockieren den Zugriff aus bestimmten geografischen Regionen. Benutzer erhalten bei OAuth- und API-Verbindungen Fehlermeldungen wie „unsupported_country_region_territory“. Dies ist besonders frustrierend für Entwickler aus Entwicklungsländern.
+
-**So löst OmniRoute das Problem:**
+
+🌐 3. "My AI provider blocks my region/country"
--**3-Level-Proxy-Konfiguration**– Konfigurierbarer Proxy auf 3 Ebenen: global (gesamter Datenverkehr), pro Anbieter (nur ein Anbieter) und pro Verbindung/Schlüssel
--**Farbcodierte Proxy-Abzeichen**– Visuelle Indikatoren: 🟢 globaler Proxy, 🟡 Anbieter-Proxy, 🔵 Verbindungs-Proxy, immer mit IP-Adresse
--**OAuth-Token-Austausch über Proxy**– Der OAuth-Fluss läuft auch über den Proxy und löst „unsupported_country_region_territory“.
--**Verbindungstests über Proxy**– Verbindungstests verwenden den konfigurierten Proxy (keine direkte Umgehung mehr)
--**SOCKS5-Unterstützung**– Vollständige SOCKS5-Proxy-Unterstützung für ausgehendes Routing
--**TLS-Fingerabdruck-Spoofing**– Browserähnlicher TLS-Fingerabdruck über „wreq-js“, um die Bot-Erkennung zu umgehen
--**🔏 CLI-Fingerabdruck-Abgleich**– Ordnet Header und Textfelder neu an, damit sie mit nativen CLI-Binärsignaturen übereinstimmen, wodurch das Risiko der Kontokennzeichnung drastisch reduziert wird. Die Proxy-IP bleibt erhalten – Sie erhalten gleichzeitig Stealth**und**IP-Maskierung
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries.
-
-🆓 4. „Ich möchte KI zum Codieren verwenden, habe aber kein Geld“
+**How OmniRoute solves it:**
-Nicht jeder kann 20–200 $/Monat für KI-Abonnements bezahlen. Studenten, Entwickler aus Schwellenländern, Bastler und Freiberufler benötigen Zugang zu hochwertigen Modellen zum Nulltarif.
+- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key
+- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP
+- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory`
+- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass)
+- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection
+- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously
-**So löst OmniRoute das Problem:**
+
--**Integrierte Free-Tier-Anbieter**– Native Unterstützung für 100 % kostenlose Anbieter: Qoder (5 unbegrenzte Modelle über OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unbegrenzte Modelle: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID kostenlos), Gemini CLI (180.000 Token/Monat kostenlos)
--**Ollama Cloud**– Cloud-gehostete Ollama-Modelle unter „api.ollama.com“ mit kostenloser Stufe „Light-Nutzung“; Verwenden Sie das Präfix „ollamacloud/“.
--**Nur kostenlose Combos**– Kette „gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus“ = 0 $/Monat ohne Ausfallzeit
--**NVIDIA NIM Free Access**– Entwickler-für immer kostenloser Zugriff auf über 70 Modelle unter build.nvidia.com mit ca. 40 U/min (Umstellung von Credits auf reine Ratenlimits)
--**Kostenoptimierte Strategie**– Routing-Strategie, die automatisch den günstigsten verfügbaren Anbieter auswählt
+
+🆓 4. "I want to use AI for coding but I have no money"
-
-🔒 5. „Ich muss mein KI-Gateway vor unbefugtem Zugriff schützen“
+Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost.
-Wenn ein KI-Gateway dem Netzwerk (LAN, VPS, Docker) zugänglich gemacht wird, kann jeder mit der Adresse die Token/Kontingente des Entwicklers verbrauchen. Ohne Schutz sind APIs anfällig für Missbrauch, sofortige Injektion und Missbrauch.
+**How OmniRoute solves it:**
-**So löst OmniRoute das Problem:**
+- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free)
+- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix
+- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime
+- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits)
+- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider
--**API-Schlüsselverwaltung**– Generierung, Rotation und Scoping pro Anbieter mit einer dedizierten „/dashboard/api-manager“-Seite
--**Berechtigungen auf Modellebene**– Beschränken Sie API-Schlüssel auf bestimmte Modelle („openai/*“, Platzhaltermuster) mit der Umschaltfunktion „Alle zulassen/Einschränken“.
--**API Endpoint Protection**– Erfordert einen Schlüssel für „/v1/models“ und blockiert bestimmte Anbieter aus der Liste
--**Auth Guard + CSRF-Schutz**– Alle Dashboard-Routen sind mit „withAuth“-Middleware + CSRF-Tokens geschützt
--**Ratenbegrenzer**– Ratenbegrenzung pro IP mit konfigurierbaren Fenstern
--**IP-Filterung**– Zulassungs-/Blockierungsliste für die Zugriffskontrolle
--**Prompt Injection Guard**– Bereinigung gegen bösartige Eingabeaufforderungsmuster
--**AES-256-GCM-Verschlüsselung**– Anmeldeinformationen im Ruhezustand verschlüsselt
+
-
-🛑 6. „Mein Provider ist ausgefallen und ich habe meinen Programmierfluss verloren“
+
+🔒 5. "I need to protect my AI gateway from unauthorized access"
-KI-Anbieter können instabil werden, 5xx-Fehler zurückgeben oder vorübergehende Ratengrenzen erreichen. Wenn ein Entwickler von einem einzelnen Anbieter abhängig ist, wird er unterbrochen. Ohne Schutzschalter können wiederholte Versuche zum Absturz der Anwendung führen.
+When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse.
-**So löst OmniRoute das Problem:**
+**How OmniRoute solves it:**
--**Leistungsschalter pro Modell**– Automatisches Öffnen/Schließen mit konfigurierbaren Schwellenwerten und Abklingzeit (Geschlossen/Offen/Halboffen), je nach Modell, um kaskadierende Blöcke zu vermeiden
--**Exponentielles Backoff**– Progressive Wiederholungsverzögerungen
--**Anti-Thundering Herd**– Mutex + Semaphor-Schutz gegen gleichzeitige Wiederholungsstürme
--**Combo-Fallback-Ketten**– Wenn der primäre Anbieter ausfällt, fällt er automatisch durch die Kette, ohne dass ein Eingreifen erforderlich ist
--**Combo Circuit Breaker**– Deaktiviert automatisch ausgefallene Anbieter innerhalb einer Combo-Kette
--**Gesundheits-Dashboard**– Betriebszeitüberwachung, Leistungsschalterzustände, Sperren, Cache-Statistiken, p50/p95/p99-Latenz
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page
+- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle
+- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing
+- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens
+- **Rate Limiter** — Per-IP rate limiting with configurable windows
+- **IP Filtering** — Allowlist/blocklist for access control
+- **Prompt Injection Guard** — Sanitization against malicious prompt patterns
+- **AES-256-GCM Encryption** — Credentials encrypted at rest
-
-🔧 7. „Die Konfiguration jedes KI-Tools ist mühsam und repetitiv“
+
-Entwickler verwenden Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code ... Jedes Tool benötigt eine andere Konfiguration (API-Endpunkt, Schlüssel, Modell). Eine Neukonfiguration bei einem Anbieter- oder Modellwechsel ist Zeitverschwendung.
+
+🛑 6. "My provider went down and I lost my coding flow"
-**So löst OmniRoute das Problem:**
+AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application.
--**CLI Tools Dashboard**– Spezielle Seite mit Ein-Klick-Einrichtung für Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
--**GitHub Copilot Config Generator**– Erzeugt „chatLanguageModels.json“ für VS-Code mit Massenmodellauswahl
--**Onboarding-Assistent**– Geführte Einrichtung in 4 Schritten für Erstbenutzer
--**Ein Endpunkt, alle Modelle**– Konfigurieren Sie „http://localhost:20128/v1“ einmal und greifen Sie auf über 60 Anbieter zu
+**How OmniRoute solves it:**
-
-🔑 8. „Die Verwaltung von OAuth-Tokens von mehreren Anbietern ist die Hölle“
+- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks
+- **Exponential Backoff** — Progressive retry delays
+- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms
+- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention
+- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain
+- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency
-Claude Code, Codex, Gemini CLI, Copilot – alle verwenden OAuth 2.0 mit ablaufenden Token. Entwickler müssen sich ständig neu authentifizieren, sich mit „client_secret fehlt“, „redirect_uri_mismatch“ und Fehlern auf Remote-Servern befassen. Besonders problematisch ist OAuth auf LAN/VPS.
+
-**So löst OmniRoute das Problem:**
+
+🔧 7. "Configuring each AI tool is tedious and repetitive"
--**Automatische Token-Aktualisierung**– OAuth-Tokens werden vor Ablauf im Hintergrund aktualisiert
--**OAuth 2.0 (PKCE) integriert**– Automatischer Ablauf für Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
--**Multi-Account OAuth**– Mehrere Konten pro Anbieter über JWT/ID-Token-Extraktion
--**OAuth LAN/Remote Fix**– Private IP-Erkennung für „redirect_uri“ + manueller URL-Modus für Remote-Server
--**OAuth hinter Nginx**– Verwendet „window.location.origin“ für Reverse-Proxy-Kompatibilität
--**Remote OAuth Guide**– Schritt-für-Schritt-Anleitung für Google Cloud-Anmeldeinformationen auf VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time.
-
-📊 9. „Ich weiß nicht, wie viel ich wo ausgebe“
+**How OmniRoute solves it:**
-Entwickler nutzen mehrere kostenpflichtige Anbieter, haben jedoch keine einheitliche Sicht auf die Ausgaben. Jeder Anbieter verfügt über ein eigenes Abrechnungs-Dashboard, es gibt jedoch keine konsolidierte Ansicht. Unerwartete Kosten können sich häufen.
+- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
+- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection
+- **Onboarding Wizard** — Guided 4-step setup for first-time users
+- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers
-**So löst OmniRoute das Problem:**
+
--**Kostenanalyse-Dashboard**– Kostenverfolgung pro Token und Budgetverwaltung pro Anbieter
--**Budgetgrenzen pro Stufe**– Ausgabenobergrenze pro Stufe, die einen automatischen Fallback auslöst
--**Preiskonfiguration pro Modell**– Konfigurierbare Preise pro Modell
--**Nutzungsstatistiken pro API-Schlüssel**– Anzahl der Anfragen und zuletzt verwendeter Zeitstempel pro Schlüssel
--**Analytics-Dashboard**– Statistikkarten, Modellnutzungsdiagramm, Anbietertabelle mit Erfolgsraten und Latenz
+
+🔑 8. "Managing OAuth tokens from multiple providers is hell"
-
-🐛 10. „Ich kann Fehler und Probleme bei KI-Aufrufen nicht diagnostizieren“
+Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic.
-Wenn ein Anruf fehlschlägt, weiß der Entwickler nicht, ob es sich um eine Ratenbegrenzung, ein abgelaufenes Token, ein falsches Format oder einen Anbieterfehler handelt. Fragmentierte Protokolle über verschiedene Terminals hinweg. Ohne Beobachtbarkeit ist das Debuggen ein Versuch und Irrtum.
+**How OmniRoute solves it:**
-**So löst OmniRoute das Problem:**
+- **Auto Token Refresh** — OAuth tokens refresh in background before expiration
+- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
+- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction
+- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers
+- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility
+- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker
--**Einheitliches Protokoll-Dashboard**– 4 Registerkarten: Anforderungsprotokolle, Proxy-Protokolle, Audit-Protokolle, Konsole
--**Console Log Viewer**– Echtzeit-Viewer im Terminal-Stil mit farbcodierten Ebenen, automatischem Scrollen, Suche und Filter
--**SQLite-Proxy-Protokolle**– Persistente Protokolle, die Serverneustarts überdauern
--**Translator Playground**– 4 Debugging-Modi: Playground (Formatübersetzung), Chat Tester (Round-Trip), Test Bench (Batch), Live Monitor (Echtzeit)
--**Telemetrie anfordern**– p50/p95/p99-Latenz + X-Request-Id-Ablaufverfolgung
--**Dateibasierte Protokollierung mit Rotation**– App-Protokolle rotieren nach Größe, Aufbewahrungstagen und Archivanzahl; Anrufprotokollartefakte rotieren nach Aufbewahrungstagen und Dateianzahl
--**Systeminfobericht**– „npm run system-info“ generiert „system-info.txt“ mit Ihrer vollständigen Umgebung (Knotenversion, OmniRoute-Version, Betriebssystem, CLI-Tools, Docker/PM2-Status). Hängen Sie es an, wenn Sie Probleme melden, um eine sofortige Einstufung zu ermöglichen.
+
-
-🏗️ 11. „Die Bereitstellung und Wartung des Gateways ist komplex“
+
+📊 9. "I don't know how much I'm spending or where"
-Die Installation, Konfiguration und Wartung eines KI-Proxys in verschiedenen Umgebungen (lokal, VPS, Docker, Cloud) ist arbeitsintensiv. Probleme wie hartcodierte Pfade, „EACCES“ für Verzeichnisse, Portkonflikte und plattformübergreifende Builds sorgen für zusätzliche Reibung.
+Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up.
-**So löst OmniRoute das Problem:**
+**How OmniRoute solves it:**
--**npm globale Installation**– „npm install -g omniroute && omniroute“ – fertig
--**Docker Multi-Platform**– AMD64 + ARM64 nativ (Apple Silicon, AWS Graviton, Raspberry Pi)
--**Docker Compose-Profile**– „base“ (keine CLI-Tools) und „cli“ (mit Claude Code, Codex, OpenClaw)
--**Electron Desktop App**– Native App für Windows/macOS/Linux mit Taskleiste, Autostart, Offline-Modus
--**Split-Port-Modus**– API und Dashboard auf separaten Ports für erweiterte Szenarien (Reverse-Proxy, Container-Netzwerk)
--**Cloud Sync**– Konfigurieren Sie die geräteübergreifende Synchronisierung über Cloudflare Workers
--**DB-Backups**– Automatische Sicherung, Wiederherstellung, Export und Import aller Einstellungen, mit „DISABLE_SQLITE_AUTO_BACKUP“ für extern verwaltete Backups
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider
+- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback
+- **Per-Model Pricing Configuration** — Configurable prices per model
+- **Usage Statistics Per API Key** — Request count and last-used timestamp per key
+- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency
-
-🌍 12. „Die Benutzeroberfläche ist nur auf Englisch verfügbar und mein Team spricht kein Englisch“
+
-Teams in nicht englischsprachigen Ländern, insbesondere in Lateinamerika, Asien und Europa, haben Probleme mit rein englischsprachigen Benutzeroberflächen. Sprachbarrieren verringern die Akzeptanz und erhöhen die Zahl von Konfigurationsfehlern.
+
+🐛 10. "I can't diagnose errors and problems in AI calls"
-**So löst OmniRoute das Problem:**
+When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error.
--**Dashboard i18n – 30 Sprachen**– Alle über 500 Tasten übersetzt, einschließlich Arabisch, Bulgarisch, Dänisch, Deutsch, Spanisch, Finnisch, Französisch, Hebräisch, Hindi, Ungarisch, Indonesisch, Italienisch, Japanisch, Koreanisch, Malaiisch, Niederländisch, Norwegisch, Polnisch, Portugiesisch (PT/BR), Rumänisch, Russisch, Slowakisch, Schwedisch, Thailändisch, Ukrainisch, Vietnamesisch, Chinesisch, Philippinisch, Englisch
--**RTL-Unterstützung**– Rechts-nach-links-Unterstützung für Arabisch und Hebräisch
--**Mehrsprachige READMEs**– 30 vollständige Dokumentationsübersetzungen
--**Sprachauswahl**– Globussymbol in der Kopfzeile zum Umschalten in Echtzeit
+**How OmniRoute solves it:**
-
-🔄 13. „Ich brauche mehr als nur Chat – ich brauche Einbettungen, Bilder, Audio“
+- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console
+- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter
+- **SQLite Proxy Logs** — Persistent logs that survive server restarts
+- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time)
+- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing
+- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count
+- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage.
-KI ist nicht nur der Abschluss eines Chats. Entwickler müssen Bilder generieren, Audio transkribieren, Einbettungen für RAG erstellen, Dokumente neu einordnen und Inhalte moderieren. Jede API hat einen anderen Endpunkt und ein anderes Format.
+
-**So löst OmniRoute das Problem:**
+
+🏗️ 11. "Deploying and maintaining the gateway is complex"
--**Embeddings**– „/v1/embeddings“ mit 6 Anbietern und 9+ Modellen
--**Image Generation**– „/v1/images/generations“ mit 10 Anbietern und über 20 Modellen (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
--**Text-zu-Video**– „/v1/videos/generations“ – ComfyUI (AnimateDiff, SVD) und SD WebUI
--**Text-zu-Musik**– „/v1/music/generations“ – ComfyUI (Stable Audio Open, MusicGen)
--**Audiotranskription**– „/v1/audio/transcriptions“ – Whisper + Nvidia NIM, HuggingFace, Qwen3
--**Text-to-Speech**– „/v1/audio/speech“ – ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + bestehende Anbieter
--**Moderationen**– „/v1/moderations“ – Überprüfung der Inhaltssicherheit
--**Reranking**– „/v1/rerank“ – Neuranking der Dokumentrelevanz
--**Responses API**– Vollständige „/v1/responses“-Unterstützung für Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction.
-
-🧪 14. „Ich habe keine Möglichkeit, die Qualität verschiedener Modelle zu testen und zu vergleichen“
+**How OmniRoute solves it:**
-Entwickler möchten wissen, welches Modell für ihren Anwendungsfall am besten geeignet ist – Code, Übersetzung, Argumentation –, aber ein manueller Vergleich ist langsam. Es sind keine integrierten Evaluierungstools vorhanden.
+- **npm global install** — `npm install -g omniroute && omniroute` — done
+- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi)
+- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw)
+- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode
+- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking)
+- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers
+- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups
-**So löst OmniRoute das Problem:**
+
--**LLM-Bewertungen**– Golden-Set-Test mit 10 vorinstallierten Fällen zu Begrüßungen, Mathematik, Geografie, Codegenerierung, JSON-Konformität, Übersetzung, Markdown und Sicherheitsverweigerung
--**4 Match-Strategien**– „exact“, „contains“, „regex“, „custom“ (JS-Funktion)
--**Translator Playground Test Bench**– Batch-Tests mit mehreren Eingaben und erwarteten Ausgaben, anbieterübergreifender Vergleich
--**Chat-Tester**– Vollständiger Roundtrip mit visueller Antwortwiedergabe
--**Live-Monitor**– Echtzeit-Stream aller Anfragen, die über den Proxy fließen
+
+🌍 12. "The interface is English-only and my team doesn't speak English"
-
-📈 15. „Ich muss skalieren, ohne an Leistung einzubüßen“
+Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors.
-Wenn das Anfragevolumen wächst, verursachen dieselben Fragen ohne Zwischenspeicherung doppelte Kosten. Ohne Idempotenz verschwenden doppelte Anfragen die Verarbeitung. Die Tarifbegrenzungen pro Anbieter müssen eingehalten werden.
+**How OmniRoute solves it:**
-**So löst OmniRoute das Problem:**
+- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English
+- **RTL Support** — Right-to-left support for Arabic and Hebrew
+- **Multi-Language READMEs** — 30 complete documentation translations
+- **Language Selector** — Globe icon in header for real-time switching
--**Semantischer Cache**– Zweistufiger Cache (Signatur + Semantik) reduziert Kosten und Latenz
--**Request Idempotency**– 5-Sekunden-Deduplizierungsfenster für identische Anfragen
--**Ratenbegrenzungserkennung**– Provider-RPM, minimale Lücke und maximale gleichzeitige Verfolgung
--**Bearbeitbare Ratengrenzen**– Konfigurierbare Standardeinstellungen unter Einstellungen → Ausfallsicherheit mit Persistenz
--**API Key Validation Cache**– 3-stufiger Cache für Produktionsleistung
--**Gesundheits-Dashboard mit Telemetrie**– p50/p95/p99-Latenz, Cache-Statistiken, Betriebszeit
+
-
-🤖 16. „Ich möchte das Modellverhalten global steuern“
+
+🔄 13. "I need more than chat — I need embeddings, images, audio"
-Entwickler, die alle Antworten in einer bestimmten Sprache oder mit einem bestimmten Ton wünschen oder die Argumentationstoken einschränken möchten. Dies in jedem Tool/jeder Anfrage zu konfigurieren, ist unpraktisch.
+AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format.
-**So löst OmniRoute das Problem:**
+**How OmniRoute solves it:**
--**System Prompt Injection**– Globale Eingabeaufforderung, die auf alle Anfragen angewendet wird
--**Thinking Budget Validation**– Reasoning-Token-Zuteilungskontrolle pro Anfrage (Passthrough, automatisch, benutzerdefiniert, adaptiv)
--**9 Routing-Strategien**– Globale Strategien, die bestimmen, wie Anfragen verteilt werden
--**Wildcard-Router**– „provider/*“-Muster leiten dynamisch an jeden Anbieter weiter
--**Combo-Aktivierung/Deaktivierung umschalten**– Combos direkt über das Dashboard umschalten
--**Provider Toggle**– Alle Verbindungen für einen Anbieter mit einem Klick aktivieren/deaktivieren
--**Blockierte Anbieter**– Bestimmte Anbieter aus der Liste „/v1/models“ ausschließen
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models
+- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
+- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI
+- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
+- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
+- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers
+- **Moderations** — `/v1/moderations` — Content safety checks
+- **Reranking** — `/v1/rerank` — Document relevance reranking
+- **Responses API** — Full `/v1/responses` support for Codex
-
-🧰 17. „Ich brauche MCP-Tools als erstklassige Produktfunktionen“
+
-Viele KI-Gateways stellen MCP nur als verstecktes Implementierungsdetail zur Verfügung. Teams benötigen eine sichtbare, überschaubare Betriebsebene.
+
+🧪 14. "I have no way to test and compare quality across models"
-**So löst OmniRoute das Problem:**
+Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist.
-– MCP wird in der Dashboard-Navigation und auf der Registerkarte „Endpunktprotokoll“ angezeigt
-- Dedizierte MCP-Verwaltungsseite mit Prozess, Tools, Bereichen und Audit
-– Integrierter Schnellstart für „omniroute --mcp“ und Client-Onboarding
+**How OmniRoute solves it:**
-
-🧠 18. „Ich benötige A2A-Orchestrierung mit Synchronisierungs- und Stream-Aufgabenpfaden“
+- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal
+- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function)
+- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison
+- **Chat Tester** — Full round-trip with visual response rendering
+- **Live Monitor** — Real-time stream of all requests flowing through the proxy
-Agenten-Workflows erfordern sowohl direkte Antworten als auch eine lang andauernde gestreamte Ausführung mit Lebenszykluskontrolle.
+
-**So löst OmniRoute das Problem:**
+
+📈 15. "I need to scale without losing performance"
-- A2A JSON-RPC-Endpunkt („POST /a2a“) mit „message/send“ und „message/stream“.
-- SSE-Streaming mit Terminal-State-Propagierung
-– Task-Lebenszyklus-APIs für „tasks/get“ und „tasks/cancel“.
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected.
-
-🛰️ 19. „Ich brauche einen echten Zustand des MCP-Prozesses, keinen erratenen Status“
+**How OmniRoute solves it:**
-Betriebsteams müssen wissen, ob MCP tatsächlich aktiv ist, und nicht nur, ob eine API erreichbar ist.
+- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency
+- **Request Idempotency** — 5s deduplication window for identical requests
+- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking
+- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence
+- **API Key Validation Cache** — 3-tier cache for production performance
+- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime
-**So löst OmniRoute das Problem:**
+
-– Laufzeit-Heartbeat-Datei mit PID, Zeitstempeln, Transport, Werkzeuganzahl und Oszilloskopmodus
-- MCP-Status-API, die Heartbeat + aktuelle Aktivität kombiniert
-- UI-Statuskarten für Prozess-/Verfügbarkeits-/Heartbeat-Aktualität
+
+🤖 16. "I want to control model behavior globally"
-
-📋 20. „Ich benötige eine überprüfbare MCP-Tool-Ausführung“
+Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical.
-Wenn Tools die Konfiguration verändern oder operative Aktionen auslösen, benötigen Teams forensische Rückverfolgbarkeit.
+**How OmniRoute solves it:**
-**So löst OmniRoute das Problem:**
+- **System Prompt Injection** — Global prompt applied to all requests
+- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive)
+- **9 Routing Strategies** — Global strategies that determine how requests are distributed
+- **Wildcard Router** — `provider/*` patterns route dynamically to any provider
+- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard
+- **Provider Toggle** — Enable/disable all connections for a provider with one click
+- **Blocked Providers** — Exclude specific providers from `/v1/models` listing
-– SQLite-gestützte Audit-Protokollierung für MCP-Tool-Aufrufe
-- Filtert nach Tool, Erfolg/Misserfolg, API-Schlüssel und Paginierung
-- Dashboard-Audit-Tabelle + Statistik-Endpunkte für die Automatisierung
+
-
-🔐 21. „Ich benötige bereichsbezogene MCP-Berechtigungen pro Integration“
+
+🧰 17. "I need MCP tools as first-class product capabilities"
-Verschiedene Clients sollten Zugriff auf die Werkzeugkategorien mit den geringsten Rechten haben.
+Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer.
-**So löst OmniRoute das Problem:**
+**How OmniRoute solves it:**
-- 10 granulare MCP-Bereiche für kontrollierten Werkzeugzugriff
-- Geltungsbereichsdurchsetzung und Sichtbarkeit in der MCP-Management-Benutzeroberfläche
-- Sichere Standardhaltung für Betriebswerkzeuge
+- MCP appears in the dashboard navigation and endpoint protocol tab
+- Dedicated MCP management page with process, tools, scopes, and audit
+- Built-in quick-start for `omniroute --mcp` and client onboarding
-
-⚙️ 22. „Ich brauche Betriebskontrollen ohne Umschichtung“
+
-Teams benötigen bei Vorfällen oder Kostenereignissen schnelle Laufzeitänderungen.
+
+🧠 18. "I need A2A orchestration with sync + stream task paths"
-**So löst OmniRoute das Problem:**
+Agent workflows need both direct replies and long-running streamed execution with lifecycle control.
-- Schalten Sie die Combo-Aktivierung direkt über das MCP-Dashboard um
-- Wenden Sie Ausfallsicherheitsprofile aus vordefinierten Richtlinienpaketen an
-- Setzen Sie den Leistungsschalterstatus über dasselbe Bedienfeld zurück
+**How OmniRoute solves it:**
-
-🔄 23. „Ich benötige Live-Sichtbarkeit und Stornierung des A2A-Aufgabenlebenszyklus“
+- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream`
+- SSE streaming with terminal state propagation
+- Task lifecycle APIs for `tasks/get` and `tasks/cancel`
-Ohne Sichtbarkeit des Lebenszyklus wird es schwierig, Aufgabenvorfälle zu selektieren.
+
-**So löst OmniRoute das Problem:**
+
+🛰️ 19. "I need real MCP process health, not guessed status"
-- Aufgabenliste/Filterung nach Bundesland/Fähigkeit mit Paginierung
-- Drilldown zu Aufgabenmetadaten, Ereignissen und Artefakten
-- Endpunkt zum Abbrechen von Aufgaben und UI-Aktion mit Bestätigung
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable.
-
-🌊 24. „Ich benötige aktive Stream-Metriken für die A2A-Last“
+**How OmniRoute solves it:**
-Streaming-Workflows erfordern betriebliche Einblicke in Parallelität und Live-Verbindungen.
+- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode
+- MCP status API combining heartbeat + recent activity
+- UI status cards for process/uptime/heartbeat freshness
-**So löst OmniRoute das Problem:**
+
-- Aktive Stream-Zähler im A2A-Status integriert
-- Zeitstempel der letzten Aufgabe und Anzahl pro Status
-- A2A-Dashboard-Karten für die Echtzeit-Betriebsüberwachung
+
+📋 20. "I need auditable MCP tool execution"
-
-🪪 25. „Ich benötige eine standardmäßige Agentenerkennung für Kunden“
+When tools mutate config or trigger ops actions, teams need forensic traceability.
-Externe Kunden und Orchestratoren benötigen für das Onboarding maschinenlesbare Metadaten.
+**How OmniRoute solves it:**
-**So löst OmniRoute das Problem:**
+- SQLite-backed audit logging for MCP tool calls
+- Filters by tool, success/failure, API key, and pagination
+- Dashboard audit table + stats endpoints for automation
-– Agentenkarte unter „/.well-known/agent.json“ verfügbar gemacht
-- Fähigkeiten und Fertigkeiten werden in der Management-Benutzeroberfläche angezeigt
-– Die A2A-Status-API enthält Erkennungsmetadaten für die Automatisierung
+
-
-🧭 26. „Ich benötige Protokollauffindbarkeit in der Produkt-UX“
+
+🔐 21. "I need scoped MCP permissions per integration"
-Wenn Benutzer Protokolloberflächen nicht entdecken können, sinken Akzeptanz und Supportqualität.
+Different clients should have least-privilege access to tool categories.
-**So löst OmniRoute das Problem:**
+**How OmniRoute solves it:**
-- Konsolidierte Seite**Endpunkte**mit Registerkarten für Proxy-, MCP-, A2A- und API-Endpunkte
-- Inline-Dienststatusumschaltung (Online/Offline) für MCP und A2A
-- Links von der Übersicht zu speziellen Verwaltungsregisterkarten
+- 10 granular MCP scopes for controlled tool access
+- Scope enforcement and visibility in MCP management UI
+- Safe default posture for operational tooling
-
-🧪 27. „Ich benötige eine End-to-End-Protokollvalidierung mit echten Clients“
+
-Probetests reichen nicht aus, um die Protokollkompatibilität vor der Veröffentlichung zu überprüfen.
+
+⚙️ 22. "I need operational controls without redeploying"
-**So löst OmniRoute das Problem:**
+Teams need quick runtime changes during incidents or cost events.
-– E2E-Suite, die die App startet und echten MCP SDK-Client-Transport verwendet
-- A2A-Clienttests für Erkennungs-, Sende-, Stream-, Get- und Abbruchflüsse
-- Vergleichen Sie Behauptungen mit MCP-Audit- und A2A-Aufgaben-APIs
+**How OmniRoute solves it:**
-
-📡 28. „Ich brauche eine einheitliche Beobachtbarkeit über alle Schnittstellen hinweg“
+- Switch combo activation directly from MCP dashboard
+- Apply resilience profiles from pre-defined policy packs
+- Reset circuit breaker state from the same operations panel
-Die Aufteilung der Beobachtbarkeit nach Protokoll führt zu blinden Flecken und einer längeren MTTR.
+
-**So löst OmniRoute das Problem:**
+
+🔄 23. "I need live A2A task lifecycle visibility and cancellation"
-- Einheitliche Dashboards/Protokolle/Analysen in einem Produkt
-- Gesundheits-, Audit- und Anforderungstelemetrie über OpenAI-, MCP- und A2A-Ebenen hinweg
-- Operative APIs für Status und Automatisierung
+Without lifecycle visibility, task incidents become hard to triage.
-
-💼 29. „Ich benötige eine Laufzeit für Proxy + Tools + Agent-Orchestrierung“
+**How OmniRoute solves it:**
-Die Ausführung vieler separater Dienste erhöht die Betriebskosten und erhöht die Fehlerhäufigkeit.
+- Task listing/filtering by state/skill with pagination
+- Drill-down on task metadata, events, and artifacts
+- Task cancellation endpoint and UI action with confirmation
-**So löst OmniRoute das Problem:**
+
-- OpenAI-kompatibler Proxy, MCP-Server und A2A-Server in einem Stack
-– Gemeinsame Authentifizierung, Ausfallsicherheit, Datenspeicher und Beobachtbarkeit
-- Konsistentes Richtlinienmodell über alle Interaktionsoberflächen hinweg
+
+🌊 24. "I need active stream metrics for A2A load"
-
-🚀 30. „Ich muss Agenten-Workflows ohne Glue-Code-Wildwuchs ausliefern“
+Streaming workflows require operational insight into concurrency and live connections.
-Teams verlieren an Geschwindigkeit, wenn sie mehrere Ad-hoc-Dienste und -Skripte zusammenfügen.
+**How OmniRoute solves it:**
-**So löst OmniRoute das Problem:**
+- Active stream counters integrated into A2A status
+- Last task timestamp and per-state counts
+- A2A dashboard cards for real-time ops monitoring
-- Einheitliche Endpunktstrategie für Kunden und Agenten
-- Integrierte Protokollverwaltungs-Benutzeroberflächen und Rauchvalidierungspfade
-- Produktionsreife Grundlagen (Sicherheit, Protokollierung, Ausfallsicherheit, Backup)
+
+
+
+🪪 25. "I need standard agent discovery for clients"
+
+External clients and orchestrators need machine-readable metadata for onboarding.
+
+**How OmniRoute solves it:**
+
+- Agent Card exposed at `/.well-known/agent.json`
+- Capabilities and skills shown in management UI
+- A2A status API includes discovery metadata for automation
+
+
+
+
+🧭 26. "I need protocol discoverability in the product UX"
+
+If users cannot discover protocol surfaces, adoption and support quality drop.
+
+**How OmniRoute solves it:**
+
+- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints
+- Inline service status toggles (Online/Offline) for MCP and A2A
+- Links from overview to dedicated management tabs
+
+
+
+
+🧪 27. "I need end-to-end protocol validation with real clients"
+
+Mock tests are not enough to validate protocol compatibility before release.
+
+**How OmniRoute solves it:**
+
+- E2E suite that boots app and uses real MCP SDK client transport
+- A2A client tests for discovery, send, stream, get, and cancel flows
+- Cross-check assertions against MCP audit and A2A tasks APIs
+
+
+
+
+📡 28. "I need unified observability across all interfaces"
+
+Splitting observability by protocol creates blind spots and longer MTTR.
+
+**How OmniRoute solves it:**
+
+- Unified dashboards/logs/analytics in one product
+- Health + audit + request telemetry across OpenAI, MCP, and A2A layers
+- Operational APIs for status and automation
+
+
+
+
+💼 29. "I need one runtime for proxy + tools + agent orchestration"
+
+Running many separate services increases operational cost and failure modes.
+
+**How OmniRoute solves it:**
+
+- OpenAI-compatible proxy, MCP server, and A2A server in one stack
+- Shared auth, resilience, data store, and observability
+- Consistent policy model across all interaction surfaces
+
+
+
+
+🚀 30. "I need to ship agentic workflows without glue-code sprawl"
+
+Teams lose velocity when stitching multiple ad-hoc services and scripts.
+
+**How OmniRoute solves it:**
+
+- Unified endpoint strategy for clients and agents
+- Built-in protocol management UIs and smoke validation paths
+- Production-ready foundations (security, logging, resilience, backup)
+
+
### Example Playbooks (Integrated Use Cases)
-**Playbook A: Bezahltes Abonnement maximieren + günstiges Backup**```txt
+**Playbook A: Maximize paid subscription + cheap backup**
+
+```txt
Combo: "maximize-claude"
1. cc/claude-opus-4-6
2. glm/glm-4.7
@@ -607,21 +690,23 @@ Combo: "maximize-claude"
Monthly cost: $20 + small backup spend
Outcome: higher quality, near-zero interruption
-````
+```
-**Playbook B: Kostenfreier Codierungsstack**```txt
+**Playbook B: Zero-cost coding stack**
+
+```txt
Combo: "free-forever"
-
-1. gc/gemini-3-flash
-2. if/kimi-k2-thinking
-3. qw/qwen3-coder-plus
+ 1. gc/gemini-3-flash
+ 2. if/kimi-k2-thinking
+ 3. qw/qwen3-coder-plus
Monthly cost: $0
Outcome: stable free coding workflow
+```
-````
+**Playbook C: 24/7 always-on fallback chain**
-**Playbook C: 24/7 Always-On-Fallback-Kette**```txt
+```txt
Combo: "always-on"
1. cc/claude-opus-4-6
2. cx/gpt-5.2-codex
@@ -630,125 +715,134 @@ Combo: "always-on"
5. if/kimi-k2-thinking
Outcome: deep fallback depth for deadline-critical workloads
-````
+```
-**Playbook D: Agenteneinsätze mit MCP + A2A**```txt
+**Playbook D: Agent ops with MCP + A2A**
-1. Start MCP transport (`omniroute --mcp`) for tool-driven operations
-2. Run A2A tasks via `message/send` and `message/stream`
-3. Observe via /dashboard/endpoint (MCP and A2A tabs)
-4. Toggle services via inline status controls
-
-````
+```txt
+1) Start MCP transport (`omniroute --mcp`) for tool-driven operations
+2) Run A2A tasks via `message/send` and `message/stream`
+3) Observe via /dashboard/endpoint (MCP and A2A tabs)
+4) Toggle services via inline status controls
+```
---
## 🆓 Start Free — Zero Configuration Cost
-> Richten Sie die KI-Codierung in wenigen Minuten für**0 $/Monat**ein. Verbinden Sie diese kostenlosen Konten und nutzen Sie die integrierte**Free Stack**-Kombination.
+> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo.
-| Schritt | Aktion | Anbieter freigeschaltet |
-| ---- | ------------------------------------------------- | ----------------------------------------------------------------- |
-| 1 | Verbinden Sie**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 –**unbegrenzt**|
-| 2 | Verbinden Sie**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**unbegrenzt**|
-| 3 | Verbinden Sie**Qwen**(Gerätecode) | qwen3-coder-plus, qwen3-coder-flash... —**unbegrenzt**|
-| 4 | Verbinden Sie**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro –**180.000/Monat kostenlos**|
-| 5 | `/dashboard/combos` → Vorlage**Free Stack ($0)**| Round-Robin aller kostenlosen Anbieter automatisch |
+| Step | Action | Providers Unlocked |
+| ---- | -------------------------------------------------- | ------------------------------------------------------------------ |
+| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** |
+| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** |
+| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** |
+| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** |
+| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically |
-**Zeigen Sie eine beliebige IDE/CLI auf:**„http://localhost:20128/v1“ · API-Schlüssel: „any-string“ · Fertig.
+**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done.
->**Optionale zusätzliche Abdeckung (auch kostenlos):**Groq API-Schlüssel (30 U/min kostenlos), NVIDIA NIM (40 U/min kostenlos, 70+ Modelle), Cerebras (1 Mio. Token/Tag), LongCat API-Schlüssel (50 Mio. Token/Tag!), Cloudflare Workers AI (10.000 Neuronen/Tag, 50+ Modelle).## Schnellstart
+> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models).
+
+## Schnellstart
### 1) Install and run
```bash
npm install -g omniroute
omniroute
-````
+```
-> **pnpm-Benutzer:**Führen Sie nach der Installation „pnpm genehmigt-builds -g“ aus, um native Build-Skripte zu aktivieren, die für „better-sqlite3“ und „@swc/core“ erforderlich sind:
+> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`:
>
-> „Bash
+> ```bash
> pnpm install -g omniroute
-> pnpm genehmigt-builds -g # Alle Pakete auswählen → genehmigen
-> Omniroute
->
-> ```
->
+> pnpm approve-builds -g # Select all packages → approve
+> omniroute
> ```
-Das Dashboard wird unter „http://localhost:20128“ geöffnet und die API-Basis-URL ist „http://localhost:20128/v1“.
+Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`.
-| Befehl | Beschreibung |
-| ----------------------- | ------------------------------------------------------------------- |
-| `omniroute` | Server starten („PORT=20128“, API und Dashboard auf demselben Port) |
-| `omniroute --port 3000` | Setzen Sie den kanonischen/API-Port auf 3000 |
-| `omniroute --mcp` | Starten Sie den MCP-Server (STDIO-Transport) |
-| `omniroute --no-open` | Browser nicht automatisch öffnen |
-| `omniroute --help` | Hilfe anzeigen |
+| Command | Description |
+| ----------------------- | ----------------------------------------------------------- |
+| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) |
+| `omniroute --port 3000` | Set canonical/API port to 3000 |
+| `omniroute --mcp` | Start MCP server (stdio transport) |
+| `omniroute --no-open` | Don't auto-open browser |
+| `omniroute --help` | Show help |
-Optionaler Split-Port-Modus:```bash
+Optional split-port mode:
+
+```bash
PORT=20128 DASHBOARD_PORT=20129 omniroute
-
-# API: http://localhost:20128/v1
-
+# API: http://localhost:20128/v1
# Dashboard: http://localhost:20129
-
-````
+```
### Long-Running Streaming Timeouts
-Für die meisten Bereitstellungen benötigen Sie lediglich:
+For most deployments, you only need:
-| Variable | Standard | Zweck |
-| ------------------------ | -------------- | ---------------------------------------------------------------------------------------------- |
-| `REQUEST_TIMEOUT_MS` | „600000“ | Gemeinsame Baseline für Upstream-Abruf, versteckte Undici-Timeouts, TLS-Fingerprint-Anfragen und API-Bridge-Request/Proxy-Timeouts |
-| `STREAM_IDLE_TIMEOUT_MS` | erbt „REQUEST_TIMEOUT_MS“ | Maximale Lücke zwischen Streaming-Blöcken, bevor OmniRoute den SSE-Stream abbricht |
+| Variable | Default | Purpose |
+| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- |
+| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts |
+| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream |
-Die Abwärtskompatibilität bleibt erhalten: Vorhandene „FETCH_TIMEOUT_MS“, „API_BRIDGE_PROXY_TIMEOUT_MS“ und andere Timeout-Variablen pro Ebene funktionieren weiterhin und überschreiben die gemeinsame Baseline.
+Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline.
-Wenn Sie eine genauere Steuerung benötigen, stehen erweiterte Überschreibungen zur Verfügung:| Variable | Standard | Zweck |
-| ---------------------------------------- | ------------------------------------------ | ------------------------------------------------------------------- |
-| `FETCH_TIMEOUT_MS` | erbt „REQUEST_TIMEOUT_MS“ | Gesamtzeitüberschreitung der Upstream-Anforderung, die vom Hauptabrufsignal | verwendet wird
-| `FETCH_HEADERS_TIMEOUT_MS` | erbt „FETCH_TIMEOUT_MS“ | Undici-Zeitlimit für den Empfang von Upstream-Antwortheadern |
-| `FETCH_BODY_TIMEOUT_MS` | erbt „FETCH_TIMEOUT_MS“ | Undici-Zeitlimit zwischen Upstream-Body-Chunks („0“ deaktiviert es) |
-| `FETCH_CONNECT_TIMEOUT_MS` | „30000“ | Undici TCP-Verbindungszeitüberschreitung |
-| `FETCH_KEEPALIVE_TIMEOUT_MS` | „4000“ | Undici Leerlauf-Keep-Alive-Socket-Timeout |
-| `TLS_CLIENT_TIMEOUT_MS` | erbt „FETCH_TIMEOUT_MS“ | Zeitüberschreitung für TLS-Fingerabdruckanfragen über „wreq-js“ |
-| `API_BRIDGE_PROXY_TIMEOUT_MS` | erbt „REQUEST_TIMEOUT_MS“ oder „30000“ | Zeitüberschreitung für „/v1“-Proxy-Weiterleitung vom API-Port zum Dashboard-Port |
-| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Zeitüberschreitung bei eingehenden Anfragen auf dem API-Bridge-Server |
-| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | „60000“ | Zeitüberschreitung beim eingehenden Header auf dem API-Bridge-Server |
-| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | „5000“ | Keep-Alive-Timeout auf dem API-Bridge-Server |
-| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Zeitüberschreitung bei Socket-Inaktivität auf dem API-Bridge-Server („0“ deaktiviert ihn) |
+Advanced overrides are available if you need finer control:
-Wenn Sie OmniRoute hinter Nginx, Caddy, Cloudflare oder einem anderen Reverse-Proxy ausführen, stellen Sie sicher, dass der Proxy vorhanden ist
-Die Zeitüberschreitungen sind auch höher als die Zeitüberschreitungen für Ihren OmniRoute-Stream/Abruf.### 2) Connect providers and create your API key
+| Variable | Default | Purpose |
+| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- |
+| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal |
+| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers |
+| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) |
+| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout |
+| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout |
+| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` |
+| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port |
+| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server |
+| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server |
+| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server |
+| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) |
-1. Öffnen Sie Dashboard → „Anbieter“ und verbinden Sie mindestens einen Anbieter (OAuth oder API-Schlüssel).
-2. Öffnen Sie Dashboard → „Endpunkte“ und erstellen Sie einen API-Schlüssel.
-3. (Optional) Öffnen Sie Dashboard → „Combos“ und legen Sie Ihre Fallback-Kette fest.### 3) Point your coding tool to OmniRoute
+If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy
+timeouts are also higher than your OmniRoute stream/fetch timeouts.
+
+### 2) Connect providers and create your API key
+
+1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key).
+2. Open Dashboard → `Endpoints` and create an API key.
+3. (Optional) Open Dashboard → `Combos` and set your fallback chain.
+
+### 3) Point your coding tool to OmniRoute
```txt
Base URL: http://localhost:20128/v1
API Key: [copy from Endpoint page]
Model: if/kimi-k2-thinking (or any provider/model prefix)
-````
+```
-Funktioniert mit Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode und OpenAI-kompatiblen SDKs.### 4) Enable and validate protocols (v2.0)
+Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs.
-**MCP (für werkzeuggesteuerte Vorgänge):**```bash
+### 4) Enable and validate protocols (v2.0)
+
+**MCP (for tool-driven operations):**
+
+```bash
omniroute --mcp
+```
-````
-
-Verbinden Sie dann Ihren MCP-Client über „stdio“ und testen Sie Tools wie:
+Then connect your MCP client over `stdio` and test tools like:
- `omniroute_get_health`
- `omniroute_list_combos`
-**A2A (für Agent-zu-Agent-Workflows):**```bash
+**A2A (for agent-to-agent workflows):**
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
```bash
curl -X POST http://localhost:20128/a2a \
@@ -762,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \
npm run test:protocols:e2e
```
-Diese Suite validiert echte MCP- und A2A-Client-Flows anhand einer laufenden App.### Alternative: run from source
+This suite validates real MCP and A2A client flows against a running app.
+
+### Alternative: run from source
```bash
cp .env.example .env
@@ -770,13 +866,13 @@ npm install
PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev
```
-
-Void Linux (Vorlage „xbps-src“)
+
+Void Linux (`xbps-src` template)
-Für Void-Linux-Benutzer können Sie mit „xbps-src“ ein natives Paket erstellen. Speichern Sie diesen Block als „srcpkgs/omniroute/template“:```bash
+For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`:
+```bash
# Template file for 'omniroute'
-
pkgname=omniroute
version=3.4.1
revision=1
@@ -788,7 +884,7 @@ license="MIT"
homepage="https://github.com/diegosouzapw/OmniRoute"
distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz"
checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b
-system_accounts="\_omniroute"
+system_accounts="_omniroute"
omniroute_homedir="/var/lib/omniroute"
export NODE_ENV=production
export npm_config_engine_strict=false
@@ -796,71 +892,70 @@ export npm_config_loglevel=error
export npm_config_fund=false
export npm_config_audit=false
-do_build() { # Determine target CPU arch for node-gyp
-local \_gyp_arch
-case "$XBPS_TARGET_MACHINE" in
-aarch64*) \_gyp_arch=arm64 ;;
-armv7*|armv6*) \_gyp_arch=arm ;;
-i686*) \_gyp_arch=ia32 ;;
-\*) \_gyp_arch=x64 ;;
-esac
+do_build() {
+ # Determine target CPU arch for node-gyp
+ local _gyp_arch
+ case "$XBPS_TARGET_MACHINE" in
+ aarch64*) _gyp_arch=arm64 ;;
+ armv7*|armv6*) _gyp_arch=arm ;;
+ i686*) _gyp_arch=ia32 ;;
+ *) _gyp_arch=x64 ;;
+ esac
- # 1) Install all deps – skip scripts (no network in do_build, native modules
- # compiled separately below; better-sqlite3 is serverExternalPackage so
- # Next.js does not execute it during next build)
- NODE_ENV=development npm ci --ignore-scripts
+ # 1) Install all deps – skip scripts (no network in do_build, native modules
+ # compiled separately below; better-sqlite3 is serverExternalPackage so
+ # Next.js does not execute it during next build)
+ NODE_ENV=development npm ci --ignore-scripts
- # 2) Build the Next.js standalone bundle
- npm run build
+ # 2) Build the Next.js standalone bundle
+ npm run build
- # 3) Copy static assets into standalone
- cp -r .next/static .next/standalone/.next/static
- [ -d public ] && cp -r public .next/standalone/public || true
+ # 3) Copy static assets into standalone
+ cp -r .next/static .next/standalone/.next/static
+ [ -d public ] && cp -r public .next/standalone/public || true
- # 4) Compile better-sqlite3 native binding for the target architecture.
- # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
- # without npm altering them.
- local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
- (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
+ # 4) Compile better-sqlite3 native binding for the target architecture.
+ # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
+ # without npm altering them.
+ local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
+ (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
- # 5) Place the compiled binding into the standalone bundle
- local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
- mkdir -p "$_bs3_release"
- cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
+ # 5) Place the compiled binding into the standalone bundle
+ local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
+ mkdir -p "$_bs3_release"
+ cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
- # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
- # so sharp is not used at runtime; x64 .so files would break aarch64 strip
- rm -rf .next/standalone/node_modules/@img
-
- # 7) Copy pino runtime deps omitted by Next.js static analysis:
- # pino-abstract-transport – required by pino's worker thread
- # split2 – dep of pino-abstract-transport
- # process-warning – dep of pino itself
- for _mod in pino-abstract-transport split2 process-warning; do
- cp -r "node_modules/$_mod" .next/standalone/node_modules/
- done
+ # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
+ # so sharp is not used at runtime; x64 .so files would break aarch64 strip
+ rm -rf .next/standalone/node_modules/@img
+ # 7) Copy pino runtime deps omitted by Next.js static analysis:
+ # pino-abstract-transport – required by pino's worker thread
+ # split2 – dep of pino-abstract-transport
+ # process-warning – dep of pino itself
+ for _mod in pino-abstract-transport split2 process-warning; do
+ cp -r "node_modules/$_mod" .next/standalone/node_modules/
+ done
}
do_check() {
-npm run test:unit
+ npm run test:unit
}
do_install() {
-vmkdir usr/lib/omniroute/.next
+ vmkdir usr/lib/omniroute/.next
- vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
+ vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
- # Prevent removal of empty Next.js app router dirs by the post-install hook
- for _d in \
- .next/standalone/.next/server/app/dashboard \
- .next/standalone/.next/server/app/dashboard/settings \
- .next/standalone/.next/server/app/dashboard/providers; do
- touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
- done
-
- cat > "${WRKDIR}/omniroute" <<'EOF'
+ # Prevent removal of empty Next.js app router dirs by the post-install hook
+ for _d in \
+ .next/standalone/.next/server/app/dashboard \
+ .next/standalone/.next/server/app/dashboard/settings \
+ .next/standalone/.next/server/app/dashboard/providers; do
+ touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
+ done
+ cat > "${WRKDIR}/omniroute" <<'EOF'
#!/bin/sh
export PORT="${PORT:-20128}"
export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}"
@@ -872,10 +967,9 @@ EOF
}
post_install() {
-vlicense LICENSE
+ vlicense LICENSE
}
-
-````
+```
@@ -883,9 +977,11 @@ vlicense LICENSE
## 🐳 Docker
-OmniRoute ist als öffentliches Docker-Image auf [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute) verfügbar.
+OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
-**Schneller Lauf:**```bash
+**Quick run:**
+
+```bash
docker run -d \
--name omniroute \
--restart unless-stopped \
@@ -893,85 +989,96 @@ docker run -d \
-p 20128:20128 \
-v omniroute-data:/app/data \
diegosouzapw/omniroute:latest
-````
+```
-**Mit Umgebungsdatei:**```bash
+**With environment file:**
+```bash
# Copy and edit .env first
-
cp .env.example .env
docker run -d \
- --name omniroute \
- --restart unless-stopped \
- --stop-timeout 40 \
- --env-file .env \
- -p 20128:20128 \
- -v omniroute-data:/app/data \
- diegosouzapw/omniroute:latest
+ --name omniroute \
+ --restart unless-stopped \
+ --stop-timeout 40 \
+ --env-file .env \
+ -p 20128:20128 \
+ -v omniroute-data:/app/data \
+ diegosouzapw/omniroute:latest
+```
-````
+**Using Docker Compose:**
-**Verwendung von Docker Compose:**```bash
+```bash
# Base profile (no CLI tools)
docker compose --profile base up -d
# CLI profile (Claude Code, Codex, OpenClaw built-in)
docker compose --profile cli up -d
-````
+```
-Die Dashboard-Unterstützung für Docker-Bereitstellungen umfasst jetzt einen**Cloudflare Quick Tunnel**mit einem Klick unter „Dashboard → Endpunkte“. Die erste Aktivierung lädt „cloudflared“ nur bei Bedarf herunter, startet einen temporären Tunnel zu Ihrem aktuellen „/v1“-Endpunkt und zeigt die generierte „https://\*.trycloudflare.com/v1“-URL direkt unter Ihrer normalen öffentlichen URL an.
+Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL.
-Hinweise:
+Notes:
-- Quick Tunnel-URLs sind temporär und ändern sich nach jedem Neustart.
- – Quick Tunnels werden nach einem OmniRoute- oder Container-Neustart nicht automatisch wiederhergestellt. Aktivieren Sie sie bei Bedarf über das Dashboard erneut.
- – Die verwaltete Installation unterstützt derzeit Linux, macOS und Windows auf „x64“ / „arm64“.
- – Managed Quick Tunnels verwenden standardmäßig den HTTP/2-Transport, um laute QUIC-UDP-Pufferwarnungen in eingeschränkten Containerumgebungen zu vermeiden. Stellen Sie „CLOUDFLARED_PROTOCOL=quic“ oder „auto“ ein, wenn Sie einen anderen Transport wünschen.
-- Docker-Images bündeln System-CA-Roots und übergeben sie an verwaltetes „Cloudflared“, wodurch TLS-Vertrauensfehler vermieden werden, wenn der Tunnel innerhalb des Containers bootet.
-- SQLite läuft im WAL-Modus. „Docker Stop“ sollte abgeschlossen werden dürfen, damit OmniRoute die neuesten Änderungen zurück in „storage.sqlite“ überprüfen kann.
- – Die gebündelten Compose-Dateien legen bereits eine Stoppfrist von 40 Sekunden fest. Wenn Sie das Image direkt ausführen, behalten Sie „--stop-timeout 40“ (oder ähnlich) bei, damit manuelle Stopps die Bereinigung beim Herunterfahren nicht unterbrechen.
-- Legen Sie „CLOUDFLARED_BIN=/absolute/path/to/cloudflared“ fest, wenn OmniRoute eine vorhandene Binärdatei verwenden soll, anstatt eine herunterzuladen.
+- Quick Tunnel URLs are temporary and change after every restart.
+- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed.
+- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`.
+- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport.
+- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container.
+- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`.
+- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup.
+- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one.
-**Verwendung von Docker Compose mit Caddy (HTTPS Auto-TLS):**
+**Using Docker Compose with Caddy (HTTPS Auto-TLS):**
-OmniRoute kann mithilfe der automatischen SSL-Bereitstellung von Caddy sicher verfügbar gemacht werden. Stellen Sie sicher, dass der DNS-A-Eintrag Ihrer Domain auf die IP Ihres Servers verweist.```yaml
+OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP.
+
+```yaml
services:
-omniroute:
-image: diegosouzapw/omniroute:latest
-container_name: omniroute
-restart: unless-stopped
-volumes: - omniroute-data:/app/data
-environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com
+ omniroute:
+ image: diegosouzapw/omniroute:latest
+ container_name: omniroute
+ restart: unless-stopped
+ volumes:
+ - omniroute-data:/app/data
+ environment:
+ - PORT=20128
+ - NEXT_PUBLIC_BASE_URL=https://your-domain.com
-caddy:
-image: caddy:latest
-container_name: caddy
-restart: unless-stopped
-ports: - "80:80" - "443:443"
-command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
+ caddy:
+ image: caddy:latest
+ container_name: caddy
+ restart: unless-stopped
+ ports:
+ - "80:80"
+ - "443:443"
+ command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
volumes:
-omniroute-data:
+ omniroute-data:
+```
-````
-
-| Bild | Tag | Größe | Beschreibung |
+| Image | Tag | Size | Description |
| ------------------------ | -------- | ------ | --------------------- |
-| `diegosouzapw/omniroute` | `neueste` | ~250 MB | Neueste stabile Version |
-| `diegosouzapw/omniroute` | `1.0.3` | ~250 MB | Aktuelle Version |---
+| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release |
+| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version |
+
+---
## 🖥️ Desktop App — Offline & Always-On
-> 🆕**NEU!**OmniRoute ist jetzt als**native Desktop-Anwendung**für Windows, macOS und Linux verfügbar.
+> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux.
-Führen Sie OmniRoute als eigenständige Desktop-App aus – kein Terminal, kein Browser, keine Internetverbindung für lokale Modelle erforderlich. Die Electron-basierte App umfasst:
+Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes:
-- 🖥️**Natives Fenster**– Spezielles App-Fenster mit Integration in die Taskleiste
-- 🔄**Auto-Start**– OmniRoute bei der Systemanmeldung starten
-- 🔔**Native Benachrichtigungen**– Erhalten Sie Benachrichtigungen bei Kontingentausschöpfung oder Anbieterproblemen
-- ⚡**One-Click-Installation**– NSIS (Windows), DMG (macOS), AppImage (Linux)
-- 🌐**Offline-Modus**– Funktioniert vollständig offline mit dem gebündelten Server### Schnellstart
+- 🖥️ **Native Window** — Dedicated app window with system tray integration
+- 🔄 **Auto-Start** — Launch OmniRoute on system login
+- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues
+- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux)
+- 🌐 **Offline Mode** — Works fully offline with bundled server
+
+### Schnellstart
```bash
# Development mode
@@ -982,308 +1089,370 @@ npm run electron:build # Current platform
npm run electron:build:win # Windows (.exe)
npm run electron:build:mac # macOS (.dmg) — x64 & arm64
npm run electron:build:linux # Linux (.AppImage)
-````
+```
### System Tray
-Wenn OmniRoute minimiert ist, befindet es sich mit schnellen Aktionen in Ihrer Taskleiste:
+When minimized, OmniRoute lives in your system tray with quick actions:
-- Dashboard öffnen
-- Server-Port ändern
-- Anwendung beenden
+- Open dashboard
+- Change server port
+- Quit application
-📖 Vollständige Dokumentation: [`electron/README.md`](electron/README.md)---
+📖 Full documentation: [`electron/README.md`](electron/README.md)
+
+---
## 💰 Pricing at a Glance
-| Stufe | Anbieter | Kosten | Kontingent zurücksetzen | Am besten für |
-| -------------------- | --------------------------- | ---------------------------------------- | ------------------------- | ------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| **💳 ABO** | Claude Code (Pro) | 20 $/Monat | 5h + wöchentlich | Bereits abonniert |
-| | Codex (Plus/Pro) | 20–200 $/Monat | 5h + wöchentlich | OpenAI-Benutzer |
-| | Gemini CLI | **KOSTENLOS** | 180.000/Monat + 1.000/Tag | Alle! |
-| | GitHub-Copilot | 10–19 $/Monat | Monatlich | GitHub-Benutzer |
-| **🔑 API-SCHLÜSSEL** | NVIDIA NIM | **KOSTENLOS**(für immer entwickeln) | ~40 U/min | Über 70 offene Modelle |
-| | Großhirn | **KOSTENLOS**(1 Mio. tok/Tag) | 60.000 TPM / 30 U/min | Der schnellste der Welt |
-| | Groq | **KOSTENLOS**(30 U/min) | 14,4K RPD | Ultraschnelles Lama/Gemma |
-| | DeepSeek V3.2 | 0,27 $/1,10 $ pro 1 Mio. | Keine | Bestes Preis-Leistungs-Verhältnis |
-| | xAI Grok-4 Schnell | **0,20 $/0,50 $ pro 1 Mio.**🆕 | Keine | Schnellster + Werkzeugaufruf, ultraniedrig |
-| | xAI Grok-4 (Standard) | 0,20 $/1,50 $ pro 1 Mio. 🆕 | Keine | Argumentations-Flaggschiff von xAI |
-| | Mistral | Kostenlose Testversion + kostenpflichtig | Tarif begrenzt | Europäische KI |
-| | OpenRouter | Pay-per-Use | Keine | Über 100 Modelle aggr. |
-| **💰 GÜNSTIG** | GLM-5 (über Z.AI) 🆕 | 0,5 $/1 Mio. | Täglich 10 Uhr | 128K-Ausgabe, neuestes Flaggschiff |
-| | GLM-4.7 | 0,6 $/1 Mio. | Täglich 10 Uhr | Budgetsicherung |
-| | MiniMax M2.5 🆕 | 0,3 $/1 Mio. Eingabe | 5-Stunden-Rollen | Argumentation + Agentenaufgaben |
-| | MiniMax M2.1 | 0,2 $/1 Mio. | 5-Stunden-Rollen | Günstigste Option |
-| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-Use | Keine | Direkter Zugriff auf die Moonshot-API |
-| | Kimi K2 | $9/Monat pauschal | 10 Millionen Token/Monat | Vorhersehbare Kosten |
-| **🆓 KOSTENLOS** | Qoder | **$0** | Unbegrenzt | 5 Modelle unbegrenzt |
-| | Qwen | **$0** | Unbegrenzt | 4 Modelle unbegrenzt |
-| | Kiro | **$0** | Unbegrenzt | Claude Sonnet/Haiku (AWS Builder) |
-| | LongCat Flash-Lite 🆕 | **$0**(50 Mio. Token/Tag 🔥) | 1 RPS | Größte kostenlose Quote der Welt |
-| | Bestäubungs-KI 🆕 | **$0**(kein Schlüssel erforderlich) | 1 Anforderung/15s | GPT-5, Claude, DeepSeek, Lama 4 |
-| | Cloudflare Workers AI 🆕 | **$0**(10.000 Neuronen/Tag) | ~150 resp/Tag | Über 50 Modelle, globaler Vorsprung |
-| | Scaleway AI 🆕 | **0 $**(insgesamt 1 Mio. Token) | Tarif begrenzt | EU/DSGVO, Qwen3 235B, Lama 70B | > 🆕**Neue Modelle hinzugefügt (März 2026):**Grok-4 Fast-Familie für 0,20 $/0,50 $/M (Benchmark bei 1143 ms – 30 % schneller als Gemini 2.5 Flash), GLM-5 über Z.AI mit 128K-Ausgabe, MiniMax M2.5-Argumentation, aktualisierte Preise für DeepSeek V3.2, Kimi K2.5 über die direkte Moonshot-API. |
+| Tier | Provider | Cost | Quota Reset | Best For |
+| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- |
+| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed |
+| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users |
+| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! |
+| | GitHub Copilot | $10-19/mo | Monthly | GitHub users |
+| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models |
+| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest |
+| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma |
+| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning |
+| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow |
+| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI |
+| | Mistral | Free trial + paid | Rate limited | European AI |
+| | OpenRouter | Pay-per-use | None | 100+ models aggr. |
+| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship |
+| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup |
+| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks |
+| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option |
+| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access |
+| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost |
+| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited |
+| | Qwen | **$0** | Unlimited | 4 models unlimited |
+| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) |
+| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth |
+| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 |
+| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge |
+| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B |
-**💡 0 $ Combo Stack – Das komplette kostenlose Setup:**```
+> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API.
+**💡 $0 Combo Stack — The Complete Free Setup:**
+
+```
# 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever
+Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
+Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
+Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
+Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
+NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+```
-Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
-Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
-LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
-Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
-Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
-Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
-Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
-Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
-Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
-NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
-Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever.
-````
-
-**Kostenlos. Hört nie auf zu programmieren.**Konfigurieren Sie dies als eine OmniRoute-Kombination und alle Fallbacks erfolgen automatisch – kein manuelles Umschalten.---
+---
---
## 🆓 Free Models — What You Actually Get
-> Alle unten aufgeführten Modelle sind**100 % kostenlos, keine Kreditkarte erforderlich**. OmniRoute leitet automatisch zwischen ihnen weiter, wenn ein Kontingent aufgebraucht ist – kombinieren Sie sie alle für eine unzerstörbare 0-Dollar-Kombination.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
+> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo.
-| Modell | Präfix | Grenze | Ratenbegrenzung |
+### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------- | ------ | ------------- | --------------------- |
-| `claude-sonett-4.5` | `kr/` |**Unbegrenzt**| Keine gemeldete Tagesobergrenze |
-| `claude-haiku-4.5` | `kr/` |**Unbegrenzt**| Keine gemeldete Tagesobergrenze |
-| `claude-opus-4.6` | `kr/` |**Unbegrenzt**| Neuestes Werk von Kiro |### 🟢 QODER MODELS (Free PAT via qodercli)
+| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro |
-| Modell | Präfix | Grenze | Ratenbegrenzung |
-| ------------------- | ------ | ------------- | --------------- |
-| `kimi-k2-thinking` | `if/` |**Unbegrenzt**| Keine gemeldete Obergrenze |
-| `qwen3-coder-plus` | `if/` |**Unbegrenzt**| Keine gemeldete Obergrenze |
-| `deepseek-r1` | `if/` |**Unbegrenzt**| Keine gemeldete Obergrenze |
-| `minimax-m2.1` | `if/` |**Unbegrenzt**| Keine gemeldete Obergrenze |
-| `kimi-k2` | `if/` |**Unbegrenzt**| Keine gemeldete Obergrenze |
+### 🟢 QODER MODELS (Free PAT via qodercli)
-> Empfohlene Verbindungsmethode:**Persönliches Zugriffstoken + „qodercli“**. Browser OAuth ist
-> experimentell und standardmäßig deaktiviert, es sei denn, die Umgebungsvariablen „QODER_OAUTH_*“ sind konfiguriert.### 🟡 QWEN MODELS (Device Code Auth)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------ | ------ | ------------- | --------------- |
+| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap |
+| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap |
+| `deepseek-r1` | `if/` | **Unlimited** | No reported cap |
+| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap |
+| `kimi-k2` | `if/` | **Unlimited** | No reported cap |
-| Modell | Präfix | Grenze | Ratenlimit |
+> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is
+> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured.
+
+### 🟡 QWEN MODELS (Device Code Auth)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------- | ------ | ------------- | ------------------- |
-| `qwen3-coder-plus` | `qw/` |**Unbegrenzt**| Keine gemeldete Obergrenze |
-| `qwen3-coder-flash` | `qw/` |**Unbegrenzt**| Keine gemeldete Obergrenze |
-| `qwen3-coder-next` | `qw/` |**Unbegrenzt**| Keine gemeldete Obergrenze |
-| „Vision-Modell“ | `qw/` |**Unbegrenzt**| Multimodal (Bilder) |### 🟣 GEMINI CLI (Google OAuth)
+| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap |
+| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) |
-| Modell | Präfix | Grenze | Ratenlimit |
-| ------------------------ | ------ | ------------ | ------------- |
-| `gemini-3-flash-preview` | `gc/` |**180.000 Token/Monat**+ 1.000/Tag | Monatlicher Reset |
-| `gemini-2.5-pro` | `gc/` | 180.000/Monat (gemeinsamer Pool) | Hohe Qualität |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
+### 🟣 GEMINI CLI (Google OAuth)
-| Stufe | Tageslimit | Ratenlimit | Notizen |
-| ---------- | ------------ | ----------- | ----------------------------------------------------- |
-| Kostenlos (Entwickler) | Keine Token-Obergrenze |**~40 U/min**| Über 70 Modelle; Übergang zu reinen Tarifbegrenzungen Mitte 2025 |
+| Model | Prefix | Limit | Rate Limit |
+| ------------------------ | ------ | --------------------------- | ------------- |
+| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset |
+| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality |
-Beliebte kostenlose Modelle: „moonshotai/kimi-k2.5“ (Kimi K2.5), „z-ai/glm4.7“ (GLM 4.7), „deepseek-ai/deepseek-v3.2“ (DeepSeek V3.2), „nvidia/llama-3.3-70b-instruct“, „deepseek/deepseek-r1“.### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
+### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
-| Stufe | Tageslimit | Ratenlimit | Notizen |
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---------- | ------------ | ----------- | ------------------------------------------------------ |
+| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 |
+
+Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`
+
+### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
+
+| Tier | Daily Limit | Rate Limit | Notes |
| ---- | ----------------- | ---------------- | ------------------------------------------- |
-| Kostenlos |**1 Mio. Token/Tag**| 60.000 TPM / 30 U/min | Weltweit schnellste LLM-Inferenz; wird täglich zurückgesetzt |
+| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily |
-Kostenlos erhältlich: „llama-3.3-70b“, „llama-3.1-8b“, „deepseek-r1-distill-llama-70b“.### 🔴 GROQ (Free API Key — console.groq.com)
+Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`
-| Stufe | Tageslimit | Ratenlimit | Notizen |
+### 🔴 GROQ (Free API Key — console.groq.com)
+
+| Tier | Daily Limit | Rate Limit | Notes |
| ---- | ------------- | ---------------- | ----------------------------------------- |
-| Kostenlos |**14,4K RPD**| 30 U/min pro Modell | Keine Kreditkarte; 429 auf Limit, nicht berechnet |
+| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged |
-Kostenlos erhältlich: „llama-3.3-70b-versatile“, „gemma2-9b-it“, „mixtral-8x7b“, „whisper-large-v3“.### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`
-| Modell | Präfix | Tägliches kostenloses Kontingent | Notizen |
-| -------------- | ------ | ----------------- | --------- |
-| `LongCat-Flash-Lite` | `lc/` |**50 Millionen Token**💥 | Größtes kostenloses Kontingent aller Zeiten |
-| `LongCat-Flash-Chat` | `lc/` | 500.000 Token | Multi-Turn-Chat |
-| „LongCat-Flash-Thinking“ | `lc/` | 500.000 Token | Begründung / CoT |
-| `LongCat-Flash-Thinking-2601` | `lc/` | 500.000 Token | Version Januar 2026 |
-| „LongCat-Flash-Omni-2603“ | `lc/` | 500.000 Token | Multimodal |
+### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
-> 100 % kostenlos während der öffentlichen Beta. Melden Sie sich per E-Mail oder Telefon bei [longcat.chat](https://longcat.chat) an. Wird täglich um 00:00 UTC zurückgesetzt.### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+| Model | Prefix | Daily Free Quota | Notes |
+| ----------------------------- | ------ | ----------------- | ----------------------- |
+| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever |
+| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat |
+| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT |
+| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version |
+| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal |
-| Modell | Präfix | Ratenlimit | Anbieter dahinter |
-| ---------- | ------ | ---------- | ------------------- |
-| `openai` | `pol/` | 1 Anforderung/15s | GPT-5 |
-| `Claude` | `pol/` | 1 Anforderung/15s | Anthropischer Claude |
-| „Zwillinge“ | `pol/` | 1 Anforderung/15s | Google Gemini |
-| `deepseek` | `pol/` | 1 Anforderung/15s | DeepSeek V3 |
-| `Lama` | `pol/` | 1 Anforderung/15s | Meta Lama 4 Scout |
-| „Mistral“ | `pol/` | 1 Anforderung/15s | Mistral KI |
+> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC.
-> ✨**Keine Reibung:**Keine Anmeldung, kein API-Schlüssel. Fügen Sie den Bestäubungsanbieter mit einem leeren Schlüsselfeld hinzu und es funktioniert sofort.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+### 🟢 POLLINATIONS AI (No API Key Required) 🆕
-| Stufe | Tägliche Neuronen | Äquivalente Verwendung | Notizen |
-| ---- | ------------- | --------------------------------------- | --------- |
-| Kostenlos |**10.000**| ~150 LLM bzw. 500 Sek. Audio / 15.000 Einbettungen | Global Edge, 50+ Modelle |
+| Model | Prefix | Rate Limit | Provider Behind |
+| ---------- | ------ | ---------- | ------------------ |
+| `openai` | `pol/` | 1 req/15s | GPT-5 |
+| `claude` | `pol/` | 1 req/15s | Anthropic Claude |
+| `gemini` | `pol/` | 1 req/15s | Google Gemini |
+| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 |
+| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout |
+| `mistral` | `pol/` | 1 req/15s | Mistral AI |
-Beliebte kostenlose Modelle: „@cf/meta/llama-3.3-70b-instruct“, „@cf/google/gemma-3-12b-it“, „@cf/openai/whisper-large-v3-turbo“ (kostenloses Audio!), „@cf/qwen/qwen2.5-coder-15b-instruct“.
+> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately.
-> Erfordert API-Token + Konto-ID von [dash.cloudflare.com](https://dash.cloudflare.com). Konto-ID in den Anbietereinstellungen hinterlegen.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
-| Stufe | Kostenloses Kontingent | Standort | Notizen |
+| Tier | Daily Neurons | Equivalent Usage | Notes |
+| ---- | ------------- | --------------------------------------- | ----------------------- |
+| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models |
+
+Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct`
+
+> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings.
+
+### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+
+| Tier | Free Quota | Location | Notes |
| ---- | ------------- | ------------ | ----------------------------------- |
-| Kostenlos |**1 Mio. Token**| 🇫🇷 Paris, EU | Innerhalb der Grenzen ist keine Kreditkarte erforderlich |
+| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits |
-Kostenlos verfügbar: „qwen3-235b-a22b-instruct-2507“ (Qwen3 235B!), „llama-3.1-70b-instruct“, „mistral-small-3.2-24b-instruct-2506“, „deepseek-v3-0324“.
+Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
-> EU/DSGVO-konform. Holen Sie sich den API-Schlüssel unter [console.scaleway.com](https://console.scaleway.com).
+> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com).
->**💡 Der ultimative kostenlose Stack (11 Anbieter, 0 $ für immer):**
+> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):**
>
> ```
-> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
-> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
-> LongCat Lite (lc/) → LongCat-Flash-Lite – 50 Millionen Token/Tag 🔥
-> Bestäubungen (pol/) → GPT-5, Claude, DeepSeek, Llama 4 – kein Schlüssel erforderlich
-> Qwen (qw/) → qwen3-Coder-Modelle UNBEGRENZT
-> Gemini (gemini/) → Gemini 2.5 Flash – 1.500 Req/Tag kostenlos
-> Cloudflare AI (cf/) → 50+ Modelle – 10.000 Neuronen/Tag
-> Scaleway (scw/) → Qwen3 235B, Llama 70B – 1 Mio. kostenlose Token (EU)
-> Groq (groq/) → Lama/Gemma – 14,4K req/Tag ultraschnell
-> NVIDIA NIM (nvidia/) → 70+ offene Modelle – 40 U/min für immer
-> Großhirn (Großhirn) → Lama/Qwen weltweit am schnellsten – 1 Mio. tok/Tag
-> ```## 🎙️ Free Transcription Combo
+> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+> Qwen (qw/) → qwen3-coder models UNLIMITED
+> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free
+> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day
+> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast
+> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+> ```
-> Transkribieren Sie jedes Audio/Video für**0 $**– Deepgram führt mit 200 $ kostenlos, AssemblyAI 50 $ Fallback, Groq Whisper als unbegrenztes Notfall-Backup.
+## 🎙️ Free Transcription Combo
-| Anbieter | Kostenlose Credits | Bestes Modell | Ratenlimit |
-| ----------------- | ---------------------- | -------------------------------------------- | ------------- |
-| 🟢**Deepgram**|**200 $ gratis**(Anmeldung) | „nova-3“ – beste Genauigkeit, über 30 Sprachen | Kein RPM-Limit für kostenlose Credits |
-| 🔵**AssemblyAI**|**50 $ gratis**(Anmeldung) | „universal-3-pro“ – Kapitel, Stimmung, PII | Kein RPM-Limit für kostenlose Credits |
-| 🔴**Groq**|**Für immer kostenlos**| „whisper-large-v3“ – OpenAI Whisper | 30 U/min (Geschwindigkeit begrenzt) |
+> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup.
-**Vorgeschlagene Kombination in „/dashboard/combos“:**```
+| Provider | Free Credits | Best Model | Rate Limit |
+| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- |
+| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits |
+| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits |
+| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) |
+
+**Suggested combo in `/dashboard/combos`:**
+
+```
Name: free-transcription
Strategy: Priority
Nodes:
[1] deepgram/nova-3 → uses $200 free first
[2] assemblyai/universal-3-pro → fallback when Deepgram credits run out
[3] groq/whisper-large-v3 → free forever, emergency fallback
-````
+```
-Dann unter „/dashboard/media“ → Registerkarte „Transkription“: Laden Sie eine beliebige Audio- oder Videodatei hoch → wählen Sie Ihren Kombinationsendpunkt aus → erhalten Sie Transkriptionen in unterstützten Formaten.## 💡 Key Features
+Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats.
-OmniRoute v2.0 ist als Betriebsplattform konzipiert und nicht nur als Relay-Proxy.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
+## 💡 Key Features
-| Funktion | Was es tut |
-| ------------------------------------- | --------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ |
-| ⚡**Grok-4 Fast Family** | xAI-Modelle für 0,20 $/0,50 $/M – im Benchmarking 1143 ms (30 % schneller als Gemini 2.5 Flash) |
-| 🧠**GLM-5 über Z.AI** | 128K-Ausgabekontext, 0,5 $/1 Mio. – neuestes Flaggschiff der GLM-Familie |
-| 🔮**MiniMax M2.5** | Argumentation + Agentenaufgaben für 0,30 $/1 Mio. – deutliche Verbesserung gegenüber M2.1 |
-| 🎯**toolCalling Flag pro Modell** | Pro Modell „toolCalling: true/false“ in der Registrierung – AutoCombo überspringt nicht-toolfähige Modelle |
-| 🌍**Mehrsprachige Absichtserkennung** | PT/ZH/ES/AR-Schlüsselwörter in der AutoCombo-Bewertung – bessere Modellauswahl für nicht-englische Inhalte |
-| 📊**Benchmark-gesteuerte Fallbacks** | Echte p95-Latenz aus der Kombinationsbewertung von Live-Anfrage-Feeds – AutoCombo lernt aus tatsächlichen Daten |
-| 🔁**Deduplizierung anfordern** | Content-Hash-basiertes Dedup-Fenster – Multi-Agent-sicher, verhindert doppelte Gebühren |
-| 🔌**Pluggable RouterStrategy** | Erweiterbare „RouterStrategy“-Schnittstelle – benutzerdefinierte Routing-Logik als Plugins hinzufügen | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP |
+OmniRoute v3.5 is built as an operational platform, not just a relay proxy.
-| Funktion | Was es tut |
-| ---------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |
-| 🎮**Modellspielplatz** | Dashboard-Seite zum direkten Testen jedes Modells – Anbieter-/Modell-/Endpunkt-Selektoren, Monaco-Editor, Streaming, Abbruch, Timing |
-| 🔏**CLI-Fingerabdruckabgleich** | Header-/Body-Reihenfolge pro Anbieter, um mit nativen CLI-Signaturen übereinzustimmen – schalten Sie pro Anbieter unter „Einstellungen“ > „Sicherheit“ um.**Ihre Proxy-IP bleibt erhalten** |
-| 🤝**ACP-Unterstützung (Agent Client Protocol)** | CLI-Agent-Erkennung (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 weitere), Prozess-Spawner, „/api/acp/agents“-Endpunkt |
-| 🤖**ACP-Agenten-Dashboard** | Debuggen › Seite „Agenten“ – Raster mit 14 Agenten mit Installationsstatus, Version und benutzerdefiniertem Agentenformular für jedes CLI-Tool.**OpenCode**-Benutzer erhalten eine Schaltfläche „Opencode.json herunterladen“, die automatisch eine gebrauchsfertige Konfiguration mit allen verfügbaren Modellen generiert. |
-| 🔧**Benutzerdefiniertes Modell „apiFormat“-Routing** | Benutzerdefinierte Modelle mit „apiFormat: „responses““ werden jetzt korrekt an den Responses-API-Übersetzer weitergeleitet |
-| 🏢**Codex Workspace Isolation** | Mehrere Codex-Arbeitsbereiche pro E-Mail – OAuth trennt Verbindungen korrekt nach Arbeitsbereichs-ID |
-| 🔄**Electron Auto-Update** | Desktop-App sucht nach Updates + automatische Installation beim Neustart | ### 🤖 Agent & Protocol Operations (v2.0) |
+### 🆕 New — v3.5.5 Highlights (Apr 2026)
-| Funktion | Was es tut |
-| ------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- |
-| 🔧**MCP-Server (25 Tools)** | IDE/Agent-Tools über 3 Transporte: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 Kerne + 3 Speicher + 4 Fertigkeitswerkzeuge |
-| 🤝**A2A-Server (JSON-RPC + SSE)** | Ausführung von Agent-zu-Agent-Aufgaben mit Synchronisierungs- und Streaming-Flows |
-| 🧭**Consolidated Endpoints-Seite** | Verwaltungsseite mit Registerkarten mit den Registerkarten „Endpunkt-Proxy“, „MCP“, „A2A“ und „API-Endpunkte“ |
-| 🎚️**Service-Aktivierung/Deaktivierung** | EIN/AUS-Schalter für MCP und A2A mit Einstellungspersistenz (Standard: AUS) |
-| 🛰️**MCP Runtime Heartbeat** | Echter Prozessstatus (PID, Betriebszeit, Heartbeat-Alter, Transport, Scope-Modus) |
-| 📋**MCP Audit Trail** | Filterbare Audit-Protokolle mit Erfolg/Misserfolg und Schlüsselzuordnung |
-| 🔐**Durchsetzung des MCP-Geltungsbereichs** | 10 granulare Umfangsberechtigungen für kontrollierten Werkzeugzugriff |
-| 📡**A2A Task Lifecycle Management** | Aufgaben auflisten/filtern, Ereignisse/Artefakte prüfen, laufende Aufgaben abbrechen |
-| 📋**Agentenkartenerkennung** | `/.well-known/agent.json` für die automatische Client-Erkennung |
-| 🧪**Protokoll-E2E-Testkabel** | Echtes MCP SDK + A2A-Client fließt in „test:protocols:e2e“ |
-| ⚙️**Betriebskontrollen** | Schaltkombination, Anwenden von Resilienzprofilen, Zurücksetzen von Leistungsschaltern über eine Bedienoberfläche | ### 🧠 Routing & Intelligence |
+| Feature | What It Does |
+| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation |
+| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs |
+| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions |
+| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection |
+| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts |
-| Funktion | Was es tut |
-| --------------------------------------- | ---------------------------------------------------------------------------------------- | ----------------------- |
-| 🎯**Intelligenter 4-Stufen-Fallback** | Automatische Route: Abonnement → API-Schlüssel → Günstig → Kostenlos |
-| 📊**Kontingentverfolgung in Echtzeit** | Live-Token-Zählung + Reset-Countdown pro Anbieter |
-| 🔄**Formatübersetzung** | OpenAI ↔ Claude ↔ Gemini ↔ Antworten mit schemasicheren Konvertierungen |
-| 👥**Unterstützung mehrerer Konten** | Mehrere Konten pro Anbieter mit intelligenter Auswahl |
-| 🔄**Automatische Token-Aktualisierung** | OAuth-Token werden bei Wiederholung automatisch aktualisiert |
-| 🎨**Benutzerdefinierte Kombinationen** | 9 Ausgleichsstrategien + Fallback-Kettenkontrolle |
-| 🌐**Wildcard-Router** | `provider/*` dynamisches Routing |
-| 🧠**Budgetkontrollen denken** | Passthrough-, automatische, benutzerdefinierte und adaptive Reasoning-Grenzwerte |
-| 🔀**Modell-Aliase** | Integrierte + benutzerdefinierte Modell-Aliasing- und Migrationssicherheit |
-| ⚡**Hintergrundverschlechterung** | Hintergrundaufgaben mit niedriger Priorität an günstigere Modelle weiterleiten |
-| 🧪**Aufgabenbewusstes Smart Routing** | Modell automatisch nach Inhaltstyp auswählen (Codierung/Vision/Analyse/Zusammenfassung) |
-| 🔄**A2A-Agent-Workflows** | Deterministischer FSM-Orchestrator für zustandsbehaftete mehrstufige Agentenausführungen |
-| 🔀**Adaptives Routing** | Dynamische Strategieüberschreibung basierend auf Token-Volumen und Prompt-Komplexität |
-| 🎲**Anbietervielfalt** | Shannon-Entropiebewertung, die die Verteilung des Auto-Combo-Verkehrs ausgleicht |
-| 💬**System-Prompt-Injektion** | Globale Verhaltenskontrollen werden konsequent angewendet |
-| 📄**Antwort-API-Kompatibilität** | Vollständige „/v1/responses“-Unterstützung für Codex und erweiterte Agenten-Workflows | ### 🎵 Multi-Modal APIs |
+### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
-| Funktion | Was es tut |
-| ------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------- |
-| 🖼️**Bilderzeugung** | `/v1/images/generations` mit Cloud- und lokalen Backends |
-| 📐**Einbettungen** | `/v1/embeddings` für Such- und RAG-Pipelines |
-| 🎤**Audio-Transkription** | „/v1/audio/transcriptions“ – 7 Anbieter (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), automatische Spracherkennung, MP4/MP3/WAV-Unterstützung |
-| 🔊**Text-to-Speech** | „/v1/audio/speech“ – 10 Anbieter (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) mit korrekten Fehlermeldungen |
-| 🎬**Videogenerierung** | `/v1/videos/generations` (ComfyUI + SD WebUI-Workflows) |
-| 🎵**Musikgeneration** | `/v1/music/generations` (ComfyUI-Workflows) |
-| 🛡️**Moderationen** | `/v1/moderations` Sicherheitsüberprüfungen |
-| 🔀**Neueinstufung** | `/v1/rerank` für Relevanzbewertung |
-| 🔍**Websuche**🆕 | „/v1/search“ – 5 Anbieter (Serper, Brave, Perplexity, Exa, Tavily), 6.500+ kostenlos/Monat, automatisches Failover, Cache | ### 🛡️ Resilience, Security & Governance |
+| Feature | What It Does |
+| ------------------------------------ | ------------------------------------------------------------------------------------------- |
+| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) |
+| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family |
+| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 |
+| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models |
+| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content |
+| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data |
+| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges |
+| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins |
-| Funktion | Was es tut |
-| ----------------------------------------------- | ------------------------------------------------------------------------------------------------------------------ | -------------------------------- |
-| 🔌**Leistungsschalter** | Auslösung/Wiederherstellung pro Modell mit Schwellenwertkontrollen |
-| 🎯**Endpunktfähige Modelle** | Benutzerdefinierte Modelle deklarieren unterstützte Endpunkte + API-Format |
-| 🛡️**Anti-Donnerende Herde** | Mutex- und Semaphorschutz bei Wiederholungs-/Ratenereignissen |
-| 🧠**Semantik + Signatur-Cache** | Kosten-/Latenzreduzierung mit zwei Cache-Schichten |
-| ⚡**Idempotenz anfordern** | Doppeltes Schutzfenster |
-| 🔒**TLS-Fingerabdruck-Spoofing** | Browserähnlicher TLS-Fingerabdruck –**reduziert die Bot-Erkennung und Kontokennzeichnung** |
-| 🔏**CLI-Fingerabdruckabgleich** | Entspricht nativen CLI-Anfragesignaturen –**reduziert das Verbotsrisiko und behält gleichzeitig die Proxy-IP bei** |
-| 🌐**IP-Filterung** | Zulassungs-/Blocklistenkontrolle für exponierte Bereitstellungen |
-| 📊**Bearbeitbare Ratenlimits** | Konfigurierbare globale/Provider-Level-Limits mit Persistenz |
-| 📉**Anmutige Degradierung** | Mehrschichtige Fallbacks zum Schutz des Kern-Gateway-Betriebs |
-| 📜**Audit-Trail konfigurieren** | Diff-basierte Änderungsverfolgung verhindert betriebliche Abweichungen durch einfache Rollbacks |
-| ⏳**Provider Health Sync** | Proaktive Überwachung des Token-Ablaufs, die Warnungen vor Autorisierungsfehlern auslöst |
-| 🚪**Gesperrte Konten automatisch deaktivieren** | Funktionsfähiger Leistungsschalter, der dauerhaft gesperrte Token-Konten automatisch verschließt |
-| 🔑**API-Schlüsselverwaltung + Scoping** | Sichere Schlüsselausgabe/-rotation und Modell-/Anbieterkontrollen |
-| 👁️**Scoped API Key Reveal**🆕 | Opt-in-Wiederherstellung von API-Schlüsseln über „ALLOW_API_KEY_REVEAL“ |
-| 🛡️**Geschützte „/Modelle“** | Optionales Authentifizierungs-Gating und Provider-Ausblenden für Modellkatalog | ### 📊 Observability & Analytics |
+### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP
-| Funktion | Was es tut |
-| ------------------------------------------ | ------------------------------------------------------------------- | ---------------------------- |
-| 📝**Anfrage + Proxy-Protokollierung** | Vollständige Anfrage/Antwort- und Proxy-Protokollierung |
-| 📉**Gestreamte detaillierte Protokolle**🆕 | Rekonstruiert SSE-Nutzlastströme sauber in der Benutzeroberfläche |
-| 📋**Einheitliches Protokoll-Dashboard** | Anforderungs-, Proxy-, Audit- und Konsolenansichten auf einer Seite |
-| 🔍**Telemetrie anfordern** | p50/p95/p99-Latenz und Anforderungsverfolgung |
-| 🏥**Gesundheits-Dashboard** | Betriebszeit, Breaker-Zustände, Sperrungen, Cache-Statistiken |
-| 💰**Kostenverfolgung** | Budgetkontrolle und Preistransparenz pro Modell |
-| 📈**Analysevisualisierungen** | Einblicke in die Modell-/Anbieternutzung und Trendansichten |
-| 🧪**Bewertungsrahmen** | Golden-Set-Test mit konfigurierbaren Match-Strategien |
-| 📡**Live-Diagnose**🆕 | Semantische Cache-Umgehung für genaue Combo-Live-Tests | ### ☁️ Deployment & Platform |
+| Feature | What It Does |
+| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing |
+| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** |
+| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint |
+| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. |
+| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator |
+| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID |
+| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart |
-| Funktion | Was es tut |
-| ------------------------------------------ | ------------------------------------------------------------------------------ | --------------------- |
-| 🌐**Überall bereitstellen** | Localhost, VPS, Docker, Cloud-Umgebungen |
-| 🚇**Cloudflare-Tunnel**🆕 | Quick-Tunnel-Integration mit einem Klick über das Dashboard |
-| 🔑**API-Schlüsselmodellfilterung** | Native /v1/models-Antwort gefiltert über zugewiesene Bearer-Kontextrollen |
-| ⚡**Smart Cache Bypass** | Konfigurierbare TTL-Heuristik und erzwungene Refetch-Kontrollen |
-| 🔄**Sichern/Wiederherstellen** | Export-/Import- und Disaster-Recovery-Abläufe |
-| 🧙**Onboarding-Assistent** | Erstmaliges geführtes Setup |
-| 🔧**CLI-Tools-Dashboard** | Ein-Klick-Setup für beliebte Codierungstools |
-| 🎮**Modellspielplatz** | Testen Sie alle Anbieter/Modelle/Endpunkte über das Dashboard |
-| 🔏**CLI-Fingerabdruck-Umschaltung** | Fingerabdruckabgleich pro Anbieter unter Einstellungen > Sicherheit |
-| 🌐**i18n (30 Sprachen)** | Vollständige Sprachunterstützung für Dashboard und Dokumente mit RTL-Abdeckung |
-| 🧹**Alle Modelle löschen** | Löschen der Modellliste in den Anbieterdetails mit einem Klick |
-| 👁️**Sidebar-Steuerelemente**🆕 | Komponenten und Integrationen in den Darstellungseinstellungen ausblenden |
-| 📋**Problemvorlagen** | Standardisierte GitHub-Vorlagen für Fehler und Funktionen |
-| 📂**Benutzerdefiniertes Datenverzeichnis** | „DATA_DIR“-Überschreibung für Speicherort | ### Feature Deep Dive |
+### 🤖 Agent & Protocol Operations (v2.0)
+
+| Feature | What It Does |
+| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
+| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools |
+| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows |
+| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs |
+| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) |
+| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) |
+| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution |
+| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access |
+| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks |
+| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery |
+| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` |
+| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface |
+
+### 🧠 Routing & Intelligence
+
+| Feature | What It Does |
+| ---------------------------------- | ------------------------------------------------------------------------ |
+| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free |
+| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider |
+| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions |
+| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection |
+| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry |
+| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control |
+| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session |
+| 🌐 **Wildcard Router** | `provider/*` dynamic routing |
+| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits |
+| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety |
+| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models |
+| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) |
+| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions |
+| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity |
+| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution |
+| 💬 **System Prompt Injection** | Global behavior controls applied consistently |
+| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows |
+
+### 🎵 Multi-Modal APIs
+
+| Feature | What It Does |
+| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends |
+| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines |
+| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support |
+| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages |
+| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) |
+| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) |
+| 🛡️ **Moderations** | `/v1/moderations` safety checks |
+| 🔀 **Reranking** | `/v1/rerank` for relevance scoring |
+| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache |
+
+### 🛡️ Resilience, Security & Governance
+
+| Feature | What It Does |
+| ----------------------------------- | -------------------------------------------------------------------------------------- |
+| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls |
+| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format |
+| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events |
+| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers |
+| ⚡ **Request Idempotency** | Duplicate protection window |
+| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** |
+| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** |
+| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments |
+| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence |
+| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations |
+| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks |
+| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures |
+| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically |
+| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls |
+| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` |
+| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog |
+
+### 📊 Observability & Analytics
+
+| Feature | What It Does |
+| -------------------------------- | ----------------------------------------------------- |
+| 📝 **Request + Proxy Logging** | Full request/response and proxy logging |
+| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI |
+| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page |
+| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing |
+| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats |
+| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility |
+| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views |
+| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies |
+| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing |
+
+### ☁️ Deployment & Platform
+
+| Feature | What It Does |
+| ------------------------------ | --------------------------------------------------------------------- |
+| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments |
+| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard |
+| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles |
+| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls |
+| 🔄 **Backup/Restore** | Export/import and disaster recovery flows |
+| 🧙 **Onboarding Wizard** | First-run guided setup |
+| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools |
+| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard |
+| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security |
+| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage |
+| 🧹 **Clear All Models** | One-click model list clearing in provider details |
+| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings |
+| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features |
+| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location |
+
+### Feature Deep Dive
#### Smart fallback with practical cost control
@@ -1295,103 +1464,132 @@ Combo: "my-coding-stack"
4. if/kimi-k2-thinking
```
-Wenn Kontingent, Rate oder Integrität fehlschlagen, wechselt OmniRoute automatisch zum nächsten Kandidaten, ohne dass ein manueller Wechsel erforderlich ist.#### Protocol management that is visible and operable
+When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching.
-- MCP + A2A sind in der Benutzeroberfläche und in den Dokumenten erkennbar (nicht ausgeblendet)
-- Protokollstatus-APIs stellen Live-Betriebsdaten bereit (`/api/mcp/*`, `/api/a2a/*`)
-- Dashboards umfassen Aktionen für Tag-2-Operationen (Kombinationsumschaltung, Zurücksetzen von Leistungsschaltern, Aufgabenabbruch).#### Translator + validation workflow
+#### Protocol management that is visible and operable
-Der Übersetzerbereich umfasst:
+- MCP + A2A are discoverable in UI and docs (not hidden)
+- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`)
+- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation)
--**Spielplatz**: Transformationsprüfungen anfordern -**Chat-Tester**: vollständiger Anfrage-/Antwort-Roundtrip -**Prüfstand**: mehrere Fälle in einem Durchgang -**Live Monitor**: Echtzeit-Verkehrsansicht
+#### Translator + validation workflow
-Plus Protokollvalidierung mit echten Clients über „npm run test:protocols:e2e“.
+The Translator area includes:
-> 📖**[MCP Server README](open-sse/mcp-server/README.md)**– Tool-Referenz, IDE-Konfigurationen und Client-Beispiele
+- **Playground**: request transformation checks
+- **Chat Tester**: full request/response round-trip
+- **Test Bench**: multiple cases in one run
+- **Live Monitor**: real-time traffic view
+
+Plus protocol validation with real clients via `npm run test:protocols:e2e`.
+
+> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples
>
-> 📖**[A2A Server README](src/lib/a2a/README.md)**– Fähigkeiten, JSON-RPC-Methoden, Streaming und Aufgabenlebenszyklus## 🧪 Evaluations (Evals)
+> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle
-OmniRoute umfasst ein integriertes Bewertungsframework zum Testen der LLM-Antwortqualität anhand eines Golden Sets. Greifen Sie darauf über**Analytics → Evals**im Dashboard zu.### Built-in Golden Set
+## 🧪 Evaluations (Evals)
-Das vorinstallierte „OmniRoute Golden Set“ enthält Testfälle für:
+OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard.
-- Grüße, Mathematik, Geographie, Codegenerierung
-- Einhaltung des JSON-Formats, Übersetzung, Markdown-Generierung
-- Sicherheitsverweigerung (schädlicher Inhalt), Zählung, boolesche Logik### Evaluation Strategies
+### Built-in Golden Set
-| Strategie | Beschreibung | Beispiel |
-| ------------------- | -------------------------------------------------------------------------------------------- | --------------------------------------- | --- |
-| „genau“ | Die Ausgabe muss genau mit | übereinstimmen „4“ |
-| „enthält“ | Die Ausgabe muss eine Teilzeichenfolge enthalten (Groß-/Kleinschreibung wird nicht beachtet) | „Paris“ |
-| `regex` | Die Ausgabe muss mit dem Regex-Muster | übereinstimmen `"1.*2.*3"` |
-| „Benutzerdefiniert“ | Benutzerdefinierte JS-Funktion gibt true/false | zurück `(Ausgabe) => Ausgabelänge > 10` | --- |
+The pre-loaded "OmniRoute Golden Set" contains test cases for:
+
+- Greetings, math, geography, code generation
+- JSON format compliance, translation, markdown generation
+- Safety refusal (harmful content), counting, boolean logic
+
+### Evaluation Strategies
+
+| Strategy | Description | Example |
+| ---------- | ------------------------------------------------ | -------------------------------- |
+| `exact` | Output must match exactly | `"4"` |
+| `contains` | Output must contain substring (case-insensitive) | `"Paris"` |
+| `regex` | Output must match regex pattern | `"1.*2.*3"` |
+| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` |
+
+---
## 📖 Setup Guide
### Protocol Setup (MCP + A2A)
-
-🧩 MCP-Setup (Model Context Protocol)
+
+🧩 MCP Setup (Model Context Protocol)
-Starten Sie den MCP-Transport im Standardmodus:```bash
+Start MCP transport in stdio mode:
+
+```bash
omniroute --mcp
+```
-````
+Recommended validation flow:
-Empfohlener Validierungsablauf:
+1. Connect your MCP client over stdio.
+2. Run `omniroute_get_health`.
+3. Run `omniroute_list_combos`.
+4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit.
-1. Verbinden Sie Ihren MCP-Client über stdio.
-2. Führen Sie „omniroute_get_health“ aus.
-3. Führen Sie „omniroute_list_combos“ aus.
-4. Öffnen Sie „/dashboard/mcp“, um Heartbeat, Aktivität und Audit zu bestätigen.
-
-Nützliche APIs für die Automatisierung:
+Useful APIs for automation:
- `GET /api/mcp/status`
- `GET /api/mcp/tools`
- `GET /api/mcp/audit`
-- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats`
-
-🤝 A2A-Setup (Agent2Agent)
+
-Entdecken Sie den Agenten:```bash
+
+🤝 A2A Setup (Agent2Agent)
+
+Discover the agent:
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
-Senden Sie eine Aufgabe:```bash
+Send a task:
+
+```bash
curl -X POST http://localhost:20128/a2a \
- -H 'content-type: application/json' \
- -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+ -H 'content-type: application/json' \
+ -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+```
-````
-
-Lebenszyklus verwalten:
+Manage lifecycle:
- `GET /api/a2a/status`
- `GET /api/a2a/tasks`
- `GET /api/a2a/tasks/:id`
- `POST /api/a2a/tasks/:id/cancel`
-Operative Benutzeroberfläche:
+Operational UI:
-- „/dashboard/a2a“ für Aufgaben-/Status-/Stream-Beobachtbarkeit und Smoke-Aktionen
+- `/dashboard/a2a` for task/state/stream observability and smoke actions
-
-🧪 End-to-End-Protokollvalidierung
+
-Validieren Sie beide Protokolle mit echten Clients:```bash
+
+🧪 End-to-end protocol validation
+
+Validate both protocols with real clients:
+
+```bash
npm run test:protocols:e2e
-````
+```
-Dies bestätigt:
+This verifies:
-- MCP SDK-Client-Verbindung/Liste/Anruf
-- A2A-Erkennung/Senden/Streamen/Get/Abbrechen
-- Vergleichen Sie die Daten in MCP-Audit- und A2A-Aufgabenverwaltungs-APIs
+- MCP SDK client connect/list/call
+- A2A discovery/send/stream/get/cancel
+- Cross-check data in MCP audit and A2A task management APIs
-
-💳 Abonnementanbieter
### Claude Code (Pro/Max)
+
+
+
+💳 Subscription Providers
+
+### Claude Code (Pro/Max)
```bash
Dashboard → Providers → Connect Claude Code
@@ -1404,7 +1602,9 @@ Models:
cc/claude-haiku-4-5-20251001
```
-**Profi-Tipp:**Verwenden Sie Opus für komplexe Aufgaben, Sonnet für Geschwindigkeit. OmniRoute verfolgt das Kontingent pro Modell!### OpenAI Codex (Plus/Pro)
+**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model!
+
+### OpenAI Codex (Plus/Pro)
```bash
Dashboard → Providers → Connect Codex
@@ -1418,20 +1618,22 @@ Models:
#### Codex Account Limit Management (5h + Weekly)
-Für jedes Codex-Konto gibt es jetzt Richtlinienumschaltungen unter „Dashboard -> Anbieter“:
+Each Codex account now has policy toggles in `Dashboard -> Providers`:
-- „5h“ (EIN/AUS): Erzwingt die 5-Stunden-Fensterschwellenrichtlinie.
-- „Wöchentlich“ (EIN/AUS): Erzwingen Sie die wöchentliche Fensterschwellenrichtlinie.
- – Schwellenwertverhalten: Wenn ein aktiviertes Fenster eine Nutzung von >=90 % erreicht, wird dieses Konto übersprungen.
-- Rotationsverhalten: OmniRoute leitet automatisch zum nächsten berechtigten Codex-Konto weiter.
-- Zurücksetzungsverhalten: Wenn die „resetAt“-Zeit des Anbieters verstrichen ist, wird das Konto automatisch wieder berechtigt.
+- `5h` (ON/OFF): enforce the 5-hour window threshold policy.
+- `Weekly` (ON/OFF): enforce the weekly window threshold policy.
+- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped.
+- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically.
+- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically.
-Szenarien:
+Scenarios:
-- „5 Stunden EIN“ + „Wöchentlich EIN“: Das Konto wird übersprungen, wenn eines der Fenster den Schwellenwert erreicht.
-- „5h AUS“ + „Wöchentlich EIN“: Nur wöchentliche Nutzung kann das Konto sperren.
-- „5h EIN“ + „Wöchentlich AUS“: Nur eine 5-stündige Nutzung kann das Konto sperren.
-- „resetAt“ übergeben: Das Konto wechselt automatisch wieder in die Rotation (keine manuelle erneute Aktivierung).### Gemini CLI (FREE 180K/month!)
+- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold.
+- `5h OFF` + `Weekly ON`: only weekly usage can block the account.
+- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account.
+- `resetAt` passed: account re-enters rotation automatically (no manual re-enable).
+
+### Gemini CLI (FREE 180K/month!)
```bash
Dashboard → Providers → Connect Gemini CLI
@@ -1443,7 +1645,9 @@ Models:
gc/gemini-2.5-pro
```
-**Bester Wert:**Riesiges kostenloses Kontingent! Verwenden Sie dies vor kostenpflichtigen Stufen.### GitHub Copilot
+**Best Value:** Huge free tier! Use this before paid tiers.
+
+### GitHub Copilot
```bash
Dashboard → Providers → Connect GitHub
@@ -1458,71 +1662,91 @@ Models:
-
-🔑 API-Schlüsselanbieter
### NVIDIA NIM (FREE developer access — 70+ models)
+
+🔑 API Key Providers
-1. Registrieren Sie sich: [build.nvidia.com](https://build.nvidia.com)
-2. Holen Sie sich einen kostenlosen API-Schlüssel (1000 Inferenz-Credits inbegriffen)
-3. Dashboard → Anbieter hinzufügen → NVIDIA NIM:
- - API-Schlüssel: „nvapi-your-key“.
+### NVIDIA NIM (FREE developer access — 70+ models)
-**Modelle:**„nvidia/llama-3.3-70b-instruct“, „nvidia/mistral-7b-instruct“ und mehr als 50 weitere
+1. Sign up: [build.nvidia.com](https://build.nvidia.com)
+2. Get free API key (1000 inference credits included)
+3. Dashboard → Add Provider → NVIDIA NIM:
+ - API Key: `nvapi-your-key`
-**Profi-Tipp:**OpenAI-kompatible API – funktioniert nahtlos mit der Formatübersetzung von OmniRoute!### DeepSeek
+**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more
-1. Registrieren Sie sich: [platform.deepseek.com](https://platform.deepseek.com)
-2. Holen Sie sich den API-Schlüssel
-3. Dashboard → Anbieter hinzufügen → DeepSeek
+**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation!
-**Modelle:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!)
+### DeepSeek
-1. Registrieren Sie sich: [console.groq.com](https://console.groq.com)
-2. Holen Sie sich den API-Schlüssel (kostenloses Kontingent inbegriffen)
-3. Dashboard → Anbieter hinzufügen → Groq
+1. Sign up: [platform.deepseek.com](https://platform.deepseek.com)
+2. Get API key
+3. Dashboard → Add Provider → DeepSeek
-**Modelle:**„groq/llama-3.3-70b“, „groq/mixtral-8x7b“.
+**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder`
-**Profi-Tipp:**Ultraschnelle Inferenz – am besten für Echtzeit-Codierung!### OpenRouter (100+ Models)
+### Groq (Free Tier Available!)
-1. Registrieren Sie sich: [openrouter.ai](https://openrouter.ai)
-2. Holen Sie sich den API-Schlüssel
-3. Dashboard → Anbieter hinzufügen → OpenRouter
+1. Sign up: [console.groq.com](https://console.groq.com)
+2. Get API key (free tier included)
+3. Dashboard → Add Provider → Groq
-**Modelle:**Greifen Sie über einen einzigen API-Schlüssel auf über 100 Modelle aller großen Anbieter zu.
+**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b`
-**Dashboard-Verhalten:**OpenRouter-Modelle werden über**Verfügbare Modelle**verwaltet. Durch manuelles Hinzufügen, Importieren und automatische Synchronisieren wird dieselbe Liste aktualisiert.
+**Pro Tip:** Ultra-fast inference — best for real-time coding!
-
-💰 Günstige Anbieter (Backup)
### GLM-4.7 (Daily reset, $0.6/1M)
+### OpenRouter (100+ Models)
-1. Registrieren Sie sich: [Zhipu AI](https://open.bigmodel.cn/)
-2. Holen Sie sich den API-Schlüssel vom Coding Plan
-3. Dashboard → API-Schlüssel hinzufügen:
- - Anbieter: `glm`
- - API-Schlüssel: „Ihr-Schlüssel“.
+1. Sign up: [openrouter.ai](https://openrouter.ai)
+2. Get API key
+3. Dashboard → Add Provider → OpenRouter
-**Verwenden Sie:**`glm/glm-4.7`
+**Models:** Access 100+ models from all major providers through a single API key.
-**Profi-Tipp:**Coding Plan bietet 3× Kontingent zu 1/7 Kosten! Täglich um 10:00 Uhr zurückgesetzt.### MiniMax M2.1 (5h reset, $0.20/1M)
+**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list.
-1. Registrieren Sie sich: [MiniMax](https://www.minimax.io/)
-2. Holen Sie sich den API-Schlüssel
-3. Dashboard → API-Schlüssel hinzufügen
+
-**Verwenden Sie:**„minimax/MiniMax-M2.1“.
+
+💰 Cheap Providers (Backup)
-**Profi-Tipp:**Günstigste Option für langen Kontext (1 Mio. Token)!### Kimi K2 ($9/month flat)
+### GLM-4.7 (Daily reset, $0.6/1M)
-1. Abonnieren: [Moonshot AI](https://platform.moonshot.ai/)
-2. Holen Sie sich den API-Schlüssel
-3. Dashboard → API-Schlüssel hinzufügen
+1. Sign up: [Zhipu AI](https://open.bigmodel.cn/)
+2. Get API key from Coding Plan
+3. Dashboard → Add API Key:
+ - Provider: `glm`
+ - API Key: `your-key`
-**Verwendung:**`kimi/kimi-latest`
+**Use:** `glm/glm-4.7`
-**Profi-Tipp:**Festpreis: 9 $/Monat für 10 Mio. Token = 0,90 $/1 Mio. effektive Kosten!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM.
-
-🆓 KOSTENLOSE Anbieter (Notfall-Backup)
### Qoder (5 FREE models via OAuth)
+### MiniMax M2.1 (5h reset, $0.20/1M)
+
+1. Sign up: [MiniMax](https://www.minimax.io/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `minimax/MiniMax-M2.1`
+
+**Pro Tip:** Cheapest option for long context (1M tokens)!
+
+### Kimi K2 ($9/month flat)
+
+1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `kimi/kimi-latest`
+
+**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost!
+
+
+
+
+🆓 FREE Providers (Emergency Backup)
+
+### Qoder (5 FREE models via OAuth)
```bash
Dashboard → Connect Qoder
@@ -1563,8 +1787,10 @@ Models:
-
-🎨 Combos erstellen
### Example 1: Maximize Subscription → Cheap Backup
+
+🎨 Create Combos
+
+### Example 1: Maximize Subscription → Cheap Backup
```
Dashboard → Combos → Create New
@@ -1592,8 +1818,10 @@ Cost: $0 forever!
-
-🔧 CLI-Integration
### Cursor IDE
+
+🔧 CLI Integration
+
+### Cursor IDE
```
Settings → Models → Advanced:
@@ -1604,7 +1832,9 @@ Settings → Models → Advanced:
### Claude Code
-Verwenden Sie die Seite**CLI-Tools**im Dashboard für die Ein-Klick-Konfiguration oder bearbeiten Sie „~/.claude/settings.json“ manuell.### Codex CLI
+Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually.
+
+### Codex CLI
```bash
export OPENAI_BASE_URL="http://localhost:20128"
@@ -1615,12 +1845,15 @@ codex "your prompt"
### OpenClaw
-**Option 1 – Dashboard (empfohlen):**```
+**Option 1 — Dashboard (recommended):**
+
+```
Dashboard → CLI Tools → OpenClaw → Select Model → Apply
+```
-````
+**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`:
-**Option 2 – Manuell:**Bearbeiten Sie „~/.openclaw/openclaw.json“:```json
+```json
{
"models": {
"providers": {
@@ -1632,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply
}
}
}
-````
+```
-> **Hinweis:**OpenClaw funktioniert nur mit lokaler OmniRoute. Verwenden Sie „127.0.0.1“ anstelle von „localhost“, um Probleme mit der IPv6-Auflösung zu vermeiden.### Cline / Continue / RooCode
+> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues.
+
+### Cline / Continue / RooCode
```
Settings → API Configuration:
@@ -1646,15 +1881,17 @@ Settings → API Configuration:
### OpenCode
-**Schritt 1:**OmniRoute als benutzerdefinierten Anbieter hinzufügen:```bash
+**Step 1:** Add OmniRoute as a custom provider:
+
+```bash
opencode
/connect
-
# Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key
+```
-````
+**Step 2:** Create/edit `opencode.json` in your project root:
-**Schritt 2:**Erstellen/bearbeiten Sie „opencode.json“ in Ihrem Projektstammverzeichnis:```json
+```json
{
"$schema": "https://opencode.ai/config.json",
"provider": {
@@ -1672,117 +1909,130 @@ opencode
}
}
}
-````
+```
-**Schritt 3:**Wählen Sie das Modell in OpenCode aus:```bash
+**Step 3:** Select the model in OpenCode:
+
+```bash
/models
-
# Select any OmniRoute model from the list
+```
-````
+> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard.
->**Tipp:**Fügen Sie alle in Ihrem OmniRoute-Endpunkt „/v1/models“ verfügbaren Modelle zum Abschnitt „Modelle“ hinzu. Verwenden Sie das Format „Anbieter/Modell-ID“ aus Ihrem OmniRoute-Dashboard.
+
---
## Fehlerbehebung
-
-Klicken Sie hier, um die Anleitung zur Fehlerbehebung zu erweitern
+
+Click to expand troubleshooting guide
-**„Sprachmodell hat keine Nachrichten bereitgestellt“**
+**"Language model did not provide messages"**
-- Anbieterkontingent erschöpft → Überprüfen Sie den Dashboard-Kontingent-Tracker
-- Lösung: Combo-Fallback verwenden oder auf günstigere Stufe wechseln
+- Provider quota exhausted → Check dashboard quota tracker
+- Solution: Use combo fallback or switch to cheaper tier
-**Ratenbegrenzung**
+**Rate limiting**
-- Abonnementkontingent aufgebraucht → Fallback auf GLM/MiniMax
-- Kombination hinzufügen: „cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking“.
+- Subscription quota out → Fallback to GLM/MiniMax
+- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-**OAuth-Token abgelaufen**
+**OAuth token expired**
-- Automatische Aktualisierung durch OmniRoute
-- Wenn die Probleme weiterhin bestehen: Dashboard → Anbieter → Verbindung wiederherstellen
+- Auto-refreshed by OmniRoute
+- If issues persist: Dashboard → Provider → Reconnect
-**Hohe Kosten**
+**High costs**
-- Überprüfen Sie die Nutzungsstatistiken im Dashboard → Kosten
-- Primärmodell auf GLM/MiniMax umstellen
-- Nutzen Sie den kostenlosen Tarif (Gemini CLI, Qoder) für unkritische Aufgaben
+- Check usage stats in Dashboard → Costs
+- Switch primary model to GLM/MiniMax
+- Use free tier (Gemini CLI, Qoder) for non-critical tasks
-**Dashboard-/API-Ports sind falsch**
+**Dashboard/API ports are wrong**
-- „PORT“ ist der kanonische Basisport (und standardmäßig API-Port)
-– „API_PORT“ überschreibt nur den OpenAI-kompatiblen API-Listener
-– „DASHBOARD_PORT“ überschreibt nur den Dashboard/Next.js-Listener
-- Setzen Sie „NEXT_PUBLIC_BASE_URL“ auf Ihr Dashboard/öffentliche URL (für OAuth-Rückrufe)
+- `PORT` is the canonical base port (and API port by default)
+- `API_PORT` overrides only OpenAI-compatible API listener
+- `DASHBOARD_PORT` overrides only dashboard/Next.js listener
+- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks)
-**Cloud-Synchronisierungsfehler**
+**Cloud sync errors**
-- Überprüfen Sie, ob „BASE_URL“ auf Ihre laufende Instanz verweist
-– Überprüfen Sie, ob „CLOUD_URL“ auf Ihren erwarteten Cloud-Endpunkt verweist
-- Halten Sie die Werte von „NEXT_PUBLIC_*“ an den serverseitigen Werten ausgerichtet
+- Verify `BASE_URL` points to your running instance
+- Verify `CLOUD_URL` points to your expected cloud endpoint
+- Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**Erste Anmeldung funktioniert nicht**
+**First login not working**
-- Überprüfen Sie „INITIAL_PASSWORD“ in „.env“.
-- Wenn nicht festgelegt, lautet das Fallback-Passwort „123456“.
+- Check `INITIAL_PASSWORD` in `.env`
+- If unset, fallback password is `123456`
-**Keine Anfrageprotokolle**
+**No request logs**
-– Anforderungsartefakte werden als eine JSON-Datei pro Anforderung in „DATA_DIR/call_logs/“ geschrieben
-- Aktivieren Sie die Pipeline-Erfassung über Dashboard → Protokolle → Protokolle anfordern, wenn Sie detaillierte Payloads pro Phase benötigen
-- Legen Sie „APP_LOG_TO_FILE=true“ fest, wenn Sie auch Anwendungskonsolenprotokolle in „logs/application/app.log“ haben möchten
-- Passen Sie „APP_LOG_MAX_FILE_SIZE“, „APP_LOG_RETENTION_DAYS“, „APP_LOG_MAX_FILES“ und „CALL_LOG_MAX_ENTRIES“ nach Bedarf an
+- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request
+- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads
+- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log`
+- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed
-**Verbindungstest zeigt „Ungültig“ für OpenAI-kompatible Anbieter**
+**Connection test shows "Invalid" for OpenAI-compatible providers**
-– Viele Anbieter stellen keinen „/models“-Endpunkt bereit
-– OmniRoute v1.0.6+ beinhaltet eine Fallback-Validierung über Chat-Abschlüsse
-– Stellen Sie sicher, dass die Basis-URL das Suffix „/v1“ enthält### 🔐 OAuth on a Remote Server
+- Many providers don't expose a `/models` endpoint
+- OmniRoute v1.0.6+ includes fallback validation via chat completions
+- Ensure base URL includes `/v1` suffix
+
+### 🔐 OAuth on a Remote Server
-
+
->**⚠️ Wichtig für Benutzer, die OmniRoute auf einem VPS, Docker oder einem anderen Remote-Server ausführen**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
+> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server**
-Die Anbieter**Antigravity**und**Gemini CLI**verwenden**Google OAuth 2.0**. Google verlangt, dass „redirect_uri“ im OAuth-Flow genau mit einem der vorregistrierten URIs in der Google Cloud Console der App übereinstimmt.
+#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
-Die in OmniRoute gebündelten OAuth-Anmeldeinformationen werden**nur für „localhost“**registriert. Wenn Sie auf OmniRoute auf einem Remote-Server zugreifen (z. B. „https://omniroute.myserver.com“), lehnt Google die Authentifizierung mit Folgendem ab:```
+The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console.
+
+The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solution: Configure your own OAuth credentials
-Sie müssen in der Google Cloud Console eine**OAuth 2.0-Client-ID**mit dem URI Ihres Servers erstellen.#### Step-by-step
+You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI.
-**1. Öffnen Sie die Google Cloud Console**
+#### Step-by-step
-Gehen Sie zu: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
+**1. Open Google Cloud Console**
-**2. Erstellen Sie eine neue OAuth 2.0-Client-ID**
+Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-- Klicken Sie auf**„+ Anmeldeinformationen erstellen“**→**„OAuth-Client-ID“**
-- Anwendungstyp:**„Webanwendung“**
-- Name: beliebig (z. B. „OmniRoute Remote“)
+**2. Create a new OAuth 2.0 Client ID**
-**3. Autorisierte Weiterleitungs-URIs hinzufügen**
+- Click **"+ Create Credentials"** → **"OAuth client ID"**
+- Application type: **"Web application"**
+- Name: anything you like (e.g. `OmniRoute Remote`)
-Fügen Sie im Feld**"Autorisierte Weiterleitungs-URIs"**Folgendes hinzu:```
+**3. Add Authorized Redirect URIs**
+
+In the **"Authorized redirect URIs"** field, add:
+
+```
https://your-server.com/callback
+```
-````
+> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`).
-> Ersetzen Sie „Ihr-Server.com“ durch die Domäne oder IP Ihres Servers (geben Sie bei Bedarf den Port ein, z. B. „http://45.33.32.156:20128/callback“).
+**4. Save and copy the credentials**
-**4. Speichern und kopieren Sie die Anmeldeinformationen**
+After creating, Google will show the **Client ID** and **Client Secret**.
-Nach der Erstellung zeigt Google die**Client-ID**und das**Client-Geheimnis**an.
+**5. Set environment variables**
-**5. Umgebungsvariablen festlegen**
+In your `.env` (or Docker environment variables):
-In Ihrer „.env“ (oder Docker-Umgebungsvariablen):```bash
+```bash
# For Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
@@ -1791,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
-````
+```
-**6. OmniRoute neu starten**```bash
+**6. Restart OmniRoute**
+```bash
# npm:
-
npm run dev
# Docker:
-
docker restart omniroute
+```
-````
+**7. Try connecting again**
-**7. Versuchen Sie erneut, eine Verbindung herzustellen**
+Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth
-Dashboard → Anbieter → Antigravity (oder Gemini CLI) → OAuth
+Google will now redirect correctly to `https://your-server.com/callback`.
-Google leitet jetzt korrekt zu „https://your-server.com/callback“ weiter.---
+---
#### Temporary workaround (without custom credentials)
-Wenn Sie jetzt keine eigenen Anmeldeinformationen einrichten möchten, können Sie dennoch den**manuellen URL-Ablauf**verwenden:
+If you don't want to set up your own credentials right now, you can still use the **manual URL flow**:
-1. OmniRoute öffnet die Google-Autorisierungs-URL
-2. Nach der Autorisierung versucht Google, auf „localhost“ umzuleiten (was auf dem Remote-Server fehlschlägt).
-3.**Kopieren Sie die vollständige URL**aus der Adressleiste Ihres Browsers (auch wenn die Seite nicht geladen wird)
-4. Fügen Sie diese URL in das Feld ein, das im OmniRoute-Verbindungsmodal angezeigt wird
-5. Klicken Sie auf**„Verbinden“**
+1. OmniRoute opens the Google authorization URL
+2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server)
+3. **Copy the full URL** from your browser's address bar (even if the page doesn't load)
+4. Paste that URL into the field shown in the OmniRoute connection modal
+5. Click **"Connect"**
-> Dies funktioniert, weil der Autorisierungscode in der URL unabhängig davon gültig ist, ob die Weiterleitungsseite geladen wurde.---
+> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded.
-
-🇧🇷 Versão em Português
#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+---
-Wir haben**Antigravity**und**Gemini CLI**mit**Google OAuth 2.0**zur Authentifizierung getestet. Google erwartet, dass „redirect_uri“ kein OAuth-Fluss verwendet, da**exatamente**ein URI vorab in die Google Cloud Console aufgenommen wurde.
+
+🇧🇷 Versão em Português
-Als OAuth-Anmelder wurde OmniRoute nicht als „localhost“**registriert. Wenn Sie auf einen Remote-Server (z. B. „https://omniroute.meuservidor.com“) auf OmniRoute zugreifen, lehnt Google die Authentifizierung ab:```
+#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+
+Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo.
+
+As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solução: Configure suas próprias credenciais OAuth
-Sie schreiben bitte eine**OAuth 2.0-Client-ID**in der Google Cloud Console mit einem URI für Ihren Server.#### Passo a passo
+Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor.
-**1. Zugriff auf die Google Cloud Console**
+#### Passo a passo
+
+**1. Acesse o Google Cloud Console**
Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-**2. Rufen Sie eine neue OAuth 2.0-Client-ID auf**
+**2. Crie um novo OAuth 2.0 Client ID**
-- Klicken Sie auf**"+ Anmeldeinformationen erstellen"**→**"OAuth-Client-ID"**
-- Anwendungstyp:**„Webanwendung“**
-- Name: Wählen Sie einen beliebigen Namen (z. B. „OmniRoute Remote“)
+- Clique em **"+ Create Credentials"** → **"OAuth client ID"**
+- Tipo de aplicativo: **"Web application"**
+- Nome: escolha qualquer nome (ex: `OmniRoute Remote`)
-**3. Adicione als autorisierte Weiterleitungs-URIs**
+**3. Adicione as Authorized Redirect URIs**
-Nein,**"Autorisierte Weiterleitungs-URIs"**, Zusatz:```
+No campo **"Authorized redirect URIs"**, adicione:
+
+```
https://seu-servidor.com/callback
+```
-````
+> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`).
-> Ersetzen Sie Ihren Server durch „seu-servidor.com“ oder die IP Ihres Servers (einschließlich der erforderlichen Portierung, z. B. „http://45.33.32.156:20128/callback“).
+**4. Salve e copie as credenciais**
-**4. Als Anmeldedaten speichern und kopieren**
+Após criar, o Google mostrará o **Client ID** e o **Client Secret**.
-Anschließend hat Google die**Client-ID**und das**Client-Geheimnis**angezeigt.
+**5. Configure as variáveis de ambiente**
-**5. Als Umgebungsvariationen konfigurieren**
+No seu `.env` (ou nas variáveis de ambiente do Docker):
-Kein `.env` (oder mehrere Docker-Umgebungsvarianten):```bash
+```bash
# Para Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
@@ -1870,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
-````
+```
-**6. Neuzugang zu OmniRoute**```bash
+**6. Reinicie o OmniRoute**
+```bash
# Se usando npm:
-
npm run dev
# Se usando Docker:
-
docker restart omniroute
-
-````
+```
**7. Tente conectar novamente**
-Dashboard → Anbieter → Antigravity (oder Gemini CLI) → OAuth
+Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth
-Dann leiten Sie Google direkt an „https://seu-servidor.com/callback“ weiter und überprüfen Sie die Funktion.---
+Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará.
+
+---
#### Workaround temporário (sem configurar credenciais próprias)
-Wenn Sie vorab keine Berechtigung erhalten möchten, besteht die Möglichkeit, das**URL-Handbuch**zu verwenden:
+Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**:
-1. OmniRoute ruft eine von Google autorisierte URL auf
-2. Nachdem Sie den Autor autorisiert haben, sendet Google eine Weiterleitung an „localhost“ (das bedeutet, dass Sie den Server nicht weiterleiten können).
-3.**Kopieren Sie eine vollständige URL**, um sie in Ihren Browser zu laden (bitte beachten Sie, dass die Seite noch nicht abgeschlossen ist).
-4. Geben Sie die URL ein, die nicht zur Verbindung mit OmniRoute verwendet werden soll
-5. Klicken Sie auf**„Connect“**
+1. O OmniRoute abrirá a URL de autorização do Google
+2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto)
+3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue)
+4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute
+5. Clique em **"Connect"**
-> Diese Problemumgehung funktioniert aufgrund des Autorisierungscodes auf der URL und ist unabhängig von der Weiterleitung oder Nicht-Weiterleitung gültig.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+
+
---
@@ -1908,64 +2171,73 @@ Wenn Sie vorab keine Berechtigung erhalten möchten, besteht die Möglichkeit, d
## 🛠️ Tech Stack
-
-Klicken Sie hier, um die Tech-Stack-Details zu erweitern
+
+Click to expand tech stack details
--**Laufzeit**: Node.js 18–22 LTS (⚠️ Node.js 24+ wird**nicht unterstützt**– native Binärdateien von „better-sqlite3“ sind inkompatibel)
--**Sprache**: TypeScript 5.9 –**100 % TypeScript**über „src/“ und „open-sse/“ (kein „any“ in Kernmodulen seit Version 2.0)
--**Framework**: Next.js 16 + React 19 + Tailwind CSS 4
--**Datenbank**: LowDB (JSON) + SQLite (Domänenstatus + Proxy-Protokolle + MCP-Prüfung + Routing-Entscheidungen)
--**Schemas**: Zod (MCP-Tool-I/O-Validierung, API-Verträge)
--**Protokolle**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
--**Streaming**: Vom Server gesendete Ereignisse (SSE)
--**Auth**: OAuth 2.0 (PKCE) + JWT + API-Schlüssel + MCP-bezogene Autorisierung
--**Testen**: Node.js-Testläufer + Vitest (über 900 Tests einschließlich Einheit, Integration, E2E)
--**CI/CD**: GitHub-Aktionen (automatische NPM-Veröffentlichung + Docker Hub bei Veröffentlichung)
--**Website**: [omniroute.online](https://omniroute.online)
--**Paket**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
--**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
--**Resilienz**: Leistungsschalter, exponentielles Backoff, Anti-Donner-Herde, TLS-Spoofing, automatische Kombinations-Selbstheilung
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible)
+- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0)
+- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4
+- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions)
+- **Schemas**: Zod (MCP tool I/O validation, API contracts)
+- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
+- **Streaming**: Server-Sent Events (SSE)
+- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization
+- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E)
+- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release)
+- **Website**: [omniroute.online](https://omniroute.online)
+- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
+- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
+- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing
+
+
---
## Dokumentation
-| Dokument | Beschreibung |
+| Document | Description |
| ---------------------------------------------- | --------------------------------------------------- |
-| [Benutzerhandbuch](docs/USER_GUIDE.md) | Anbieter, Kombinationen, CLI-Integration, Bereitstellung |
-| [API-Referenz](docs/API_REFERENCE.md) | Alle Endpunkte mit Beispielen |
-| [MCP-Server](open-sse/mcp-server/README.md) | 16 MCP-Tools, IDE-Konfigurationen, Python/TS/Go-Clients |
-| [A2A-Server](src/lib/a2a/README.md) | JSON-RPC 2.0-Protokoll, Fähigkeiten, Streaming, Aufgabenverwaltung |
-| [Auto-Combo-Engine](docs/auto-combo.md) | 6-Faktor-Bewertung, Moduspakete, Selbstheilung |
-| [Fehlerbehebung](docs/TROUBLESHOOTING.md) | Häufige Probleme und Lösungen |
-| [Architektur](docs/ARCHITECTURE.md) | Systemarchitektur und Interna |
-| [Mitwirken](CONTRIBUTING.md) | Entwicklungsaufbau und Richtlinien |
-| [OpenAPI-Spezifikation](docs/openapi.yaml) | OpenAPI 3.0-Spezifikation |
-| [Sicherheitsrichtlinie](SECURITY.md) | Schwachstellenmeldung und Sicherheitspraktiken |
-| [VM-Bereitstellung](docs/VM_DEPLOYMENT_GUIDE.md) | Vollständige Anleitung: VM + Nginx + Cloudflare-Setup |
-| [Features-Galerie](docs/FEATURES.md) | Visuelle Dashboard-Tour mit Screenshots |
-| [Release-Checkliste](docs/RELEASE_CHECKLIST.md) | Validierungsschritte vor der Veröffentlichung |---
+| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment |
+| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples |
+| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients |
+| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt |
+| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing |
+| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation |
+| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions |
+| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals |
+| [Contributing](CONTRIBUTING.md) | Development setup and guidelines |
+| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification |
+| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices |
+| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup |
+| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots |
+| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps |
+
+---
## 🗺️ Roadmap
-Für OmniRoute sind**210+ Funktionen**in mehreren Entwicklungsphasen geplant. Hier sind die Schlüsselbereiche:
+OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas:
-| Kategorie | Geplante Funktionen | Höhepunkte |
-| -------------- | ---------------- | -------------------------------------------------------------------------------------- |
-| 🧠**Routing & Intelligenz**| 25+ | Routing mit der niedrigsten Latenz, Tag-basiertes Routing, Quoten-Preflight, P2C-Kontoauswahl |
-| 🔒**Sicherheit & Compliance**| 20+ | SSRF-Härtung, Credential-Cloaking, Ratenbegrenzung pro Endpunkt, Verwaltungsschlüssel-Scoping |
-| 📊**Beobachtbarkeit**| 15+ | OpenTelemetry-Integration, Echtzeit-Kontingentüberwachung, Kostenverfolgung pro Modell |
-| 🔄**Anbieterintegrationen**| 20+ | Dynamische Modellregistrierung, Anbieter-Abklingzeiten, Multi-Account-Codex, Copilot-Kontingentanalyse |
-| ⚡**Leistung**| 15+ | Duale Cache-Schicht, Prompt-Cache, Antwort-Cache, Streaming-Keepalive, Batch-API |
-| 🌐**Ökosystem**| 10+ | WebSocket-API, Hot-Reload der Konfiguration, verteilter Konfigurationsspeicher, kommerzieller Modus |### 🔜 Coming Soon
+| Category | Planned Features | Highlights |
+| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- |
+| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection |
+| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping |
+| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model |
+| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing |
+| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API |
+| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode |
-- 🔗**OpenCode-Integration**– Native Anbieterunterstützung für die OpenCode AI-Codierungs-IDE
-- 🔗**TRAE-Integration**– Volle Unterstützung für das TRAE AI-Entwicklungsframework
-- 📦**Batch-API**– Asynchrone Stapelverarbeitung für Massenanfragen
-- 🎯**Tag-basiertes Routing**– Leiten Sie Anfragen basierend auf benutzerdefinierten Tags und Metadaten weiter
-- 💰**Niedrigste Kostenstrategie**– Wählen Sie automatisch den günstigsten verfügbaren Anbieter aus
+### 🔜 Coming Soon
-> 📝 Vollständige Funktionsspezifikationen verfügbar unter [`docs/new-features/`](docs/new-features/) (217 detaillierte Spezifikationen)---
+- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE
+- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework
+- 📦 **Batch API** — Asynchronous batch processing for bulk requests
+- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata
+- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider
+
+> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs)
+
+---
## 👥 Contributors
@@ -1973,18 +2245,20 @@ Für OmniRoute sind**210+ Funktionen**in mehreren Entwicklungsphasen geplant. Hi
### How to Contribute
-1. Forken Sie das Repository
-2. Erstellen Sie Ihren Feature-Zweig („git checkout -b feature/amazing-feature“)
-3. Übernehmen Sie Ihre Änderungen („git commit -m ‚Erstaunliche Funktion hinzufügen‘“)
-4. Zum Zweig pushen („git push origin feature/amazing-feature“)
-5. Öffnen Sie eine Pull-Anfrage
+1. Fork the repository
+2. Create your feature branch (`git checkout -b feature/amazing-feature`)
+3. Commit your changes (`git commit -m 'Add amazing feature'`)
+4. Push to the branch (`git push origin feature/amazing-feature`)
+5. Open a Pull Request
-Detaillierte Richtlinien finden Sie unter [CONTRIBUTING.md](CONTRIBUTING.md).### Releasing a New Version
+See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines.
+
+### Releasing a New Version
```bash
# Create a release — npm publish happens automatically
gh release create v2.0.0 --title "v2.0.0" --generate-notes
-````
+```
---
@@ -1996,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes
## 🙏 Acknowledgments
-Besonderer Dank geht an**[9router](https://github.com/decolua/9router)**von**[decolua](https://github.com/decolua)**– das ursprüngliche Projekt, das diesen Fork inspiriert hat. OmniRoute baut auf dieser unglaublichen Grundlage mit zusätzlichen Funktionen, multimodalen APIs und einer vollständigen Neufassung von TypeScript auf.
+Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite.
-Besonderer Dank geht an**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**– die ursprüngliche Go-Implementierung, die diese JavaScript-Portierung inspiriert hat.---
+Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port.
+
+---
## Lizenz
-MIT-Lizenz – Einzelheiten finden Sie unter [LIZENZ](LIZENZ).---
+MIT License - see [LICENSE](LICENSE) for details.
+
+---
Built with ❤️ for developers who code 24/7
diff --git a/docs/i18n/de/docs/ARCHITECTURE.md b/docs/i18n/de/docs/ARCHITECTURE.md
index 61a1ecca90..807f2a4df7 100644
--- a/docs/i18n/de/docs/ARCHITECTURE.md
+++ b/docs/i18n/de/docs/ARCHITECTURE.md
@@ -4,80 +4,93 @@
---
-_Letzte Aktualisierung: 28.03.2026_## Executive Summary
-OmniRoute ist ein lokales KI-Routing-Gateway und Dashboard, das auf Next.js basiert.
-Es bietet einen einzigen OpenAI-kompatiblen Endpunkt („/v1/\*“) und leitet den Datenverkehr über mehrere Upstream-Anbieter mit Übersetzung, Fallback, Token-Aktualisierung und Nutzungsverfolgung weiter.
-Kernkompetenzen:
+_Last updated: 2026-03-28_
-- OpenAI-kompatible API-Oberfläche für CLI/Tools (28 Anbieter)
-- Anforderungs-/Antwortübersetzung über Anbieterformate hinweg
-- Modell-Combo-Fallback (Multi-Modell-Sequenz)
-- Fallback auf Kontoebene (mehrere Konten pro Anbieter)
-- OAuth + API-Schlüssel-Provider-Verbindungsverwaltung
-- Einbettungsgenerierung über „/v1/embeddings“ (6 Anbieter, 9 Modelle)
-- Bildgenerierung über „/v1/images/generations“ (4 Anbieter, 9 Modelle)
-- Think-Tag-Parsing (`
...`) für Argumentationsmodelle
-- Antwortbereinigung für strikte OpenAI SDK-Kompatibilität
-- Rollennormalisierung (Entwickler→System, System→Benutzer) für anbieterübergreifende Kompatibilität
-- Strukturierte Ausgabekonvertierung (json_schema → Gemini ResponseSchema)
-- Lokale Persistenz für Anbieter, Schlüssel, Aliase, Kombinationen, Einstellungen, Preise
-- Nutzungs-/Kostenverfolgung und Anforderungsprotokollierung
-- Optionale Cloud-Synchronisierung für die Synchronisierung mehrerer Geräte/Status
-- IP-Zulassungs-/Blockierungsliste für die API-Zugriffskontrolle
-- Denken Sie an die Budgetverwaltung (Passthrough/Auto/Benutzerdefiniert/Adaptiv)
-- Sofortige Injektion des globalen Systems
-- Sitzungsverfolgung und Fingerabdruck
-- Erweiterte Ratenbegrenzung pro Konto mit anbieterspezifischen Profilen
-- Leistungsschaltermuster für die Ausfallsicherheit des Anbieters
-- Donnernder Herdenschutz mit Mutex-Sperre
- – Signaturbasierter Anforderungsdeduplizierungs-Cache
-- Domänenschicht: Modellverfügbarkeit, Kostenregeln, Fallback-Richtlinie, Sperrrichtlinie
-- Persistenz des Domänenstatus (SQLite-Durchschreibcache für Fallbacks, Budgets, Sperrungen, Leistungsschalter)
-- Richtlinien-Engine für zentralisierte Anfrageauswertung (Sperrung → Budget → Fallback)
-- Fordern Sie Telemetrie mit p50/p95/p99-Latenzaggregation an
-- Korrelations-ID (X-Request-Id) für eine durchgängige Nachverfolgung
-- Compliance-Audit-Protokollierung mit Opt-out pro API-Schlüssel
-- Evaluierungsrahmen für die LLM-Qualitätssicherung
-- Resilience-UI-Dashboard mit Echtzeit-Leistungsschalterstatus
-- Modulare OAuth-Anbieter (12 einzelne Module unter „src/lib/oauth/providers/“)
+## Executive Summary
-Primäres Laufzeitmodell:
+OmniRoute is a local AI routing gateway and dashboard built on Next.js.
+It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking.
-– Next.js-App-Routen unter „src/app/api/_“ implementieren sowohl Dashboard-APIs als auch Kompatibilitäts-APIs
-– Ein gemeinsam genutzter SSE/Routing-Kern in „src/sse/_“ + „open-sse/\*“ kümmert sich um die Ausführung, Übersetzung, Streaming, Fallback und Nutzung des Anbieters## Scope and Boundaries
+Core capabilities:
+
+- OpenAI-compatible API surface for CLI/tools (28 providers)
+- Request/response translation across provider formats
+- Model combo fallback (multi-model sequence)
+- Account-level fallback (multi-account per provider)
+- OAuth + API-key provider connection management
+- Embedding generation via `/v1/embeddings` (6 providers, 9 models)
+- Image generation via `/v1/images/generations` (4 providers, 9 models)
+- Think tag parsing (`
...`) for reasoning models
+- Response sanitization for strict OpenAI SDK compatibility
+- Role normalization (developer→system, system→user) for cross-provider compatibility
+- Structured output conversion (json_schema → Gemini responseSchema)
+- Local persistence for providers, keys, aliases, combos, settings, pricing
+- Usage/cost tracking and request logging
+- Optional cloud sync for multi-device/state sync
+- IP allowlist/blocklist for API access control
+- Thinking budget management (passthrough/auto/custom/adaptive)
+- Global system prompt injection
+- Session tracking and fingerprinting
+- Per-account enhanced rate limiting with provider-specific profiles
+- Circuit breaker pattern for provider resilience
+- Anti-thundering herd protection with mutex locking
+- Signature-based request deduplication cache
+- Domain layer: model availability, cost rules, fallback policy, lockout policy
+- Context Relay: session handoff summaries for account rotation continuity
+- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers)
+- Policy engine for centralized request evaluation (lockout → budget → fallback)
+- Request telemetry with p50/p95/p99 latency aggregation
+- Correlation ID (X-Request-Id) for end-to-end tracing
+- Compliance audit logging with opt-out per API key
+- Eval framework for LLM quality assurance
+- Resilience UI dashboard with real-time circuit breaker status
+- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`)
+
+Primary runtime model:
+
+- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs
+- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage
+
+## Scope and Boundaries
### In Scope
-- Lokale Gateway-Laufzeit
-- Dashboard-Verwaltungs-APIs
-- Anbieterauthentifizierung und Token-Aktualisierung
-- Fordern Sie Übersetzung und SSE-Streaming an
-- Lokaler Status + Nutzungspersistenz
-- Optionale Orchestrierung der Cloud-Synchronisierung### Out of Scope
+- Local gateway runtime
+- Dashboard management APIs
+- Provider authentication and token refresh
+- Request translation and SSE streaming
+- Local state + usage persistence
+- Optional cloud sync orchestration
-- Cloud-Service-Implementierung hinter „NEXT_PUBLIC_CLOUD_URL“.
-- Anbieter-SLA/Kontrollebene außerhalb des lokalen Prozesses
-- Externe CLI-Binärdateien selbst (Claude CLI, Codex CLI usw.)## Dashboard Surface (Current)
+### Out of Scope
-Hauptseiten unter „src/app/(dashboard)/dashboard/“:
+- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL`
+- Provider SLA/control plane outside local process
+- External CLI binaries themselves (Claude CLI, Codex CLI, etc.)
-- „/dashboard“ – Schnellstart + Anbieterübersicht
-- „/dashboard/endpoint“ – Endpunkt-Proxy + MCP + A2A + API-Endpunkt-Registerkarten
-- „/dashboard/providers“ – Anbieterverbindungen und Anmeldeinformationen
-- „/dashboard/combos“ – Kombinationsstrategien, Vorlagen, Modell-Routing-Regeln
-- „/dashboard/costs“ – Kostenaggregation und Preistransparenz
-- „/dashboard/analytics“ – Nutzungsanalysen und Auswertungen
-- „/dashboard/limits“ – Kontingent-/Ratenkontrolle
-- „/dashboard/cli-tools“ – CLI-Onboarding, Laufzeiterkennung, Konfigurationsgenerierung
-- „/dashboard/agents“ – erkannte ACP-Agenten + benutzerdefinierte Agentenregistrierung
-- „/dashboard/media“ – Bild-/Video-/Musikspielplatz
-- „/dashboard/search-tools“ – Tests und Verlauf des Suchanbieters
-- „/dashboard/health“ – Betriebszeit, Leistungsschalter, Ratenbegrenzungen
-- „/dashboard/logs“ – Anforderungs-/Proxy-/Audit-/Konsolenprotokolle
-- „/dashboard/settings“ – Registerkarten für Systemeinstellungen (Allgemein, Routing, Combo-Standardeinstellungen usw.)
-- „/dashboard/api-manager“ – API-Schlüssellebenszyklus und Modellberechtigungen## High-Level System Context
+## Dashboard Surface (Current)
+
+Main pages under `src/app/(dashboard)/dashboard/`:
+
+- `/dashboard` — quick start + provider overview
+- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs
+- `/dashboard/providers` — provider connections and credentials
+- `/dashboard/combos` — combo strategies, templates, model routing rules
+- `/dashboard/costs` — cost aggregation and pricing visibility
+- `/dashboard/analytics` — usage analytics and evaluations
+- `/dashboard/limits` — quota/rate controls
+- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation
+- `/dashboard/agents` — detected ACP agents + custom agent registration
+- `/dashboard/media` — image/video/music playground
+- `/dashboard/search-tools` — search provider testing and history
+- `/dashboard/health` — uptime, circuit breakers, rate limits
+- `/dashboard/logs` — request/proxy/audit/console logs
+- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.)
+- `/dashboard/api-manager` — API key lifecycle and model permissions
+
+## High-Level System Context
```mermaid
flowchart LR
@@ -129,140 +142,151 @@ flowchart LR
## 1) API and Routing Layer (Next.js App Routes)
-Hauptverzeichnisse:
+Main directories:
-- „src/app/api/v1/_“ und „src/app/api/v1beta/_“ für Kompatibilitäts-APIs
-- „src/app/api/\*“ für Verwaltungs-/Konfigurations-APIs
-- Next schreibt in „next.config.mjs“ die Zuordnung von „/v1/_“ zu „/api/v1/_“ um
+- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs
+- `src/app/api/*` for management/configuration APIs
+- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*`
-Wichtige Kompatibilitätsrouten:
+Important compatibility routes:
- `src/app/api/v1/chat/completions/route.ts`
- `src/app/api/v1/messages/route.ts`
- `src/app/api/v1/responses/route.ts`
-- „src/app/api/v1/models/route.ts“ – enthält benutzerdefinierte Modelle mit „custom: true“.
-- „src/app/api/v1/embeddings/route.ts“ – Einbettungsgenerierung (6 Anbieter)
-- `src/app/api/v1/images/generations/route.ts` — Bildgenerierung (4+ Anbieter inkl. Antigravity/Nebius)
+- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true`
+- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers)
+- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius)
- `src/app/api/v1/messages/count_tokens/route.ts`
-- „src/app/api/v1/providers/[provider]/chat/completions/route.ts“ – dedizierter Chat pro Anbieter
-- „src/app/api/v1/providers/[provider]/embeddings/route.ts“ – dedizierte Einbettungen pro Anbieter
-- „src/app/api/v1/providers/[provider]/images/generations/route.ts“ – dedizierte Bilder pro Anbieter
+- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat
+- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings
+- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images
- `src/app/api/v1beta/models/route.ts`
-- `src/app/api/v1beta/models/[...pfad]/route.ts`
+- `src/app/api/v1beta/models/[...path]/route.ts`
-Verwaltungsdomänen:
+Management domains:
-- Authentifizierung/Einstellungen: `src/app/api/auth/*`, `src/app/api/settings/*`
-- Anbieter/Verbindungen: `src/app/api/providers*`
-- Anbieterknoten: `src/app/api/provider-nodes*`
-- Benutzerdefinierte Modelle: `src/app/api/provider-models` (GET/POST/DELETE)
-- Modellkatalog: `src/app/api/models/route.ts` (GET)
-- Proxy-Konfiguration: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
+- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*`
+- Providers/connections: `src/app/api/providers*`
+- Provider nodes: `src/app/api/provider-nodes*`
+- Custom models: `src/app/api/provider-models` (GET/POST/DELETE)
+- Model catalog: `src/app/api/models/route.ts` (GET)
+- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
- OAuth: `src/app/api/oauth/*`
-- Schlüssel/Aliase/Combos/Preise: „src/app/api/keys*“, „src/app/api/models/alias“, „src/app/api/combos*“, „src/app/api/pricing“.
-- Verwendung: `src/app/api/usage/*`
-- Synchronisierung/Cloud: `src/app/api/sync/*`, `src/app/api/cloud/*`
-- CLI-Tool-Helfer: `src/app/api/cli-tools/*`
-- IP-Filter: `src/app/api/settings/ip-filter` (GET/PUT)
-- Thinking-Budget: `src/app/api/settings/thinking-budget` (GET/PUT)
-- Systemeingabeaufforderung: `src/app/api/settings/system-prompt` (GET/PUT)
-- Sitzungen: `src/app/api/sessions` (GET)
-- Ratenlimits: `src/app/api/rate-limits` (GET)
- – Resilienz: „src/app/api/resilience“ (GET/PATCH) – Anbieterprofile, Leistungsschalter, Ratengrenzstatus
-- Resilience-Reset: `src/app/api/resilience/reset` (POST) – Breaker + Abklingzeiten zurücksetzen
-- Cache-Statistiken: `src/app/api/cache/stats` (GET/DELETE)
-- Modellverfügbarkeit: `src/app/api/models/availability` (GET/POST)
-- Telemetrie: `src/app/api/telemetry/summary` (GET)
+- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing`
+- Usage: `src/app/api/usage/*`
+- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*`
+- CLI tooling helpers: `src/app/api/cli-tools/*`
+- IP filter: `src/app/api/settings/ip-filter` (GET/PUT)
+- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT)
+- System prompt: `src/app/api/settings/system-prompt` (GET/PUT)
+- Sessions: `src/app/api/sessions` (GET)
+- Rate limits: `src/app/api/rate-limits` (GET)
+- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state
+- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns
+- Cache stats: `src/app/api/cache/stats` (GET/DELETE)
+- Model availability: `src/app/api/models/availability` (GET/POST)
+- Telemetry: `src/app/api/telemetry/summary` (GET)
- Budget: `src/app/api/usage/budget` (GET/POST)
-- Fallback-Ketten: `src/app/api/fallback/chains` (GET/POST/DELETE)
-- Compliance-Audit: `src/app/api/compliance/audit-log` (GET)
-- Auswertungen: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET)
-- Richtlinien: `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core
+- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE)
+- Compliance audit: `src/app/api/compliance/audit-log` (GET)
+- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET)
+- Policies: `src/app/api/policies` (GET/POST)
-Hauptflussmodule:
+## 2) SSE + Translation Core
-- Eintrag: `src/sse/handlers/chat.ts`
-- Kernorchestrierung: „open-sse/handlers/chatCore.ts“.
-- Anbieterausführungsadapter: „open-sse/executors/\*“.
-- Formaterkennung/Anbieterkonfiguration: „open-sse/services/provider.ts“.
-- Modellanalyse/-auflösung: `src/sse/services/model.ts`, `open-sse/services/model.ts`
-- Konto-Fallback-Logik: „open-sse/services/accountFallback.ts“.
-- Übersetzungsregister: „open-sse/translator/index.ts“.
-- Stream-Transformationen: „open-sse/utils/stream.ts“, „open-sse/utils/streamHandler.ts“.
-- Nutzungsextraktion/-normalisierung: `open-sse/utils/usageTracking.ts`
-- Think-Tag-Parser: „open-sse/utils/thinkTagParser.ts“.
-- Einbettungshandler: `open-sse/handlers/embeddings.ts`
-- Einbettungsanbieter-Registrierung: „open-sse/config/embeddingRegistry.ts“.
-- Handler für die Bildgenerierung: „open-sse/handlers/imageGeneration.ts“.
-- Registrierung des Bildanbieters: „open-sse/config/imageRegistry.ts“.
-- Antwortbereinigung: `open-sse/handlers/responseSanitizer.ts`
-- Rollennormalisierung: `open-sse/services/roleNormalizer.ts`
+Main flow modules:
-Dienste (Geschäftslogik):
+- Entry: `src/sse/handlers/chat.ts`
+- Core orchestration: `open-sse/handlers/chatCore.ts`
+- Provider execution adapters: `open-sse/executors/*`
+- Format detection/provider config: `open-sse/services/provider.ts`
+- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts`
+- Account fallback logic: `open-sse/services/accountFallback.ts`
+- Translation registry: `open-sse/translator/index.ts`
+- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
+- Usage extraction/normalization: `open-sse/utils/usageTracking.ts`
+- Think tag parser: `open-sse/utils/thinkTagParser.ts`
+- Embedding handler: `open-sse/handlers/embeddings.ts`
+- Embedding provider registry: `open-sse/config/embeddingRegistry.ts`
+- Image generation handler: `open-sse/handlers/imageGeneration.ts`
+- Image provider registry: `open-sse/config/imageRegistry.ts`
+- Response sanitization: `open-sse/handlers/responseSanitizer.ts`
+- Role normalization: `open-sse/services/roleNormalizer.ts`
-- Kontoauswahl/-bewertung: `open-sse/services/accountSelector.ts`
-- Kontextlebenszyklusverwaltung: „open-sse/services/contextManager.ts“.
-- Durchsetzung des IP-Filters: „open-sse/services/ipFilter.ts“.
-- Sitzungsverfolgung: `open-sse/services/sessionManager.ts`
-- Deduplizierung anfordern: „open-sse/services/signatureCache.ts“.
-- System-Prompt-Injection: „open-sse/services/systemPrompt.ts“.
-- Thinking Budget Management: „open-sse/services/thinkingBudget.ts“.
-- Wildcard-Modell-Routing: „open-sse/services/wildcardRouter.ts“.
-- Ratenlimitverwaltung: `open-sse/services/rateLimitManager.ts`
-- Leistungsschalter: `open-sse/services/CircuitBreaker.ts`
+Services (business logic):
-Module der Domänenschicht:
+- Account selection/scoring: `open-sse/services/accountSelector.ts`
+- Context lifecycle management: `open-sse/services/contextManager.ts`
+- IP filter enforcement: `open-sse/services/ipFilter.ts`
+- Session tracking: `open-sse/services/sessionManager.ts`
+- Request deduplication: `open-sse/services/signatureCache.ts`
+- System prompt injection: `open-sse/services/systemPrompt.ts`
+- Thinking budget management: `open-sse/services/thinkingBudget.ts`
+- Wildcard model routing: `open-sse/services/wildcardRouter.ts`
+- Rate limit management: `open-sse/services/rateLimitManager.ts`
+- Circuit breaker: `open-sse/services/circuitBreaker.ts`
+- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy
+- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions
-- Modellverfügbarkeit: `src/lib/domain/modelAvailability.ts`
-- Kostenregeln/Budgets: `src/lib/domain/costRules.ts`
-- Fallback-Richtlinie: `src/lib/domain/fallbackPolicy.ts`
-- Combo-Resolver: `src/lib/domain/comboResolver.ts`
-- Sperrrichtlinie: `src/lib/domain/lockoutPolicy.ts`
- – Richtlinien-Engine: „src/domain/policyEngine.ts“ – zentralisierte Sperrung → Budget → Fallback-Auswertung
-- Fehlercodekatalog: `src/lib/domain/errorCodes.ts`
-- Anforderungs-ID: `src/lib/domain/requestId.ts`
-- Abrufzeitüberschreitung: `src/lib/domain/fetchTimeout.ts`
-- Telemetrie anfordern: `src/lib/domain/requestTelemetry.ts`
-- Compliance/Audit: `src/lib/domain/compliance/index.ts`
-- Eval-Runner: `src/lib/domain/evalRunner.ts`
- – Domänenstatus-Persistenz: „src/lib/db/domainState.ts“ – SQLite CRUD für Fallback-Ketten, Budgets, Kostenverlauf, Sperrstatus, Leistungsschalter
+Domain layer modules:
-OAuth-Provider-Module (12 einzelne Dateien unter „src/lib/oauth/providers/“):
+- Model availability: `src/lib/domain/modelAvailability.ts`
+- Cost rules/budgets: `src/lib/domain/costRules.ts`
+- Fallback policy: `src/lib/domain/fallbackPolicy.ts`
+- Combo resolver: `src/lib/domain/comboResolver.ts`
+- Lockout policy: `src/lib/domain/lockoutPolicy.ts`
+- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation
+- Error codes catalog: `src/lib/domain/errorCodes.ts`
+- Request ID: `src/lib/domain/requestId.ts`
+- Fetch timeout: `src/lib/domain/fetchTimeout.ts`
+- Request telemetry: `src/lib/domain/requestTelemetry.ts`
+- Compliance/audit: `src/lib/domain/compliance/index.ts`
+- Eval runner: `src/lib/domain/evalRunner.ts`
+- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers
-- Registrierungsindex: `src/lib/oauth/providers/index.ts`
-- Einzelne Anbieter: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
- – Thin Wrapper: „src/lib/oauth/providers.ts“ – erneuter Export aus einzelnen Modulen## 3) Persistence Layer
+OAuth provider modules (12 individual files under `src/lib/oauth/providers/`):
-Primärstatus-DB (SQLite):
+- Registry index: `src/lib/oauth/providers/index.ts`
+- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
+- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules
-- Kerninfra: `src/lib/db/core.ts` (better-sqlite3, Migrationen, WAL)
-- Fassade erneut exportieren: `src/lib/localDb.ts` (dünne Kompatibilitätsschicht für Aufrufer)
-- Datei: „${DATA_DIR}/storage.sqlite“ (oder „$XDG_CONFIG_HOME/omniroute/storage.sqlite“, wenn festgelegt, sonst „~/.omniroute/storage.sqlite“)
-- Entitäten (Tabellen + KV-Namespaces): ProviderConnections, ProviderNodes, ModelAliases, Combos, APIKeys, Einstellungen, Preise,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt**
+## 3) Persistence Layer
-Nutzungsdauer:
+Primary state DB (SQLite):
-- Fassade: `src/lib/usageDb.ts` (zerlegte Module in `src/lib/usage/*`)
-- SQLite-Tabellen in „storage.sqlite“: „usage_history“, „call_logs“, „proxy_logs“.
-- Optionale Dateiartefakte bleiben aus Kompatibilitäts-/Debuggründen erhalten (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `
/logs/...`)
- – Ältere JSON-Dateien werden durch Startmigrationen nach SQLite migriert, sofern vorhanden
+- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL)
+- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers)
+- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`)
+- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt**
+
+Usage persistence:
+
+- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`)
+- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs`
+- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`)
+- legacy JSON files are migrated to SQLite by startup migrations when present
Domain State DB (SQLite):
-– „src/lib/db/domainState.ts“ – CRUD-Operationen für den Domänenstatus
+- `src/lib/db/domainState.ts` — CRUD operations for domain state
+- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
+- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start
-- Tabellen (erstellt in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_Circuit_breakers`
-- Write-Through-Cache-Muster: In-Memory-Maps sind zur Laufzeit maßgeblich; Mutationen werden synchron zu SQLite geschrieben; Der Status wird beim Kaltstart aus der DB wiederhergestellt## 4) Auth + Security Surfaces
+## 4) Auth + Security Surfaces
-- Dashboard-Cookie-Authentifizierung: „src/proxy.ts“, „src/app/api/auth/login/route.ts“.
-- API-Schlüsselgenerierung/-überprüfung: `src/shared/utils/apiKey.ts`
- – Provider-Geheimnisse blieben in „providerConnections“-Einträgen bestehen
-- Unterstützung für ausgehende Proxys über „open-sse/utils/proxyFetch.ts“ (env vars) und „open-sse/utils/networkProxy.ts“ (pro Anbieter oder global konfigurierbar)## 5) Cloud Sync
+- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts`
+- API key generation/verification: `src/shared/utils/apiKey.ts`
+- Provider secrets persisted in `providerConnections` entries
+- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global)
-- Scheduler-Init: „src/lib/initCloudSync.ts“, „src/shared/services/initializeCloudSync.ts“, „src/shared/services/modelSyncScheduler.ts“.
-- Periodische Aufgabe: `src/shared/services/cloudSyncScheduler.ts`
-- Periodische Aufgabe: `src/shared/services/modelSyncScheduler.ts`
-- Kontrollroute: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`)
+## 5) Cloud Sync
+
+- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
+- Periodic task: `src/shared/services/cloudSyncScheduler.ts`
+- Periodic task: `src/shared/services/modelSyncScheduler.ts`
+- Control route: `src/app/api/sync/cloud/route.ts`
+
+## Request Lifecycle (`/v1/chat/completions`)
```mermaid
sequenceDiagram
@@ -339,7 +363,9 @@ flowchart TD
Q -- No --> R[Return all unavailable]
```
-Fallback-Entscheidungen werden von „open-sse/services/accountFallback.ts“ unter Verwendung von Statuscodes und Fehlermeldungsheuristiken gesteuert. Combo-Routing fügt einen zusätzlichen Schutz hinzu: 400-Fehler im Anbieterbereich wie Upstream-Inhaltsblockierungs- und Rollenvalidierungsfehler werden als modelllokale Fehler behandelt, sodass spätere Combo-Ziele weiterhin ausgeführt werden können.## OAuth Onboarding and Token Refresh Lifecycle
+Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run.
+
+## OAuth Onboarding and Token Refresh Lifecycle
```mermaid
sequenceDiagram
@@ -369,7 +395,9 @@ sequenceDiagram
Test-->>UI: validation result
```
-Die Aktualisierung während des Live-Verkehrs wird in „open-sse/handlers/chatCore.ts“ über den Executor „refreshCredentials()“ ausgeführt.## Cloud Sync Lifecycle (Enable / Sync / Disable)
+Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`.
+
+## Cloud Sync Lifecycle (Enable / Sync / Disable)
```mermaid
sequenceDiagram
@@ -401,7 +429,9 @@ sequenceDiagram
Sync-->>UI: disabled
```
-Die regelmäßige Synchronisierung wird durch „CloudSyncScheduler“ ausgelöst, wenn die Cloud aktiviert ist.## Data Model and Storage Map
+Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled.
+
+## Data Model and Storage Map
```mermaid
erDiagram
@@ -502,12 +532,14 @@ erDiagram
}
```
-Physische Speicherdateien:
+Physical storage files:
-- Primäre Laufzeit-DB: „${DATA_DIR}/storage.sqlite“.
-- Protokollzeilen anfordern: „${DATA_DIR}/log.txt“ (Kompatibilitäts-/Debug-Artefakt)
-- Strukturierte Anrufnutzlastarchive: „${DATA_DIR}/call_logs/“.
-- optionale Übersetzer-/Request-Debug-Sitzungen: `/logs/...`## Deployment Topology
+- primary runtime DB: `${DATA_DIR}/storage.sqlite`
+- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact)
+- structured call payload archives: `${DATA_DIR}/call_logs/`
+- optional translator/request debug sessions: `/logs/...`
+
+## Deployment Topology
```mermaid
flowchart LR
@@ -542,205 +574,249 @@ flowchart LR
### Route and API Modules
-- `src/app/api/v1/*`, `src/app/api/v1beta/*`: Kompatibilitäts-APIs
-- `src/app/api/v1/providers/[provider]/*`: dedizierte Routen pro Anbieter (Chat, Einbettungen, Bilder)
-- „src/app/api/providers\*“: Anbieter-CRUD, Validierung, Tests
-- „src/app/api/provider-nodes\*“: benutzerdefinierte kompatible Knotenverwaltung
-- „src/app/api/provider-models“: benutzerdefinierte Modellverwaltung (CRUD)
-- `src/app/api/models/route.ts`: Modellkatalog-API (Aliase + benutzerdefinierte Modelle)
-- `src/app/api/oauth/*`: OAuth/Gerätecodeflüsse
-- „src/app/api/keys\*“: Lebenszyklus des lokalen API-Schlüssels
-- `src/app/api/models/alias`: Alias-Verwaltung
-- `src/app/api/combos*`: Fallback-Combo-Verwaltung
-- „src/app/api/pricing“: Preisüberschreibungen für die Kostenberechnung
-- `src/app/api/settings/proxy`: Proxy-Konfiguration (GET/PUT/DELETE)
-- „src/app/api/settings/proxy/test“: Test der ausgehenden Proxy-Konnektivität (POST)
-- `src/app/api/usage/*`: Nutzungs- und Protokoll-APIs
-- `src/app/api/sync/*` + `src/app/api/cloud/*`: Cloud-Synchronisierung und Cloud-orientierte Helfer
-- `src/app/api/cli-tools/*`: lokale CLI-Konfigurationsschreiber/-prüfer
-- `src/app/api/settings/ip-filter`: IP-Zulassungsliste/Blockliste (GET/PUT)
-- `src/app/api/settings/thinking-budget`: Thinking-Token-Budgetkonfiguration (GET/PUT)
-- `src/app/api/settings/system-prompt`: globale Systemeingabeaufforderung (GET/PUT)
-- `src/app/api/sessions`: Auflistung der aktiven Sitzungen (GET)
-- `src/app/api/rate-limits`: Status des Ratenlimits pro Konto (GET)### Routing and Execution Core
+- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs
+- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images)
+- `src/app/api/providers*`: provider CRUD, validation, testing
+- `src/app/api/provider-nodes*`: custom compatible node management
+- `src/app/api/provider-models`: custom model management (CRUD)
+- `src/app/api/models/route.ts`: model catalog API (aliases + custom models)
+- `src/app/api/oauth/*`: OAuth/device-code flows
+- `src/app/api/keys*`: local API key lifecycle
+- `src/app/api/models/alias`: alias management
+- `src/app/api/combos*`: fallback combo management
+- `src/app/api/pricing`: pricing overrides for cost calculation
+- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE)
+- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST)
+- `src/app/api/usage/*`: usage and logs APIs
+- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers
+- `src/app/api/cli-tools/*`: local CLI config writers/checkers
+- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT)
+- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT)
+- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT)
+- `src/app/api/sessions`: active session listing (GET)
+- `src/app/api/rate-limits`: per-account rate limit status (GET)
-- „src/sse/handlers/chat.ts“: Anforderungsanalyse, Kombinationsverarbeitung, Kontoauswahlschleife
-- „open-sse/handlers/chatCore.ts“: Übersetzung, Executor-Dispatch, Wiederholungs-/Aktualisierungsbehandlung, Stream-Setup
-- „open-sse/executors/\*“: anbieterspezifisches Netzwerk- und Formatverhalten### Translation Registry and Format Converters
+### Routing and Execution Core
-- „open-sse/translator/index.ts“: Übersetzerregistrierung und Orchestrierung
-- Übersetzer anfordern: `open-sse/translator/request/*`
-- Antwortübersetzer: `open-sse/translator/response/*`
-- Formatkonstanten: „open-sse/translator/formats.ts“.### Persistence
+- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop
+- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup
+- `open-sse/executors/*`: provider-specific network and format behavior
-- `src/lib/db/*`: persistente Konfiguration/Status und Domänenpersistenz auf SQLite
-- `src/lib/localDb.ts`: Kompatibilitäts-Neuexport für DB-Module
-- „src/lib/usageDb.ts“: Fassade der Nutzungshistorie/Anrufprotokolle über SQLite-Tabellen## Provider Executor Coverage (Strategy Pattern)
+### Translation Registry and Format Converters
-Jeder Anbieter verfügt über einen speziellen Executor, der „BaseExecutor“ (in „open-sse/executors/base.ts“) erweitert und URL-Erstellung, Header-Konstruktion, Wiederholung mit exponentiellem Backoff, Hooks für die Aktualisierung von Anmeldeinformationen und die Orchestrierungsmethode „execute()“ bereitstellt.
+- `open-sse/translator/index.ts`: translator registry and orchestration
+- Request translators: `open-sse/translator/request/*`
+- Response translators: `open-sse/translator/response/*`
+- Format constants: `open-sse/translator/formats.ts`
-| Testamentsvollstrecker | Anbieter(n) | Besondere Handhabung |
-| ---------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------------------------------------------- |
-| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamische URL-/Header-Konfiguration pro Anbieter |
-| `AntigravityExecutor` | Google Antigravitation | Benutzerdefinierte Projekt-/Sitzungs-IDs, Wiederholen nach dem Parsen |
-| `CodexExecutor` | OpenAI-Codex | Fügt Systemanweisungen ein und erzwingt den Denkaufwand |
-| `CursorExecutor` | Cursor-IDE | ConnectRPC-Protokoll, Protobuf-Kodierung, Anforderungssignatur über Prüfsumme |
-| `GithubExecutor` | GitHub-Copilot | Copilot-Token-Aktualisierung, VSCode-imitierende Header |
-| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream-Binärformat → SSE-Konvertierung |
-| `GeminiCLIExecutor` | Gemini CLI | Aktualisierungszyklus des Google OAuth-Tokens |
+### Persistence
-Alle anderen Anbieter (einschließlich benutzerdefinierter kompatibler Knoten) verwenden den „DefaultExecutor“.## Provider Compatibility Matrix
+- `src/lib/db/*`: persistent config/state and domain persistence on SQLite
+- `src/lib/localDb.ts`: compatibility re-export for DB modules
+- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables
-| Anbieter | Formatieren | Authentifizierung | Stream | Nicht-Stream | Token-Aktualisierung | Nutzungs-API |
-| ---------------- | ---------------- | ---------------------------- | ---------------- | ------------ | -------------------- | ------------------------------ | ------------------------------ |
-| Claude | Claude | API-Schlüssel / OAuth | ✅ | ✅ | ✅ | ⚠️ Nur Administrator |
-| Zwillinge | Zwillinge | API-Schlüssel / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud-Konsole |
-| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud-Konsole |
-| Antigravitation | Antigravitation | OAuth | ✅ | ✅ | ✅ | ✅ Vollständige Kontingent-API |
-| OpenAI | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ |
-| Kodex | Openai-Antworten | OAuth | ✅ gezwungen | ❌ | ✅ | ✅ Tariflimits |
-| GitHub-Copilot | openai | OAuth + Copilot-Token | ✅ | ✅ | ✅ | ✅ Kontingent-Snapshots |
-| Cursor | Cursor | Benutzerdefinierte Prüfsumme | ✅ | ✅ | ❌ | ❌ |
-| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Nutzungsbeschränkungen |
-| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Auf Anfrage |
-| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Auf Anfrage |
-| OpenRouter | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ |
-| GLM/Kimi/MiniMax | Claude | API-Schlüssel | ✅ | ✅ | ❌ | ❌ |
-| DeepSeek | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ |
-| Groq | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ |
-| xAI (Grok) | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ |
-| Mistral | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ |
-| Ratlosigkeit | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ |
-| Zusammen KI | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ |
-| Feuerwerk KI | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ |
-| Großhirn | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ |
-| Kohärent | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ |
-| NVIDIA NIM | openai | API-Schlüssel | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage |
+## Provider Executor Coverage (Strategy Pattern)
-Zu den erkannten Quellformaten gehören:
+Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method.
+
+| Executor | Provider(s) | Special Handling |
+| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- |
+| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider |
+| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing |
+| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort |
+| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum |
+| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers |
+| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion |
+| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle |
+
+All other providers (including custom compatible nodes) use the `DefaultExecutor`.
+
+## Provider Compatibility Matrix
+
+| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API |
+| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ |
+| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only |
+| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API |
+| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits |
+| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots |
+| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ |
+| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits |
+| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request |
+| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request |
+| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ |
+| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+
+## Format Translation Coverage
+
+Detected source formats include:
- `openai`
-- „Openai-Antworten“.
-- `Claude`
-- „Zwillinge“.
+- `openai-responses`
+- `claude`
+- `gemini`
-Zu den Zielformaten gehören:
+Target formats include:
-- OpenAI-Chat/Antworten
+- OpenAI chat/Responses
- Claude
-- Gemini/Gemini-CLI/Antigravity-Umschlag
+- Gemini/Gemini-CLI/Antigravity envelope
- Kiro
- Cursor
-Übersetzungen verwenden**OpenAI als Hub-Format**– alle Konvertierungen durchlaufen OpenAI als Zwischenformat:```
+Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate:
+
+```
Source Format → OpenAI (hub) → Target Format
+```
-````
+Translations are selected dynamically based on source payload shape and provider target format.
-Übersetzungen werden dynamisch basierend auf der Form der Quellnutzlast und dem Zielformat des Anbieters ausgewählt.
+Additional processing layers in the translation pipeline:
-Zusätzliche Verarbeitungsebenen in der Übersetzungspipeline:
+- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance
+- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE)
+- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field
+- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema`
--**Antwortbereinigung**– Entfernt nicht standardmäßige Felder aus Antworten im OpenAI-Format (sowohl Streaming als auch Nicht-Streaming), um eine strikte SDK-Konformität sicherzustellen
--**Rollennormalisierung**– Konvertiert „Entwickler“ → „System“ für Nicht-OpenAI-Ziele; führt „System“ → „Benutzer“ für Modelle zusammen, die die Systemrolle ablehnen (GLM, ERNIE)
--**Think-Tag-Extraktion**– Analysiert „...“-Blöcke aus dem Inhalt in das Feld „reasoning_content“.
--**Strukturierte Ausgabe**– Konvertiert OpenAI „response_format.json_schema“ in „responseMimeType“ + „responseSchema“ von Gemini## Supported API Endpoints
+## Supported API Endpoints
-| Endpunkt | Formatieren | Handler |
-| ------------------------------------------------- | ------------------- | ------------------------------------------------------------------- |
-| `POST /v1/chat/completions` | OpenAI-Chat | `src/sse/handlers/chat.ts` |
-| `POST /v1/messages` | Claude-Nachrichten | Gleicher Handler (automatisch erkannt) |
-| `POST /v1/responses` | OpenAI-Antworten | `open-sse/handlers/responsesHandler.ts` |
-| `POST /v1/embeddings` | OpenAI-Einbettungen | `open-sse/handlers/embeddings.ts` |
-| `GET /v1/embeddings` | Modellliste | API-Route |
-| `POST /v1/images/generations` | OpenAI-Bilder | `open-sse/handlers/imageGeneration.ts` |
-| `GET /v1/images/generations` | Modellliste | API-Route |
-| `POST /v1/providers/{provider}/chat/completions` | OpenAI-Chat | Dedizierter pro Anbieter mit Modellvalidierung |
-| `POST /v1/providers/{provider}/embeddings` | OpenAI-Einbettungen | Dedizierter pro Anbieter mit Modellvalidierung |
-| `POST /v1/providers/{provider}/images/generations` | OpenAI-Bilder | Dedizierter pro Anbieter mit Modellvalidierung |
-| `POST /v1/messages/count_tokens` | Claude Token Count | API-Route |
-| `GET /v1/models` | Liste der OpenAI-Modelle | API-Route (Chat + Einbettung + Bild + benutzerdefinierte Modelle) |
-| `GET /api/models/catalog` | Katalog | Alle Modelle gruppiert nach Anbieter + Typ |
-| `POST /v1beta/models/*:streamGenerateContent` | Zwillinge heimisch | API-Route |
-| `GET/PUT/DELETE /api/settings/proxy` | Proxy-Konfiguration | Netzwerk-Proxy-Konfiguration |
-| `POST /api/settings/proxy/test` | Proxy-Konnektivität | Proxy-Zustands-/Konnektivitätstest-Endpunkt |
-| `GET/POST/DELETE /api/provider-models` | Anbietermodelle | Metadaten des Anbietermodells, die benutzerdefinierte und verwaltete verfügbare Modelle unterstützen |## Bypass Handler
+| Endpoint | Format | Handler |
+| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- |
+| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` |
+| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) |
+| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` |
+| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` |
+| `GET /v1/embeddings` | Model listing | API route |
+| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` |
+| `GET /v1/images/generations` | Model listing | API route |
+| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation |
+| `POST /v1/messages/count_tokens` | Claude Token Count | API route |
+| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) |
+| `GET /api/models/catalog` | Catalog | All models grouped by provider + type |
+| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route |
+| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration |
+| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint |
+| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models |
-Der Bypass-Handler („open-sse/utils/bypassHandler.ts“) fängt bekannte „Wegwerf“-Anfragen von Claude CLI ab – Warmup-Pings, Titelextraktionen und Token-Zählungen – und gibt eine**falsche Antwort**zurück, ohne Upstream-Anbieter-Tokens zu verbrauchen. Dies wird nur ausgelöst, wenn „User-Agent“ „claude-cli“ enthält.## Request Logger Pipeline
+## Bypass Handler
-Der Anforderungslogger („open-sse/utils/requestLogger.ts“) bietet eine 7-stufige Debug-Protokollierungspipeline, die standardmäßig deaktiviert und über „ENABLE_REQUEST_LOGS=true“ aktiviert ist:```
+The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`.
+
+## Request Logger Pipeline
+
+The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`:
+
+```
1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json
→ 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt
-````
+```
-Dateien werden für jede Anforderungssitzung in „/logs//“ geschrieben.## Failure Modes and Resilience
+Files are written to `/logs//` for each request session.
+
+## Failure Modes and Resilience
## 1) Account/Provider Availability
-- Abklingzeit des Anbieterkontos bei vorübergehenden/Raten-/Authentifizierungsfehlern
-- Konto-Fallback vor fehlgeschlagener Anfrage
-- Combo-Modell-Fallback, wenn der aktuelle Modell-/Anbieterpfad erschöpft ist## 2) Token Expiry
+- provider account cooldown on transient/rate/auth errors
+- account fallback before failing request
+- combo model fallback when current model/provider path is exhausted
-- Vorabprüfung und Aktualisierung mit erneutem Versuch für aktualisierbare Anbieter
- – 401/403-Wiederholungsversuch nach Aktualisierungsversuch im Kernpfad## 3) Stream Safety
+## 2) Token Expiry
-- Trennungsfähiger Stream-Controller
-- Übersetzungsstream mit Stream-Ende-Flush und „[FERTIG]“-Behandlung
-- Fallback der Nutzungsschätzung, wenn Metadaten zur Anbieternutzung fehlen## 4) Cloud Sync Degradation
+- pre-check and refresh with retry for refreshable providers
+- 401/403 retry after refresh attempt in core path
-– Synchronisierungsfehler werden angezeigt, die lokale Laufzeit wird jedoch fortgesetzt
-– Der Scheduler verfügt über eine wiederholfähige Logik, aber die regelmäßige Ausführung ruft derzeit standardmäßig eine Einzelversuchssynchronisierung auf## 5) Data Integrity
+## 3) Stream Safety
-- SQLite-Schemamigrationen und automatische Upgrade-Hooks beim Start
-- Legacy-JSON → SQLite-Migrationskompatibilitätspfad## Observability and Operational Signals
+- disconnect-aware stream controller
+- translation stream with end-of-stream flush and `[DONE]` handling
+- usage estimation fallback when provider usage metadata is missing
-Quellen für die Laufzeitsichtbarkeit:
+## 4) Cloud Sync Degradation
-- Konsolenprotokolle von „src/sse/utils/logger.ts“.
-- Nutzungsaggregate pro Anfrage in SQLite („usage_history“, „call_logs“, „proxy_logs“)
-- Vierstufige detaillierte Nutzlasterfassungen in SQLite (`request_detail_logs`), wenn `settings.detailed_logs_enabled=true`
-- Statusprotokoll der Textanfrage in „log.txt“ (optional/kompatibel)
-- optionale tiefe Anforderungs-/Übersetzungsprotokolle unter „logs/“, wenn „ENABLE_REQUEST_LOGS=true“ ist
-- Dashboard-Nutzungsendpunkte (`/api/usage/*`) für die UI-Nutzung
+- sync errors are surfaced but local runtime continues
+- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default
-Die detaillierte Anforderungsnutzlasterfassung speichert bis zu vier JSON-Nutzlaststufen pro weitergeleitetem Anruf:
+## 5) Data Integrity
-- Rohanfrage vom Client erhalten
-- Die übersetzte Anfrage wurde tatsächlich an den Upstream gesendet
-- Anbieterantwort als JSON rekonstruiert; Gestreamte Antworten werden zur endgültigen Zusammenfassung plus Stream-Metadaten komprimiert
- – endgültige Client-Antwort, die von OmniRoute zurückgegeben wird; Gestreamte Antworten werden in derselben kompakten Zusammenfassungsform gespeichert## Security-Sensitive Boundaries
+- SQLite schema migrations and auto-upgrade hooks at startup
+- legacy JSON → SQLite migration compatibility path
-- JWT-Geheimnis („JWT_SECRET“) sichert die Überprüfung/Signatur von Dashboard-Sitzungscookies
- – Der anfängliche Passwort-Bootstrap („INITIAL_PASSWORD“) sollte explizit für die erstmalige Bereitstellung konfiguriert werden
-- Das HMAC-Geheimnis des API-Schlüssels („API_KEY_SECRET“) sichert das generierte lokale API-Schlüsselformat
- – Anbietergeheimnisse (API-Schlüssel/Tokens) werden in der lokalen Datenbank gespeichert und sollten auf Dateisystemebene geschützt werden
- – Cloud-Synchronisierungsendpunkte basieren auf der API-Schlüsselauthentifizierung und der Maschinen-ID-Semantik## Environment and Runtime Matrix
+## Observability and Operational Signals
-Vom Code aktiv verwendete Umgebungsvariablen:
+Runtime visibility sources:
-- App/Auth: „JWT_SECRET“, „INITIAL_PASSWORD“.
-- Speicher: `DATA_DIR`
-- Kompatibles Knotenverhalten: „ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE“.
-- Optionale Speicherbasisüberschreibung (Linux/macOS, wenn „DATA_DIR“ nicht festgelegt ist): „XDG_CONFIG_HOME“.
-- Sicherheits-Hashing: „API_KEY_SECRET“, „MACHINE_ID_SALT“.
-- Protokollierung: „ENABLE_REQUEST_LOGS“.
-- Synchronisierungs-/Cloud-URLing: „NEXT_PUBLIC_BASE_URL“, „NEXT_PUBLIC_CLOUD_URL“.
-- Ausgehender Proxy: „HTTP_PROXY“, „HTTPS_PROXY“, „ALL_PROXY“, „NO_PROXY“ und Varianten in Kleinbuchstaben
-- SOCKS5-Funktionsflags: „ENABLE_SOCKS5_PROXY“, „NEXT_PUBLIC_ENABLE_SOCKS5_PROXY“.
-- Plattform-/Laufzeit-Helfer (keine App-spezifische Konfiguration): „APPDATA“, „NODE_ENV“, „PORT“, „HOSTNAME“.## Known Architectural Notes
+- console logs from `src/sse/utils/logger.ts`
+- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`)
+- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true`
+- textual request status log in `log.txt` (optional/compat)
+- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true`
+- dashboard usage endpoints (`/api/usage/*`) for UI consumption
-1. „usageDb“ und „localDb“ verwenden dieselbe Basisverzeichnisrichtlinie („DATA_DIR“ -> „XDG_CONFIG_HOME/omniroute“ -> „~/.omniroute“) bei der Migration älterer Dateien.
-2. „/api/v1/route.ts“ delegiert an denselben einheitlichen Katalog-Builder, der von „/api/v1/models“ verwendet wird (`src/app/api/v1/models/catalog.ts`), um semantische Abweichungen zu vermeiden.
-3. Der Anforderungslogger schreibt bei Aktivierung vollständige Header/Textkörper. Behandeln Sie das Protokollverzeichnis als vertraulich.
-4. Das Cloud-Verhalten hängt von der korrekten „NEXT_PUBLIC_BASE_URL“ und der Erreichbarkeit des Cloud-Endpunkts ab.
-5. Das Verzeichnis „open-sse/“ wird als „@omniroute/open-sse“**npm-Workspace-Paket**veröffentlicht. Der Quellcode importiert es über „@omniroute/open-sse/...“ (aufgelöst durch Next.js „transpilePackages“). Dateipfade in diesem Dokument verwenden aus Konsistenzgründen weiterhin den Verzeichnisnamen „open-sse/“.
-6. Diagramme im Dashboard verwenden**Recharts**(SVG-basiert) für zugängliche, interaktive Analysevisualisierungen (Modellnutzungs-Balkendiagramme, Anbieteraufschlüsselungstabellen mit Erfolgsquoten).
-7. E2E-Tests verwenden**Playwright**(`tests/e2e/`) und werden über `npm run test:e2e` ausgeführt. Unit-Tests verwenden**Node.js Test Runner**(`tests/unit/`) und werden über `npm run test:unit` ausgeführt. Der Quellcode unter „src/“ ist**TypeScript**(`.ts`/`.tsx`); Der `open-sse/`-Arbeitsbereich bleibt JavaScript (`.js`).
-8. Die Einstellungsseite ist in 5 Registerkarten unterteilt: Sicherheit, Routing (6 globale Strategien: Fill-First, Round-Robin, P2C, Random, Least-Used, Cost-Optimized), Resilience (bearbeitbare Ratenlimits, Leistungsschalter, Richtlinien), AI (Thinking Budget, System Prompt, Prompt Cache), Advanced (Proxy).## Operational Verification Checklist
+Detailed request payload capture stores up to four JSON payload stages per routed call:
-- Aus der Quelle erstellen: „npm run build“.
-- Docker-Image erstellen: `docker build -t omniroute .`
-- Starten Sie den Dienst und überprüfen Sie:
+- raw request received from the client
+- translated request actually sent upstream
+- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata
+- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form
+
+## Security-Sensitive Boundaries
+
+- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing
+- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning
+- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format
+- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level
+- Cloud sync endpoints rely on API key auth + machine id semantics
+
+## Environment and Runtime Matrix
+
+Environment variables actively used by code:
+
+- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD`
+- Storage: `DATA_DIR`
+- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
+- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME`
+- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT`
+- Logging: `ENABLE_REQUEST_LOGS`
+- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
+- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants
+- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
+- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`
+
+## Known Architectural Notes
+
+1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration.
+2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift.
+3. Request logger writes full headers/body when enabled; treat log directory as sensitive.
+4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability.
+5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency.
+6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates).
+7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`).
+8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy).
+9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed.
+10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22.
+11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible.
+
+## Operational Verification Checklist
+
+- Build from source: `npm run build`
+- Build Docker image: `docker build -t omniroute .`
+- Start service and verify:
- `GET /api/settings`
- `GET /api/v1/models`
- – Die Basis-URL des CLI-Ziels sollte „http://:20128/v1“ lauten, wenn „PORT=20128“.
+- CLI target base URL should be `http://:20128/v1` when `PORT=20128`
diff --git a/docs/i18n/de/docs/FEATURES.md b/docs/i18n/de/docs/FEATURES.md
index 926bd473db..d76662f8aa 100644
--- a/docs/i18n/de/docs/FEATURES.md
+++ b/docs/i18n/de/docs/FEATURES.md
@@ -4,102 +4,168 @@
---
-Visuelle Anleitung zu jedem Abschnitt des OmniRoute-Dashboards.---
+
+
+Visual guide to every section of the OmniRoute dashboard.
+
+---
## 🔌 Providers
-Verwalten Sie KI-Anbieterverbindungen: OAuth-Anbieter (Claude Code, Codex, Gemini CLI), API-Schlüsselanbieter (Groq, DeepSeek, OpenRouter) und kostenlose Anbieter (Qoder, Qwen, Kiro). Bei Kiro-Konten ist die Nachverfolgung des Guthabens möglich – verbleibende Guthaben, Gesamtguthaben und Verlängerungsdatum sind im Dashboard → Nutzung sichtbar.
+Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage.
+
+
---
## 🎨 Combos
-Erstellen Sie Modell-Routing-Kombinationen mit 6 Strategien: Priorität, gewichtet, Round-Robin, zufällig, am wenigsten verwendet und kostenoptimiert. Jede Kombination verkettet mehrere Modelle mit automatischem Fallback und umfasst schnelle Vorlagen und Bereitschaftsprüfungen.
+Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks.
+
+
---
## 📊 Analytics
-Umfassende Nutzungsanalysen mit Token-Verbrauch, Kostenschätzungen, Aktivitäts-Heatmaps, wöchentlichen Verteilungsdiagrammen und Aufschlüsselungen pro Anbieter.
+Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns.
+
+
---
## 🏥 System Health
-Echtzeitüberwachung: Betriebszeit, Speicher, Version, Latenzperzentile (p50/p95/p99), Cache-Statistiken und Leistungsschalterzustände des Anbieters.
+Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states.
+
+
---
## 🔧 Translator Playground
-Vier Modi zum Debuggen von API-Übersetzungen:**Playground**(Formatkonverter),**Chat Tester**(Live-Anfragen),**Test Bench**(Batch-Tests) und**Live Monitor**(Echtzeit-Stream).
+Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream).
+
+
---
## 🎮 Model Playground _(v2.0.9+)_
-Testen Sie jedes Modell direkt vom Dashboard aus. Wählen Sie Anbieter, Modell und Endpunkt aus, schreiben Sie Eingabeaufforderungen mit Monaco Editor, streamen Sie Antworten in Echtzeit, brechen Sie mitten im Stream ab und sehen Sie sich Timing-Metriken an.---
+Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics.
+
+---
## 🎨 Themes _(v2.0.5+)_
-Anpassbare Farbthemen für das gesamte Dashboard. Wählen Sie aus 7 voreingestellten Farben (Koralle, Blau, Rot, Grün, Violett, Orange, Cyan) oder erstellen Sie ein individuelles Design, indem Sie eine beliebige Hex-Farbe auswählen. Unterstützt Hell-, Dunkel- und Systemmodus.---
+Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode.
+
+---
## ⚙️ Settings
-Umfangreiches Einstellungsfeld mit Registerkarten:
+Comprehensive settings panel with tabs:
--**Allgemein**– Systemspeicher, Backup-Management (Datenbank exportieren/importieren) -**Erscheinungsbild**– Themenauswahl (Dunkel/Hell/System), Voreinstellungen für Farbthemen und benutzerdefinierte Farben, Sichtbarkeit des Gesundheitsprotokolls, Steuerelemente für die Sichtbarkeit von Elementen in der Seitenleiste -**Sicherheit**– API-Endpunktschutz, benutzerdefinierte Anbieterblockierung, IP-Filterung, Sitzungsinformationen -**Routing**– Modellaliase, Verschlechterung der Hintergrundaufgabe -**Resilienz**– Persistenz der Ratenbegrenzung, Leistungsschalter-Optimierung, automatische Deaktivierung gesperrter Konten, Überwachung des Anbieterablaufs -**Erweitert**– Konfigurationsüberschreibungen, Konfigurations-Audit-Trail, Fallback-Verschlechterungsmodus
+- **General** — System storage, backup management (export/import database)
+- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls
+- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info
+- **Routing** — Model aliases, background task degradation
+- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration
+- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode
+
+
---
## 🔧 CLI Tools
-Ein-Klick-Konfiguration für KI-Codierungstools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor und Factory Droid. Bietet automatisches Anwenden/Zurücksetzen der Konfiguration, Verbindungsprofile und Modellzuordnung.
+One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping.
+
+
---
## 🤖 CLI Agents _(v2.0.11+)_
-Dashboard zum Erkennen und Verwalten von CLI-Agenten. Zeigt ein Raster mit 14 integrierten Agenten (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) mit:
+Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with:
--**Installationsstatus**– Installiert/Nicht gefunden mit Versionserkennung -**Protokollabzeichen**– stdio, HTTP usw. -**Benutzerdefinierte Agents**– Registrieren Sie jedes CLI-Tool über ein Formular (Name, Binärdatei, Versionsbefehl, Spawn-Argumente). -**CLI-Fingerabdruck-Abgleich**– Umschalten pro Anbieter, um native CLI-Anfragesignaturen abzugleichen, wodurch das Verbotsrisiko verringert und gleichzeitig die Proxy-IP erhalten bleibt---
+- **Installation status** — Installed / Not Found with version detection
+- **Protocol badges** — stdio, HTTP, etc.
+- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args)
+- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP
+
+---
+
+## 🔗 Context Relay _(v3.5.5+)_
+
+A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context.
+
+Configurable via combo-level or global settings:
+- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%)
+- **Max Messages For Summary** — How much recent history to condense
+- **Summary Model** — Optional override model for generating the handoff summary
+
+Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md).
+
+---
+
+## 🛡️ Proxy Hardening _(v3.5.5+)_
+
+Comprehensive proxy configuration enforcement across the entire request pipeline:
+
+- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments
+- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings
+- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22
+- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS
+
+---
## 🖼️ Media _(v2.0.3+)_
-Generieren Sie Bilder, Videos und Musik über das Dashboard. Unterstützt OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open und MusicGen.---
+Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen.
+
+---
## 📝 Request Logs
-Echtzeit-Anfrageprotokollierung mit Filterung nach Anbieter, Modell, Konto und API-Schlüssel. Zeigt Statuscodes, Token-Nutzung, Latenz und Antwortdetails an.
+Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details.
+
+
---
## 🌐 API Endpoint
-Ihr einheitlicher API-Endpunkt mit Aufschlüsselung der Funktionen: Chat-Abschlüsse, Antwort-API, Einbettungen, Bildgenerierung, Neuranking, Audiotranskription, Text-to-Speech, Moderationen und registrierte API-Schlüssel. Cloudflare Quick Tunnel-Integration und Cloud-Proxy-Unterstützung für Fernzugriff.
+Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access.
+
+
---
## 🔑 API Key Management
-API-Schlüssel erstellen, festlegen und widerrufen. Jeder Schlüssel kann auf bestimmte Modelle/Anbieter mit Vollzugriff oder Nur-Lese-Berechtigungen beschränkt werden. Visuelle Schlüsselverwaltung mit Nutzungsverfolgung.---
+Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking.
+
+---
## 📋 Audit Log
-Verwaltungsaktionsverfolgung mit Filterung nach Aktionstyp, Akteur, Ziel, IP-Adresse und Zeitstempel. Vollständiger Sicherheitsereignisverlauf.---
+Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history.
+
+---
## 🖥️ Desktop Application
-Native Electron-Desktop-App für Windows, macOS und Linux. Führen Sie OmniRoute als eigenständige Anwendung mit Taskleistenintegration, Offline-Unterstützung, automatischer Aktualisierung und Installation mit einem Klick aus.
+Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install.
-Hauptmerkmale:
+Key features:
-- Abfrage der Serverbereitschaft (kein leerer Bildschirm beim Kaltstart)
-- Taskleiste mit Portverwaltung
-- Inhaltssicherheitsrichtlinie
-- Einzelinstanzsperre
-- Automatische Aktualisierung beim Neustart
-- Plattformabhängige Benutzeroberfläche (Ampeln für macOS, Standardtitelleiste für Windows/Linux)
-- Hardened Electron Build-Paketierung – symbolisch verknüpfte „node_modules“ im Standalone-Bundle werden vor dem Paketieren erkannt und abgelehnt, wodurch eine Laufzeitabhängigkeit von der Build-Maschine verhindert wird (v2.5.5+)
+- Server readiness polling (no blank screen on cold start)
+- System tray with port management
+- Content Security Policy
+- Single-instance lock
+- Auto-update on restart
+- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar)
+- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+)
-📖 Die vollständige Dokumentation finden Sie unter [`electron/README.md`](../electron/README.md).
+📖 See [`electron/README.md`](../electron/README.md) for full documentation.
diff --git a/docs/i18n/de/docs/TROUBLESHOOTING.md b/docs/i18n/de/docs/TROUBLESHOOTING.md
index c1dcb822a2..8dfa2dffff 100644
--- a/docs/i18n/de/docs/TROUBLESHOOTING.md
+++ b/docs/i18n/de/docs/TROUBLESHOOTING.md
@@ -4,68 +4,142 @@
---
-Häufige Probleme und Lösungen für OmniRoute.---
+
+
+Common problems and solutions for OmniRoute.
+
+---
## Quick Fixes
-| Problem | Lösung |
-| ------------------------------------------ | ------------------------------------------------------------------------------------- | --- |
-| Erster Login funktioniert nicht | Legen Sie „INITIAL_PASSWORD“ in „.env“ fest (keine fest codierte Standardeinstellung) |
-| Dashboard wird am falschen Port geöffnet | Setzen Sie „PORT=20128“ und „NEXT_PUBLIC_BASE_URL=http://localhost:20128“ |
-| Keine Anforderungsprotokolle unter „logs/“ | Setzen Sie „ENABLE_REQUEST_LOGS=true“ |
-| EACCES: Berechtigung verweigert | Setzen Sie „DATA_DIR=/path/to/writable/dir“, um „~/.omniroute“ zu überschreiben |
-| Routing-Strategie wird nicht gespeichert | Update auf v1.4.11+ (Zod-Schema-Korrektur für Einstellungspersistenz) | --- |
+| Problem | Solution |
+| ----------------------------- | ------------------------------------------------------------------ |
+| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) |
+| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
+| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` |
+| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` |
+| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) |
+| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below |
+| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below |
+
+---
+
+## Node.js Compatibility
+
+
+
+### Login page crashes or shows "Module self-registration" error
+
+**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database.
+
+**Symptoms:**
+- Login page shows a blank screen or a server error
+- Console shows `Error: Module did not self-register` or similar native binding errors
+- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected
+
+**Fix:**
+
+1. Install Node.js 22 LTS (recommended):
+ ```bash
+ nvm install 22
+ nvm use 22
+ ```
+2. Verify your version: `node --version` should show `v22.x.x`
+3. Reinstall OmniRoute: `npm install -g omniroute`
+4. Restart: `omniroute`
+
+> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**.
+
+---
+
+## Proxy Issues
+
+
+
+### Provider validation shows "fetch failed"
+
+**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing.
+
+**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically.
+
+### Token health check fails with "fetch failed"
+
+**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection.
+
+**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+.
+
+### SOCKS5 proxy returns "invalid onRequestStart method"
+
+**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation.
+
+**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+.
+
+---
## Provider Issues
### "Language model did not provide messages"
-**Ursache:**Anbieterkontingent erschöpft.
+**Cause:** Provider quota exhausted.
**Fix:**
-1. Überprüfen Sie den Quoten-Tracker im Dashboard
-2. Verwenden Sie eine Kombination mit Fallback-Stufen
-3. Wechseln Sie zum günstigeren/kostenlosen Tarif### Rate Limiting
+1. Check dashboard quota tracker
+2. Use a combo with fallback tiers
+3. Switch to cheaper/free tier
-**Ursache:**Das Abonnementkontingent ist erschöpft.
+### Rate Limiting
+
+**Cause:** Subscription quota exhausted.
**Fix:**
-- Fallback hinzufügen: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-- Verwenden Sie GLM/MiniMax als günstiges Backup### OAuth Token Expired
+- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Use GLM/MiniMax as cheap backup
-OmniRoute aktualisiert Token automatisch. Wenn die Probleme weiterhin bestehen:
+### OAuth Token Expired
-1. Dashboard → Anbieter → Erneut verbinden
-2. Löschen Sie die Provider-Verbindung und fügen Sie sie erneut hinzu---
+OmniRoute auto-refreshes tokens. If issues persist:
+
+1. Dashboard → Provider → Reconnect
+2. Delete and re-add the provider connection
+
+---
## Cloud Issues
### Cloud Sync Errors
-1. Überprüfen Sie, ob „BASE_URL“ auf Ihre laufende Instanz verweist (z. B. „http://localhost:20128“).
-2. Überprüfen Sie, ob „CLOUD_URL“ auf Ihren Cloud-Endpunkt verweist (z. B. „https://omniroute.dev“).
-3. Halten Sie die Werte von „NEXT*PUBLIC*\*“ an den serverseitigen Werten ausgerichtet### Cloud `stream=false` Returns 500
+1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`)
+2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`)
+3. Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**Symptom:**„Unerwartetes Token „d“...“ auf dem Cloud-Endpunkt für Nicht-Streaming-Aufrufe.
+### Cloud `stream=false` Returns 500
-**Ursache:**Upstream gibt SSE-Nutzdaten zurück, während der Client JSON erwartet.
+**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls.
-**Problemumgehung:**Verwenden Sie „stream=true“ für Cloud-Direktaufrufe. Die lokale Laufzeit umfasst SSE→JSON-Fallback.### Cloud Says Connected but "Invalid API key"
+**Cause:** Upstream returns SSE payload while client expects JSON.
-1. Erstellen Sie einen neuen Schlüssel aus dem lokalen Dashboard („/api/keys“).
-2. Führen Sie die Cloud-Synchronisierung aus: Cloud aktivieren → Jetzt synchronisieren
-3. Alte/nicht synchronisierte Schlüssel können in der Cloud immer noch „401“ zurückgeben---
+**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback.
+
+### Cloud Says Connected but "Invalid API key"
+
+1. Create a fresh key from local dashboard (`/api/keys`)
+2. Run cloud sync: Enable Cloud → Sync Now
+3. Old/non-synced keys can still return `401` on cloud
+
+---
## Docker Issues
### CLI Tool Shows Not Installed
-1. Überprüfen Sie die Laufzeitfelder: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
-2. Für den tragbaren Modus: Verwenden Sie das Image-Ziel „runner-cli“ (gebündelte CLIs).
-3. Für den Host-Mount-Modus: Legen Sie „CLI_EXTRA_PATHS“ fest und mounten Sie das Host-Bin-Verzeichnis als schreibgeschützt
-4. Wenn „installed=true“ und „runnable=false“: Binärdatei wurde gefunden, aber die Integritätsprüfung ist fehlgeschlagen### Quick Runtime Validation
+1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
+2. For portable mode: use image target `runner-cli` (bundled CLIs)
+3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only
+4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck
+
+### Quick Runtime Validation
```bash
curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
@@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,
### High Costs
-1. Überprüfen Sie die Nutzungsstatistiken im Dashboard → Nutzung
-2. Primärmodell auf GLM/MiniMax umstellen
-3. Nutzen Sie den kostenlosen Tarif (Gemini CLI, Qoder) für unkritische Aufgaben
-4. Legen Sie Kostenbudgets pro API-Schlüssel fest: Dashboard → API-Schlüssel → Budget---
+1. Check usage stats in Dashboard → Usage
+2. Switch primary model to GLM/MiniMax
+3. Use free tier (Gemini CLI, Qoder) for non-critical tasks
+4. Set cost budgets per API key: Dashboard → API Keys → Budget
+
+---
## Debugging
### Enable Request Logs
-Setzen Sie „ENABLE_REQUEST_LOGS=true“ in Ihrer „.env“-Datei. Protokolle werden im Verzeichnis „logs/“ angezeigt.### Check Provider Health
+Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory.
+
+### Check Provider Health
```bash
# Health dashboard
@@ -100,102 +178,135 @@ curl http://localhost:20128/api/monitoring/health
### Runtime Storage
-- Hauptstatus: „${DATA_DIR}/storage.sqlite“ (Anbieter, Kombinationen, Aliase, Schlüssel, Einstellungen)
-- Verwendung: SQLite-Tabellen in „storage.sqlite“ („usage_history“, „call_logs“, „proxy_logs“) + optional „${DATA_DIR}/log.txt“ und „${DATA_DIR}/call_logs/“.
-- Protokolle anfordern: `/logs/...` (wenn `ENABLE_REQUEST_LOGS=true`)---
+- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings)
+- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/`
+- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`)
+
+---
## Circuit Breaker Issues
### Provider stuck in OPEN state
-Wenn der Leistungsschalter eines Anbieters OFFEN ist, werden Anfragen blockiert, bis die Abklingzeit abgelaufen ist.
+When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires.
**Fix:**
-1. Gehen Sie zu**Dashboard → Einstellungen → Resilienz**
-2. Überprüfen Sie die Leistungsschalterkarte des betroffenen Anbieters
-3. Klicken Sie auf**Alle zurücksetzen**, um alle Unterbrecher zu löschen, oder warten Sie, bis die Abklingzeit abgelaufen ist
-4. Stellen Sie vor dem Zurücksetzen sicher, dass der Anbieter tatsächlich verfügbar ist### Provider keeps tripping the circuit breaker
+1. Go to **Dashboard → Settings → Resilience**
+2. Check the circuit breaker card for the affected provider
+3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire
+4. Verify the provider is actually available before resetting
-Wenn ein Anbieter wiederholt in den OPEN-Zustand wechselt:
+### Provider keeps tripping the circuit breaker
-1. Überprüfen Sie**Dashboard → Health → Provider Health**auf das Fehlermuster
-2. Gehen Sie zu**Einstellungen → Ausfallsicherheit → Anbieterprofile**und erhöhen Sie den Fehlerschwellenwert
-3. Überprüfen Sie, ob der Anbieter die API-Grenzwerte geändert hat oder eine erneute Authentifizierung erfordert
-4. Überprüfen Sie die Latenz-Telemetrie – hohe Latenz kann zu zeitüberschreitungsbedingten Fehlern führen---
+If a provider repeatedly enters OPEN state:
+
+1. Check **Dashboard → Health → Provider Health** for the failure pattern
+2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold
+3. Check if the provider has changed API limits or requires re-authentication
+4. Review latency telemetry — high latency may cause timeout-based failures
+
+---
## Audio Transcription Issues
### "Unsupported model" error
-- Stellen Sie sicher, dass Sie das richtige Präfix verwenden: „deepgram/nova-3“ oder „assemblyai/best“.
-- Überprüfen Sie, ob der Anbieter unter**Dashboard → Anbieter**verbunden ist.### Transcription returns empty or fails
+- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best`
+- Verify the provider is connected in **Dashboard → Providers**
-- Überprüfen Sie die unterstützten Audioformate: „mp3“, „wav“, „m4a“, „flac“, „ogg“, „webm“.
-- Stellen Sie sicher, dass die Dateigröße innerhalb der Anbietergrenzen liegt (normalerweise < 25 MB).
-- Überprüfen Sie die Gültigkeit des API-Schlüssels des Anbieters auf der Anbieterkarte---
+### Transcription returns empty or fails
+
+- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm`
+- Verify file size is within provider limits (typically < 25MB)
+- Check provider API key validity in the provider card
+
+---
## Translator Debugging
-Verwenden Sie**Dashboard → Übersetzer**, um Formatübersetzungsprobleme zu beheben:
+Use **Dashboard → Translator** to debug format translation issues:
-| Modus | Wann zu verwenden |
-| ---------------- | --------------------------------------------------------------------------------------------------------------------------------------- | ------------------------ |
-| **Spielplatz** | Vergleichen Sie Eingabe-/Ausgabeformate nebeneinander – fügen Sie eine fehlgeschlagene Anfrage ein, um zu sehen, wie sie übersetzt wird |
-| **Chat-Tester** | Senden Sie Live-Nachrichten und überprüfen Sie die vollständige Anfrage-/Antwort-Nutzlast einschließlich Header |
-| **Prüfstand** | Führen Sie Stapeltests über Formatkombinationen hinweg durch, um herauszufinden, welche Übersetzungen fehlerhaft sind |
-| **Live-Monitor** | Beobachten Sie den Anfragefluss in Echtzeit, um zeitweise auftretende Übersetzungsprobleme zu erkennen | ### Common format issues |
+| Mode | When to Use |
+| ---------------- | -------------------------------------------------------------------------------------------- |
+| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates |
+| **Chat Tester** | Send live messages and inspect the full request/response payload including headers |
+| **Test Bench** | Run batch tests across format combinations to find which translations are broken |
+| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues |
--**Thinking-Tags werden nicht angezeigt**– Überprüfen Sie, ob der Zielanbieter Thinking und die Einstellung des Thinking-Budgets unterstützt -**Tool-Aufrufe löschen**– Bei einigen Formatübersetzungen werden möglicherweise nicht unterstützte Felder entfernt. im Playground-Modus überprüfen -**Systemaufforderung fehlt**– Claude und Gemini gehen unterschiedlich mit Systemaufforderungen um; Überprüfen Sie die Übersetzungsausgabe -**SDK gibt Rohzeichenfolge statt Objekt zurück**– In Version 1.1.0 behoben: Antwortbereinigung entfernt jetzt nicht standardmäßige Felder (`x_groq`, `usage_breakdown` usw.), die zu OpenAI SDK Pydantic-Validierungsfehlern führen -**GLM/ERNIE lehnt „System“-Rolle ab**– In Version 1.1.0 behoben: Der Rollennormalisierer führt automatisch Systemmeldungen in Benutzermeldungen für inkompatible Modelle zusammen -**Rolle „Entwickler“ nicht erkannt**– In Version 1.1.0 behoben: Für Nicht-OpenAI-Anbieter automatisch in „System“ konvertiert -**`json_schema` funktioniert nicht mit Gemini**– In v1.1.0 behoben: `response_format` wird jetzt in Geminis `responseMimeType` + `responseSchema` konvertiert---
+### Common format issues
+
+- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting
+- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode
+- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output
+- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures
+- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models
+- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers
+- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema`
+
+---
## Resilience Settings
### Auto rate-limit not triggering
-– Die automatische Ratenbegrenzung gilt nur für API-Schlüsselanbieter (nicht OAuth/Abonnement).
+- Auto rate-limit only applies to API key providers (not OAuth/subscription)
+- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled
+- Check if the provider returns `429` status codes or `Retry-After` headers
-- Überprüfen Sie, ob in**Einstellungen → Ausfallsicherheit → Anbieterprofile**die automatische Ratenbegrenzung aktiviert ist
-- Überprüfen Sie, ob der Anbieter „429“-Statuscodes oder „Retry-After“-Header zurückgibt### Tuning exponential backoff
+### Tuning exponential backoff
-Anbieterprofile unterstützen diese Einstellungen:
+Provider profiles support these settings:
--**Basisverzögerung**– Anfängliche Wartezeit nach dem ersten Fehler (Standard: 1 s) -**Max. Verzögerung**– Maximale Wartezeitobergrenze (Standard: 30 s) -**Multiplikator**– Wie viel Verzögerung pro aufeinanderfolgendem Fehler erhöht werden soll (Standard: 2x)### Anti-thundering herd
+- **Base delay** — Initial wait time after first failure (default: 1s)
+- **Max delay** — Maximum wait time cap (default: 30s)
+- **Multiplier** — How much to increase delay per consecutive failure (default: 2x)
-Wenn viele gleichzeitige Anfragen einen Anbieter mit begrenzter Rate treffen, verwendet OmniRoute Mutex + automatische Ratenbegrenzung, um Anfragen zu serialisieren und kaskadierende Fehler zu verhindern. Dies geschieht automatisch für API-Schlüsselanbieter.---
+### Anti-thundering herd
+
+When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers.
+
+---
## Optional RAG / LLM failure taxonomy (16 problems)
-Einige OmniRoute-Benutzer platzieren das Gateway vor RAG- oder Agent-Stacks. In diesen Setups ist es üblich, ein seltsames Muster zu erkennen: OmniRoute sieht fehlerfrei aus (Anbieter aktiv, Routing-Profile in Ordnung, keine Ratenbegrenzungswarnungen), aber die endgültige Antwort ist immer noch falsch.
+Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong.
-In der Praxis gehen diese Vorfälle meist von der nachgelagerten RAG-Pipeline aus, nicht vom Gateway selbst.
+In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself.
-Wenn Sie ein gemeinsames Vokabular zur Beschreibung dieser Fehler wünschen, können Sie die WFGY ProblemMap verwenden, eine externe MIT-Lizenztextressource, die sechzehn wiederkehrende RAG-/LLM-Fehlermuster definiert. Auf hohem Niveau umfasst es:
+If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers:
-- Abrufdrift und gebrochene Kontextgrenzen
-- leere oder veraltete Indizes und Vektorspeicher
-- Einbettung versus semantische Nichtübereinstimmung
-- Probleme mit der Eingabeaufforderung und dem Kontextfenster
-- Zusammenbruch der Logik und übertriebene Antworten
-- Fehler bei der Koordinierung langer Ketten und Agenten
-- Multiagentengedächtnis und Rollendrift
-- Probleme bei der Bereitstellung und Bootstrap-Reihenfolge
+- retrieval drift and broken context boundaries
+- empty or stale indexes and vector stores
+- embedding versus semantic mismatch
+- prompt assembly and context window issues
+- logic collapse and overconfident answers
+- long chain and agent coordination failures
+- multi agent memory and role drift
+- deployment and bootstrap ordering problems
-Die Idee ist einfach:
+The idea is simple:
-1. Wenn Sie eine schlechte Antwort untersuchen, erfassen Sie Folgendes:
- - Benutzeraufgabe und -anfrage
- - Routen- oder Anbieterkombination in OmniRoute
- - jeglicher RAG-Kontext, der nachgelagert verwendet wird (abgerufene Dokumente, Tool-Aufrufe usw.)
-2. Ordnen Sie den Vorfall einer oder zwei WFGY ProblemMap-Nummern („Nr. 1“ … „Nr. 16“) zu.
-3. Speichern Sie die Nummer in Ihrem eigenen Dashboard, Runbook oder Incident-Tracker neben den OmniRoute-Protokollen.
-4. Verwenden Sie die entsprechende WFGY-Seite, um zu entscheiden, ob Sie Ihren RAG-Stack, Retriever oder Ihre Routing-Strategie ändern müssen.
+1. When you investigate a bad response, capture:
+ - user task and request
+ - route or provider combo in OmniRoute
+ - any RAG context used downstream (retrieved documents, tool calls, etc)
+2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`).
+3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs.
+4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy.
-Volltext und konkrete Rezepte gibt es hier (MIT-Lizenz, nur Text):
+Full text and concrete recipes live here (MIT license, text only):
[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md)
-Sie können diesen Abschnitt ignorieren, wenn Sie keine RAG- oder Agent-Pipelines hinter OmniRoute ausführen.---
+You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute.
+
+---
## Still Stuck?
--**GitHub-Probleme**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architektur**: Interne Details finden Sie unter [`docs/ARCHITECTURE.md`](ARCHITECTURE.md). -**API-Referenz**: Siehe [`docs/API_REFERENCE.md`](API_REFERENCE.md) für alle Endpunkte -**Gesundheits-Dashboard**: Überprüfen Sie**Dashboard → Gesundheit**auf den Echtzeit-Systemstatus -**Übersetzer**: Verwenden Sie**Dashboard → Übersetzer**, um Formatprobleme zu beheben
+- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details
+- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints
+- **Health Dashboard**: Check **Dashboard → Health** for real-time system status
+- **Translator**: Use **Dashboard → Translator** to debug format issues
diff --git a/docs/i18n/de/llm.txt b/docs/i18n/de/llm.txt
new file mode 100644
index 0000000000..82c5317d30
--- /dev/null
+++ b/docs/i18n/de/llm.txt
@@ -0,0 +1,476 @@
+# OmniRoute (Deutsch)
+
+🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt)
+
+---
+
+
+> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app.
+
+## Übersicht
+
+OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free.
+
+**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost.
+
+**Current version:** 3.5.5
+
+## Tech Stack
+
+- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`)
+- **Framework:** Next.js 16 (App Router) with TypeScript 5.9
+- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations)
+- **State management:** Zustand (client), SQLite (server persistence)
+- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons
+- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth
+- **Schemas:** Zod v4 for all API / MCP input validation
+- **Background jobs:** Custom token health check scheduler, 24h model auto-sync
+- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses
+- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine
+- **i18n:** next-intl with 30 languages
+- **Desktop:** Electron (cross-platform: Windows, macOS, Linux)
+- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`)
+
+## Project Structure
+
+```
+/
+├── src/ # Main application source
+│ ├── app/ # Next.js App Router pages and API routes
+│ │ ├── (dashboard)/ # Dashboard UI pages
+│ │ │ └── dashboard/
+│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents)
+│ │ │ ├── analytics/ # Usage analytics and charts
+│ │ │ ├── api-manager/ # API key management
+│ │ │ ├── audit/ # Audit logs
+│ │ │ ├── auto-combo/ # Auto-combo engine dashboard
+│ │ │ ├── cache/ # Cache dashboard (semantic cache stats)
+│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.)
+│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates)
+│ │ │ ├── costs/ # Cost tracking per provider/model
+│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs
+│ │ │ ├── health/ # System health (uptime, circuit breakers, latency)
+│ │ │ ├── limits/ # Rate limits dashboard
+│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed)
+│ │ │ ├── media/ # Image/video/music generation + transcription
+│ │ │ ├── memory/ # Memory system dashboard
+│ │ │ ├── onboarding/ # Onboarding wizard
+│ │ │ ├── playground/ # Model playground (Monaco editor, streaming)
+│ │ │ ├── providers/ # Provider management (OAuth + API key + free)
+│ │ │ ├── search-tools/ # Search tools configuration
+│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced)
+│ │ │ ├── skills/ # Skills system dashboard
+│ │ │ ├── translator/ # Format translator + debug tools
+│ │ │ └── usage/ # Usage history
+│ │ ├── api/ # REST API endpoints (51 route directories)
+│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings,
+│ │ │ │ # images, audio, videos, music, moderations, rerank, search,
+│ │ │ │ # responses, messages, registered-keys, quotas, accounts)
+│ │ │ ├── v1beta/ # Gemini-compatible API
+│ │ │ ├── a2a/ # A2A agent management API
+│ │ │ ├── acp/ # ACP agent management API
+│ │ │ ├── oauth/ # OAuth flows per provider
+│ │ │ ├── providers/ # Provider CRUD and batch testing
+│ │ │ ├── models/ # Dashboard model listing and aliases
+│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains)
+│ │ │ ├── memory/ # Memory system API
+│ │ │ ├── skills/ # Skills system API
+│ │ │ ├── evals/ # Eval runner API
+│ │ │ ├── mcp/ # MCP HTTP transport API
+│ │ │ ├── search/ # Search provider API
+│ │ │ ├── webhooks/ # Webhook management
+│ │ │ ├── tunnels/ # Cloudflare tunnel management
+│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.)
+│ │ ├── landing/ # Landing page
+│ │ ├── login/ # Login page
+│ │ ├── forgot-password/ # Password recovery
+│ │ ├── status/ # Status page
+│ │ └── docs/ # In-app documentation
+│ ├── domain/ # Domain types and policy engine
+│ │ ├── policyEngine.ts # Central policy engine
+│ │ ├── comboResolver.ts # Combo resolution logic
+│ │ ├── costRules.ts # Cost calculation rules
+│ │ ├── degradation.ts # Graceful degradation
+│ │ ├── fallbackPolicy.ts # Fallback behavior
+│ │ ├── lockoutPolicy.ts # Account lockout logic
+│ │ ├── modelAvailability.ts # Model availability checks
+│ │ ├── providerExpiration.ts # Provider credential expiration
+│ │ ├── quotaCache.ts # Quota caching layer
+│ │ ├── configAudit.ts # Configuration auditing
+│ │ └── responses.ts # Domain response types
+│ ├── i18n/ # Internationalization
+│ │ └── messages/ # 30 language JSON files
+│ ├── lib/ # Core libraries
+│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server
+│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting)
+│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup
+│ │ │ └── streaming.ts # SSE streaming for A2A
+│ │ ├── acp/ # Agent Communication Protocol registry and manager
+│ │ ├── compliance/ # Compliance policy engine
+│ │ ├── db/ # SQLite database layer (21 modules + migrations)
+│ │ │ ├── core.ts # Database initialization, connection, schema
+│ │ │ ├── providers.ts # Provider connection CRUD
+│ │ │ ├── models.ts # Model catalog management
+│ │ │ ├── combos.ts # Combo configuration
+│ │ │ ├── apiKeys.ts # API key management
+│ │ │ ├── settings.ts # Settings persistence
+│ │ │ ├── backup.ts # Database backup/restore
+│ │ │ ├── proxies.ts # Proxy registry
+│ │ │ ├── prompts.ts # Prompt templates
+│ │ │ ├── webhooks.ts # Webhook subscriptions
+│ │ │ ├── detailedLogs.ts # Detailed request logging
+│ │ │ ├── domainState.ts # Domain state persistence
+│ │ │ ├── registeredKeys.ts # Registered API keys with quotas
+│ │ │ ├── quotaSnapshots.ts # Quota snapshot history
+│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings
+│ │ │ ├── cliToolState.ts # CLI tool state tracking
+│ │ │ ├── encryption.ts # Data encryption
+│ │ │ ├── readCache.ts # Read-through cache layer
+│ │ │ ├── secrets.ts # Secrets management
+│ │ │ ├── stateReset.ts # State reset utilities
+│ │ │ ├── migrationRunner.ts # Schema migration runner
+│ │ │ └── migrations/ # 16 SQL migration files
+│ │ ├── evals/ # Eval runner and scheduler
+│ │ ├── memory/ # Persistent conversational memory
+│ │ │ ├── extraction.ts # Memory extraction from conversations
+│ │ │ ├── injection.ts # Memory injection into context
+│ │ │ ├── retrieval.ts # Memory retrieval/search
+│ │ │ ├── store.ts # Memory persistence layer
+│ │ │ └── summarization.ts # Memory summarization
+│ │ ├── oauth/ # OAuth providers, services, and utilities
+│ │ │ ├── constants/ # Default OAuth credentials (overridable via env)
+│ │ │ ├── providers/ # Provider-specific OAuth configs
+│ │ │ ├── services/ # Provider-specific token exchange logic
+│ │ │ └── utils/ # PKCE, callback server, token helpers
+│ │ ├── plugins/ # Plugin system
+│ │ ├── skills/ # Extensible skill framework
+│ │ │ ├── registry.ts # Skill registration
+│ │ │ ├── executor.ts # Skill execution engine
+│ │ │ ├── sandbox.ts # Skill sandbox environment
+│ │ │ ├── builtin/ # Built-in skills
+│ │ │ ├── interception.ts # Skill request interception
+│ │ │ └── injection.ts # Skill context injection
+│ │ ├── usage/ # Usage tracking system
+│ │ │ ├── callLogs.ts # Call log persistence
+│ │ │ ├── costCalculator.ts # Cost calculation engine
+│ │ │ └── usageHistory.ts # Usage history queries
+│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers
+│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management
+│ │ ├── pricingSync.ts # LiteLLM pricing data sync
+│ │ ├── semanticCache.ts # Semantic caching layer
+│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler
+│ │ ├── webhookDispatcher.ts # Webhook event dispatcher
+│ │ └── localDb.ts # Unified re-export layer for all DB modules
+│ ├── middleware/ # Request middleware
+│ │ └── promptInjectionGuard.ts # Prompt injection detection
+│ ├── mitm/ # MITM proxy capability
+│ │ ├── cert/ # Certificate management
+│ │ ├── dns/ # DNS handling
+│ │ ├── targets/ # Target routing
+│ │ └── manager.ts # MITM proxy manager
+│ ├── shared/ # Shared utilities, components, and constants
+│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.)
+│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes
+│ │ ├── contracts/ # Shared API contracts
+│ │ ├── hooks/ # React hooks
+│ │ ├── middleware/ # Shared middleware utilities
+│ │ ├── schemas/ # Shared Zod schemas
+│ │ ├── services/ # Shared services
+│ │ ├── types/ # Shared TypeScript types
+│ │ ├── validation/ # Zod schemas (settings, providers, routes)
+│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID)
+│ ├── sse/ # SSE proxy pipeline
+│ │ ├── services/ # Auth resolution, format translation, response handling
+│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency
+│ ├── store/ # Zustand client-side stores (theme, providers, etc.)
+│ └── types/ # TypeScript type definitions
+├── open-sse/ # Standalone SSE server (npm workspace)
+│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video,
+│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models)
+│ ├── executors/ # Provider-specific request executors (14 executors)
+│ │ ├── base.ts # Base executor with shared logic
+│ │ ├── default.ts # Default OpenAI-compatible executor
+│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum)
+│ │ ├── codex.ts # OpenAI Codex CLI
+│ │ ├── antigravity.ts # Antigravity IDE
+│ │ ├── github.ts # GitHub Copilot
+│ │ ├── gemini-cli.ts # Gemini CLI
+│ │ ├── kiro.ts # Kiro AI
+│ │ ├── qoder.ts # Qoder AI
+│ │ ├── vertex.ts # Vertex AI (Service Account JSON)
+│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI
+│ │ ├── opencode.ts # OpenCode Zen/Go
+│ │ ├── pollinations.ts # Pollinations AI
+│ │ └── puter.ts # Puter AI
+│ ├── handlers/ # Request handlers per API type (11 handlers)
+│ │ ├── chatCore.ts # Main chat completions handler
+│ │ ├── responsesHandler.ts # OpenAI Responses API handler
+│ │ ├── embeddings.ts # Embedding generation
+│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.)
+│ │ ├── videoGeneration.ts # Video generation
+│ │ ├── musicGeneration.ts # Music generation
+│ │ ├── audioSpeech.ts # Text-to-speech
+│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI)
+│ │ ├── moderations.ts # Content moderation
+│ │ ├── rerank.ts # Reranking API
+│ │ └── search.ts # Web search API
+│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP)
+│ │ ├── server.ts # MCP server core (tool registration, scope enforcement)
+│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools)
+│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a)
+│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes)
+│ │ ├── audit.ts # Tool call audit logging
+│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat
+│ │ └── httpTransport.ts # HTTP transport handler
+│ ├── services/ # 36+ service modules
+│ │ ├── combo.ts # Core routing engine
+│ │ ├── usage.ts # Usage tracking
+│ │ ├── tokenRefresh.ts # OAuth token refresh
+│ │ ├── rateLimitManager.ts # Rate limit management
+│ │ ├── accountFallback.ts # Multi-account fallback
+│ │ ├── sessionManager.ts # Session management
+│ │ ├── wildcardRouter.ts # Wildcard model routing
+│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration)
+│ │ ├── intentClassifier.ts # Request intent classification
+│ │ ├── taskAwareRouter.ts # Task-aware routing
+│ │ ├── thinkingBudget.ts # Thinking budget management
+│ │ ├── contextManager.ts # Context window management
+│ │ ├── modelDeprecation.ts # Model deprecation handling
+│ │ ├── modelFamilyFallback.ts # Intra-family model fallback
+│ │ ├── emergencyFallback.ts # Emergency fallback
+│ │ ├── workflowFSM.ts # Workflow state machine
+│ │ ├── backgroundTaskDetector.ts # Background task detection
+│ │ ├── ipFilter.ts # IP-based access control
+│ │ ├── signatureCache.ts # CLI signature caching
+│ │ ├── volumeDetector.ts # Request volume detection
+│ │ ├── contextHandoff.ts # Context relay handoff generation and injection
+│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay
+│ │ └── ... # Additional services (14 more modules)
+│ ├── transformer/ # Responses API transformer
+│ │ └── responsesTransformer.ts
+│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek)
+│ │ ├── request/ # Request translators per provider
+│ │ ├── response/ # Response translators per provider
+│ │ ├── helpers/ # Translation helpers
+│ │ └── image/ # Image format translation
+│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.)
+├── electron/ # Electron desktop app (cross-platform)
+│ ├── main.js # Electron main process
+│ ├── preload.js # Preload script (IPC bridge)
+│ └── assets/ # App icons and assets
+├── tests/ # Test suites
+│ ├── unit/ # 122 unit test files
+│ ├── integration/ # Integration tests
+│ ├── e2e/ # Playwright E2E tests
+│ ├── security/ # Security tests
+│ ├── translator/ # Translator-specific tests
+│ └── load/ # Load tests
+├── docs/ # Documentation
+│ ├── i18n/ # 30-language translated docs
+│ ├── ARCHITECTURE.md # Full architecture documentation
+│ ├── API_REFERENCE.md # API reference
+│ ├── USER_GUIDE.md # User guide
+│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview
+│ ├── CLI-TOOLS.md # CLI tools integration guide
+│ ├── A2A-SERVER.md # A2A agent protocol documentation
+│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring)
+│ ├── MCP-SERVER.md # MCP server (25 tools)
+│ ├── TROUBLESHOOTING.md # Troubleshooting guide
+│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide
+│ ├── openapi.yaml # OpenAPI specification
+│ └── screenshots/ # Dashboard screenshots
+├── bin/ # CLI entry points (omniroute, reset-password)
+├── scripts/ # Build and utility scripts
+└── .env.example # Environment variable template
+```
+
+## Key Features (v3.5.5)
+
+### Core Proxy
+- **60+ AI providers** with automatic format translation
+- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible)
+- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay
+- **4-tier fallback**: Subscription → API Key → Cheap → Free
+- **Context Relay strategy**: Session handoff summaries on account rotation for continuity
+- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown
+- **Semantic caching** with cache hit/miss headers
+- **Idempotency** with configurable dedup window
+- **Circuit breaker** per provider with configurable thresholds
+- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback
+- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers
+- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement
+- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization
+- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills
+- **Prompt Injection Guard**: Middleware-level prompt injection detection
+- **MITM Proxy**: Certificate management, DNS handling, and target routing
+- **Cloudflare Tunnels**: Managed tunnel creation for remote access
+- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches)
+
+### Sicherheit
+- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs)
+- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()`
+- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()`
+- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection
+- **CLI Fingerprint Matching** — Per-provider request signature matching
+- **Prompt injection guard** — Request middleware detection
+- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`)
+- **PII sanitizer** — Sensitive data scrubbing in logs
+
+### Dashboard Pages (23 sections)
+- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons
+- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies
+- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics
+- **Analytics** — Token consumption, cost, heatmaps, distributions
+- **Health** — Uptime, memory, latency percentiles, circuit breakers
+- **Logs** — Request, Proxy, Audit, Console (tabbed)
+- **Audit** — Audit trail and compliance logging
+- **Costs** — Cost tracking per provider/model
+- **Limits** — Rate limit monitoring
+- **Cache** — Semantic cache statistics and management
+- **CLI Tools** — One-click configuration for 10+ AI CLI tools
+- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration
+- **Playground** — Test any model with Monaco editor, streaming responses
+- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files)
+- **Search Tools** — Search provider configuration and testing
+- **Memory** — Memory system management and visualization
+- **Skills** — Skills framework management and execution
+- **Translator** — Format debugging: playground, chat tester, test bench, live monitor
+- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced
+- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed)
+- **Onboarding** — Setup wizard for new users
+- **Usage** — Usage history and analytics
+- **API Manager** — API key management with scoped permissions
+
+### Protocol Support
+- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations`
+- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens`
+- **OpenAI Responses** — `/v1/responses`
+- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}`
+- **Ollama** — `/v1/api/chat`, `/api/tags`
+- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily)
+- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP)
+- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills)
+- **ACP** — Agent Communication Protocol registry and manager
+
+### MCP Server (25 Tools)
+| Category | Tools |
+|-----------|-------|
+| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` |
+| Memory (3) | `memory_search`, `memory_add`, `memory_clear` |
+| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` |
+
+**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience`
+
+### Provider Categories
+
+**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI
+
+**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline
+
+**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan
+
+**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs
+
+### Internationalization
+- 30 languages for UI (all dashboard pages)
+- 30 translated documentation sets in docs/i18n/
+- Language switcher in documentation
+
+## Key Architectural Decisions
+
+1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints.
+
+2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently.
+
+3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation.
+
+4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity.
+
+5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client.
+
+6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes.
+
+7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`).
+
+8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages.
+
+9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations.
+
+10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`.
+
+11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation.
+
+12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline.
+
+13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast.
+
+## Main Flows
+
+### Proxy Request Flow
+1. Client sends OpenAI-format request to `/v1/chat/completions`
+2. API key validation
+3. Model resolution: direct model or combo lookup
+4. For combos: iterate through models with selected strategy
+5. Auth resolution: get credentials for the target provider
+6. Format translation: OpenAI → provider native format
+7. CLI fingerprint matching (if enabled for provider)
+8. Upstream request with circuit breaker and rate limiting
+9. Response translation: provider → OpenAI format
+10. omniModel tag sanitization (strip internal tags)
+11. SSE streaming back to client
+12. Memory extraction (if memory system enabled)
+13. Usage logging and cost calculation
+
+### OAuth Flow
+1. Dashboard initiates `/api/oauth/[provider]/authorize`
+2. User completes OAuth login in browser
+3. Callback hits `/api/oauth/[provider]/exchange`
+4. Tokens stored as a provider connection in SQLite
+5. Background job refreshes tokens before expiry
+
+## Important Notes for LLMs
+
+1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only).
+
+2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`).
+
+3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services.
+
+4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`.
+
+5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module.
+
+6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`).
+
+7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes.
+
+8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB.
+
+9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown.
+
+10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130).
+
+11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux).
+
+12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint.
+
+13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API.
+
+14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API.
+
+15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time.
+
+16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`.
+
+17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22.
+
+18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false.
+
+## Links
+
+- Repository: https://github.com/diegosouzapw/OmniRoute
+- Website: https://omniroute.online
+- npm: https://www.npmjs.com/package/omniroute
+- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute
+- Documentation: See `/docs/` directory
diff --git a/docs/i18n/es/README.md b/docs/i18n/es/README.md
index ced39b01f1..47720abff4 100644
--- a/docs/i18n/es/README.md
+++ b/docs/i18n/es/README.md
@@ -4,11 +4,14 @@
---
+
### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback.
-_Su proxy API universal: un punto final, más de 60 proveedores, cero tiempo de inactividad. Ahora con**Servidor MCP (25 herramientas)**,**Protocolo A2A**,**Sistemas de memoria/habilidades**y**Aplicación de escritorio Electron**._
+_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._
-**Finalización de chat • Incrustaciones • Generación de imágenes • Vídeo • Música • Audio • Reclasificación •**Búsqueda web**• Servidor MCP • Protocolo A2A • 100% TypeScript**---
+**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript**
+
+---
@@ -39,9 +42,13 @@ _Su proxy API universal: un punto final, más de 60 proveedores, cero tiempo de
[](https://omniroute.online)
[](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-[🌐 Sitio web](https://omniroute.online) • [🚀 Inicio rápido](#-inicio rápido) • [💡 Funciones](#-key-features) • [📖 Documentos](#-documentación) • [💰 Precios](#-precios-de-un-vistazo) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-🌐**Disponible en:**🇺🇸 [Inglés](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [English](docs/i18n/es/README.md) | 🇫🇷 [Francés](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magiar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Países Bajos](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Esloveno](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)---
+
+
+🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)
+
+---
## 🖼️ Main Dashboard
@@ -53,553 +60,629 @@ _Su proxy API universal: un punto final, más de 60 proveedores, cero tiempo de
## 📸 Dashboard Preview
-
-Haga clic para ver capturas de pantalla del panel
+
+Click to see dashboard screenshots
-| Página | Captura de pantalla |
-| -------------------- | ------------------------------------------------------ | ---------- |
-| **Proveedores** |  |
-| **Combinaciones** |  |
-| **Análisis** |  |
-| **Salud** |  |
-| **Traductor** |  |
-| **Configuración** |  |
-| **Herramientas CLI** |  |
-| **Registros de uso** |  |
-| **Puntos finales** |  | |
+| Page | Screenshot |
+| -------------- | ------------------------------------------------- |
+| **Providers** |  |
+| **Combos** |  |
+| **Analytics** |  |
+| **Health** |  |
+| **Translator** |  |
+| **Settings** |  |
+| **CLI Tools** |  |
+| **Usage Logs** |  |
+| **Endpoints** |  |
+
+
---
### 🤖 Free AI Provider for your favorite coding agents
-_Conecte cualquier herramienta IDE o CLI con tecnología de IA a través de OmniRoute: puerta de enlace API gratuita para codificación ilimitada._
+_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._
-
+
-📡 Todos los agentes se conectan a través de http://localhost:20128/v1 o http://cloud.omniroute.online/v1: una configuración, modelos y cuotas ilimitados---
+📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota
+
+---
## 🤔 Why OmniRoute?
-**Deja de gastar dinero y alcanzar límites:**
+**Stop wasting money and hitting limits:**
--
La cuota de suscripción vence cada mes sin usarse
--
Los límites de velocidad le impiden codificar a mitad de camino
--
API costosas ($20-50/mes por proveedor)
--
Cambio manual entre proveedores
+-
Subscription quota expires unused every month
+-
Rate limits stop you mid-coding
+-
Expensive APIs ($20-50/month per provider)
+-
Manual switching between providers
-**OmniRoute resuelve esto:**
+**OmniRoute solves this:**
-- ✅**Maximizar suscripciones**- Realice un seguimiento de la cuota, use cada bit antes de restablecer
-- ✅**Retroceso automático**- Suscripción → Clave API → Barato → Gratis, sin tiempo de inactividad
-- ✅**Multicuenta**- Round-robin entre cuentas por proveedor
-- ✅**Universal**- Funciona con Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw y cualquier herramienta CLI---
+- ✅ **Maximize subscriptions** - Track quota, use every bit before reset
+- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime
+- ✅ **Multi-account** - Round-robin between accounts per provider
+- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool
+
+---
## 📧 Support
-> 💬**¡Únase a nuestra comunidad!**[Grupo de WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t): obtenga ayuda, comparta consejos y manténgase actualizado.
+> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated.
--**Sitio web**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problemas**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Grupo comunitario](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Contribuyendo**: consulte [CONTRIBUTING.md](CONTRIBUTING.md), abra un PR o elija un "buen primer número". -**Proyecto original**: [9router de decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug?
+- **Website**: [omniroute.online](https://omniroute.online)
+- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute)
+- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue`
+- **Original Project**: [9router by decolua](https://github.com/decolua/9router)
-Al abrir un problema, ejecute el comando system-info y adjunte el archivo generado:```bash
+### 🐛 Reporting a Bug?
+
+When opening an issue, please run the system-info command and attach the generated file:
+
+```bash
npm run system-info
-
```
-Esto genera un `system-info.txt` con su versión de Node.js, versión de OmniRoute, detalles del sistema operativo, herramientas CLI instaladas (qoder, gemini, claude, codex, antigravity, droid, etc.), estado de Docker/PM2 y paquetes del sistema: todo lo que necesitamos para reproducir su problema rápidamente. Adjunte el archivo directamente a su problema de GitHub.---
+This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue.
+
+---
## 🔄 How It Works
```
-
┌─────────────┐
-│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
-│ Tool │
+│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
+│ Tool │
└──────┬──────┘
-│ http://localhost:20128/v1
-↓
+ │ http://localhost:20128/v1
+ ↓
┌─────────────────────────────────────────┐
-│ OmniRoute (Smart Router) │
-│ • Format translation (OpenAI ↔ Claude) │
-│ • Quota tracking + Embeddings + Images │
-│ • Auto token refresh │
+│ OmniRoute (Smart Router) │
+│ • Format translation (OpenAI ↔ Claude) │
+│ • Quota tracking + Embeddings + Images │
+│ • Auto token refresh │
└──────┬──────────────────────────────────┘
-│
-├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
-│ ↓ quota exhausted
-├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
-│ ↓ budget limit
-├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
-│ ↓ budget limit
-└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
+ │
+ ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
+ │ ↓ quota exhausted
+ ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
+ │ ↓ budget limit
+ ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
+ │ ↓ budget limit
+ └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
Result: Never stop coding, minimal cost
-
-````
+```
---
## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases
->**Todos los desarrolladores que utilizan herramientas de IA se enfrentan a estos problemas a diario.**OmniRoute se creó para resolverlos todos: desde sobrecostos hasta bloqueos regionales, desde flujos rotos de OAuth hasta operaciones de protocolo y observabilidad empresarial.
+> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability.
-
-💸 1. "Pago una suscripción costosa pero aún así me interrumpen los límites"
+
+💸 1. "I pay for an expensive subscription but still get interrupted by limits"
-Los desarrolladores pagan entre 20 y 200 dólares al mes por Claude Pro, Codex Pro o GitHub Copilot. Incluso pagando, la cuota tiene un límite: 5 horas de uso, límites semanales o límites de tarifa por minuto. A mitad de la sesión de codificación, el proveedor deja de responder y el desarrollador pierde flujo y productividad.
+Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity.
-**Cómo lo resuelve OmniRoute:**
+**How OmniRoute solves it:**
--**Reserva inteligente de 4 niveles**: si se agota la cuota de suscripción, se redirige automáticamente a la clave API → Barato → Gratis sin intervención manual
--**Seguimiento de límites del proveedor**: las instantáneas de cuota almacenadas en caché se actualizan según una programación del lado del servidor (predeterminado `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) con actualización manual disponible en la interfaz de usuario
--**Soporte multicuenta**: varias cuentas por proveedor con rotación automática: cuando una se agota, cambia a la siguiente
--**Combinaciones personalizadas**: cadenas de respaldo personalizables con 9 estrategias de equilibrio (prioridad, ponderada, llenado primero, round-robin, P2C, aleatoria, menos utilizada, de costo optimizado, estrictamente aleatoria)
--**Cuotas comerciales de Codex**: monitoreo de cuotas del espacio de trabajo empresarial/de equipo directamente en el panel
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention
+- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI
+- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next
+- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**)
+- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard
-
-🔌 2. "Necesito usar varios proveedores pero cada uno tiene una API diferente"
+
-OpenAI usa un formato, Claude (Anthropic) usa otro, Gemini otro más. Si un desarrollador quiere probar modelos de diferentes proveedores o recurrir a ellos, debe reconfigurar los SDK, cambiar los puntos finales y lidiar con formatos incompatibles. Los proveedores personalizados (FriendLI, NIM) tienen puntos finales de modelo no estándar.
+
+🔌 2. "I need to use multiple providers but each has a different API"
-**Cómo lo resuelve OmniRoute:**
+OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints.
--**Punto final unificado**: un único `http://localhost:20128/v1` sirve como proxy para los más de 60 proveedores
--**Traducción de formato**: automática y transparente: OpenAI ↔ Claude ↔ Gemini ↔ API de respuestas
--**Desinfección de respuesta**: elimina los campos no estándar (`x_groq`, `usage_breakdown`, `service_tier`) que interrumpen OpenAI SDK v1.83+
--**Normalización de roles**: convierte `desarrollador` → `sistema` para proveedores que no son OpenAI; `sistema` → `usuario` para GLM/ERNIE
--**Think Tag Extraction**: extrae bloques `` de modelos como DeepSeek R1 en `reasoning_content` estandarizado.
--**Salida estructurada para Gemini**— conversión automática `json_schema` → `responseMimeType`/`responseSchema`
--**`stream` por defecto es `false`**: se alinea con las especificaciones de OpenAI, evitando SSE inesperado en los SDK de Python/Rust/Go
+**How OmniRoute solves it:**
-
-🌐 3. "Mi proveedor de IA bloquea mi región/país"
+- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers
+- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
+- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+
+- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE
+- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content`
+- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion
+- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs
-Proveedores como OpenAI/Codex bloquean el acceso desde ciertas regiones geográficas. Los usuarios reciben errores como `unsupported_country_region_territory` durante las conexiones OAuth y API. Esto resulta especialmente frustrante para los desarrolladores de los países en desarrollo.
+
-**Cómo lo resuelve OmniRoute:**
+
+🌐 3. "My AI provider blocks my region/country"
--**Configuración de proxy de 3 niveles**: Proxy configurable en 3 niveles: global (todo el tráfico), por proveedor (un solo proveedor) y por conexión/clave.
--**Insignias de proxy codificadas por colores**— Indicadores visuales: 🟢 proxy global, 🟡 proxy de proveedor, 🔵 proxy de conexión, que siempre muestra la IP
--**Intercambio de tokens de OAuth a través de proxy**: el flujo de OAuth también pasa a través del proxy, lo que resuelve `unsupported_country_region_territory`
--**Pruebas de conexión a través de proxy**: las pruebas de conexión utilizan el proxy configurado (no más derivación directa)
--**Soporte SOCKS5**: soporte completo de proxy SOCKS5 para enrutamiento saliente
--**Suplantación de huellas dactilares TLS**: huella digital TLS similar a la de un navegador a través de `wreq-js` para evitar la detección de bots
--**🔏 Coincidencia de huellas dactilares CLI**: reordena los encabezados y los campos del cuerpo para que coincidan con las firmas binarias CLI nativas, lo que reduce drásticamente el riesgo de marcación de cuentas. La IP del proxy se conserva: obtienes enmascaramiento de IP oculto**y**simultáneamente
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries.
-
-🆓 4. "Quiero usar IA para codificar pero no tengo dinero"
+**How OmniRoute solves it:**
-No todo el mundo puede pagar entre 20 y 200 dólares al mes por suscripciones a IA. Los estudiantes, desarrolladores de países emergentes, aficionados y autónomos necesitan acceso a modelos de calidad sin coste alguno.
+- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key
+- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP
+- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory`
+- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass)
+- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection
+- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously
-**Cómo lo resuelve OmniRoute:**
+
--**Proveedores de nivel gratuito integrados**: soporte nativo para proveedores 100 % gratuitos: Qoder (5 modelos ilimitados a través de OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 modelos ilimitados: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID gratis), Gemini CLI (180.000 tokens/mes gratis)
--**Ollama Cloud**: modelos de Ollama alojados en la nube en `api.ollama.com` con nivel gratuito de "Uso ligero"; use el prefijo `ollamacloud/`
--**Combos solo gratuitos**— Cadena `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/mes sin tiempo de inactividad
--**Acceso gratuito a NVIDIA NIM**: desarrollo de ~40 RPM, acceso gratuito para siempre a más de 70 modelos en build.nvidia.com (transición de créditos a límites de velocidad pura)
--**Estrategia de optimización de costos**: estrategia de enrutamiento que elige automáticamente el proveedor más barato disponible
+
+🆓 4. "I want to use AI for coding but I have no money"
-
-🔒 5. "Necesito proteger mi puerta de enlace de IA del acceso no autorizado"
+Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost.
-Al exponer una puerta de enlace de IA a la red (LAN, VPS, Docker), cualquiera con la dirección puede consumir los tokens/cuota del desarrollador. Sin protección, las API son vulnerables al mal uso, la inyección rápida y el abuso.
+**How OmniRoute solves it:**
-**Cómo lo resuelve OmniRoute:**
+- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free)
+- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix
+- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime
+- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits)
+- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider
--**Administración de claves API**: generación, rotación y alcance por proveedor con una página dedicada `/dashboard/api-manager`
--**Permisos a nivel de modelo**: restrinja las claves API a modelos específicos (`openai/*`, patrones comodín), con la opción Permitir todo/Restringir
--**API Endpoint Protection**: requiere una clave para `/v1/models` y bloquea proveedores específicos del listado
--**Auth Guard + Protección CSRF**: todas las rutas del panel protegidas con middleware `withAuth` + tokens CSRF
--**Limitador de velocidad**: limitación de velocidad por IP con ventanas configurables
--**Filtrado de IP**: lista permitida/lista bloqueada para control de acceso
--**Prompt injection guard**: desinfección contra patrones de avisos maliciosos
--**Cifrado AES-256-GCM**: credenciales cifradas en reposo
+
-
-🛑 6. "Mi proveedor dejó de funcionar y perdí mi flujo de codificación"
+
+🔒 5. "I need to protect my AI gateway from unauthorized access"
-Los proveedores de IA pueden volverse inestables, devolver errores 5xx o alcanzar límites de velocidad temporales. Si un desarrollador depende de un solo proveedor, se le interrumpe. Sin disyuntores, los reintentos repetidos pueden bloquear la aplicación.
+When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse.
-**Cómo lo resuelve OmniRoute:**
+**How OmniRoute solves it:**
--**Disyuntor por modelo**: apertura/cierre automático con umbrales configurables y enfriamiento (cerrado/abierto/medio abierto), con alcance por modelo para evitar bloqueos en cascada
--**Retroceso exponencial**: retrasos progresivos en los reintentos
--**Anti-Thundering Herd**— Mutex + protección de semáforo contra tormentas de reintentos simultáneos
--**Cadenas alternativas combinadas**: si el proveedor principal falla, automáticamente pasa por la cadena sin intervención.
--**Disyuntor combinado**: desactiva automáticamente los proveedores defectuosos dentro de una cadena combinada
--**Panel de estado**: monitoreo del tiempo de actividad, estados de disyuntores, bloqueos, estadísticas de caché, latencia p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page
+- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle
+- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing
+- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens
+- **Rate Limiter** — Per-IP rate limiting with configurable windows
+- **IP Filtering** — Allowlist/blocklist for access control
+- **Prompt Injection Guard** — Sanitization against malicious prompt patterns
+- **AES-256-GCM Encryption** — Credentials encrypted at rest
-
-🔧 7. "Configurar cada herramienta de IA es tedioso y repetitivo"
+
-Los desarrolladores utilizan Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Cada herramienta necesita una configuración diferente (punto final API, clave, modelo). Reconfigurar al cambiar de proveedor o modelo es una pérdida de tiempo.
+
+🛑 6. "My provider went down and I lost my coding flow"
-**Cómo lo resuelve OmniRoute:**
+AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application.
--**Panel de herramientas CLI**: página dedicada con configuración con un solo clic para Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
--**Generador de configuración de GitHub Copilot**: genera `chatLanguageModels.json` para código VS con selección masiva de modelos
--**Asistente de incorporación**: configuración guiada de 4 pasos para usuarios nuevos
--**Un punto final, todos los modelos**: configure `http://localhost:20128/v1` una vez, acceda a más de 60 proveedores
+**How OmniRoute solves it:**
-
-🔑 8. "Administrar tokens OAuth de múltiples proveedores es un infierno"
+- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks
+- **Exponential Backoff** — Progressive retry delays
+- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms
+- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention
+- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain
+- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency
-Claude Code, Codex, Gemini CLI, Copilot: todos usan OAuth 2.0 con tokens que caducan. Los desarrolladores necesitan volver a autenticarse constantemente, lidiar con "falta client_secret", "redirect_uri_mismatch" y fallas en servidores remotos. OAuth en LAN/VPS es particularmente problemático.
+
-**Cómo lo resuelve OmniRoute:**
+
+🔧 7. "Configuring each AI tool is tedious and repetitive"
--**Actualización automática de tokens**: los tokens de OAuth se actualizan en segundo plano antes de que caduquen
--**OAuth 2.0 (PKCE) integrado**: flujo automático para Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
--**OAuth multicuenta**: varias cuentas por proveedor mediante extracción de token JWT/ID
--**OAuth LAN/Remote Fix**— Detección de IP privada para `redirect_uri` + modo URL manual para servidores remotos
--**OAuth detrás de Nginx**: utiliza `window.location.origin` para compatibilidad con proxy inverso
--**Guía remota de OAuth**: guía paso a paso para las credenciales de Google Cloud en VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time.
-
-📊 9. "No sé cuánto estoy gastando ni dónde"
+**How OmniRoute solves it:**
-Los desarrolladores utilizan múltiples proveedores pagos pero no tienen una visión unificada del gasto. Cada proveedor tiene su propio panel de facturación, pero no hay una vista consolidada. Los costos inesperados pueden acumularse.
+- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
+- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection
+- **Onboarding Wizard** — Guided 4-step setup for first-time users
+- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers
-**Cómo lo resuelve OmniRoute:**
+
--**Panel de análisis de costos**: seguimiento de costos por token y gestión de presupuesto por proveedor
--**Límites de presupuesto por nivel**: límite de gasto por nivel que activa el respaldo automático
--**Configuración de precios por modelo**: precios configurables por modelo
--**Estadísticas de uso por clave API**: recuento de solicitudes y marca de tiempo utilizada por última vez por clave
--**Panel de análisis**: tarjetas de estadísticas, tabla de uso de modelos, tabla de proveedores con tasas de éxito y latencia.
+
+🔑 8. "Managing OAuth tokens from multiple providers is hell"
-
-🐛 10. "No puedo diagnosticar errores ni problemas en las llamadas de IA"
+Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic.
-Cuando falla una llamada, el desarrollador no sabe si se trata de un límite de velocidad, un token caducado, un formato incorrecto o un error del proveedor. Registros fragmentados en diferentes terminales. Sin observabilidad, la depuración es de prueba y error.
+**How OmniRoute solves it:**
-**Cómo lo resuelve OmniRoute:**
+- **Auto Token Refresh** — OAuth tokens refresh in background before expiration
+- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
+- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction
+- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers
+- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility
+- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker
--**Panel de registros unificados**: 4 pestañas: registros de solicitudes, registros de proxy, registros de auditoría y consola
--**Visor de registros de consola**: visor estilo terminal en tiempo real con niveles codificados por colores, desplazamiento automático, búsqueda y filtro
--**Registros de proxy SQLite**: registros persistentes que sobreviven a los reinicios del servidor
--**Translator Playground**: 4 modos de depuración: Playground (traducción de formato), Chat Tester (ida y vuelta), Test Bench (por lotes), Live Monitor (en tiempo real)
--**Solicitud de telemetría**: latencia p50/p95/p99 + seguimiento de X-Request-Id
--**Registro basado en archivos con rotación**: los registros de aplicaciones rotan por tamaño, días de retención y recuento de archivos; Los artefactos del registro de llamadas rotan según los días de retención y el recuento de archivos.
--**Informe de información del sistema**: `npm run system-info` genera `system-info.txt` con su entorno completo (versión de nodo, versión de OmniRoute, sistema operativo, herramientas CLI, estado de Docker/PM2). Adjúntelo cuando informe problemas para una clasificación instantánea.
+
-
-🏗️ 11. "Implementar y mantener la puerta de enlace es complejo"
+
+📊 9. "I don't know how much I'm spending or where"
-Instalar, configurar y mantener un proxy de IA en diferentes entornos (local, VPS, Docker, nube) requiere mucha mano de obra. Problemas como rutas codificadas, "EACCES" en directorios, conflictos de puertos y compilaciones multiplataforma añaden fricción.
+Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up.
-**Cómo lo resuelve OmniRoute:**
+**How OmniRoute solves it:**
--**npm global install**— `npm install -g omniroute && omniroute` — hecho
--**Docker multiplataforma**: AMD64 + ARM64 nativo (Apple Silicon, AWS Graviton, Raspberry Pi)
--**Docker Compose Profiles**— `base` (sin herramientas CLI) y `cli` (con Claude Code, Codex, OpenClaw)
--**Aplicación de escritorio Electron**: aplicación nativa para Windows/macOS/Linux con bandeja del sistema, inicio automático y modo sin conexión
--**Modo de puerto dividido**: API y panel en puertos separados para escenarios avanzados (proxy inverso, redes de contenedores)
--**Cloud Sync**: sincronización de configuración entre dispositivos a través de Cloudflare Workers
--**Copias de seguridad de base de datos**: copia de seguridad, restauración, exportación e importación automáticas de todas las configuraciones, con `DISABLE_SQLITE_AUTO_BACKUP` para copias de seguridad administradas externamente
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider
+- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback
+- **Per-Model Pricing Configuration** — Configurable prices per model
+- **Usage Statistics Per API Key** — Request count and last-used timestamp per key
+- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency
-
-🌍 12. "La interfaz es solo en inglés y mi equipo no habla inglés"
+
-Los equipos en países que no hablan inglés, especialmente en América Latina, Asia y Europa, tienen dificultades con las interfaces solo en inglés. Las barreras del idioma reducen la adopción y aumentan los errores de configuración.
+
+🐛 10. "I can't diagnose errors and problems in AI calls"
-**Cómo lo resuelve OmniRoute:**
+When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error.
--**Panel i18n — 30 idiomas**— Las más de 500 teclas traducidas, incluidas árabe, búlgaro, danés, alemán, español, finlandés, francés, hebreo, hindi, húngaro, indonesio, italiano, japonés, coreano, malayo, holandés, noruego, polaco, portugués (PT/BR), rumano, ruso, eslovaco, sueco, tailandés, ucraniano, vietnamita, chino, filipino, inglés.
--**Soporte RTL**: soporte de derecha a izquierda para árabe y hebreo
--**README multilingüe**: 30 traducciones de documentación completa
--**Selector de idioma**: ícono de globo en el encabezado para cambiar en tiempo real
+**How OmniRoute solves it:**
-
-🔄 13. "Necesito más que chat: necesito incrustaciones, imágenes y audio"
+- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console
+- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter
+- **SQLite Proxy Logs** — Persistent logs that survive server restarts
+- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time)
+- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing
+- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count
+- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage.
-La IA no es solo completar un chat. Los desarrolladores necesitan generar imágenes, transcribir audio, crear incrustaciones para RAG, reclasificar documentos y moderar contenido. Cada API tiene un punto final y un formato diferentes.
+
-**Cómo lo resuelve OmniRoute:**
+
+🏗️ 11. "Deploying and maintaining the gateway is complex"
--**Incrustaciones**— `/v1/embeddings` con 6 proveedores y más de 9 modelos
--**Generación de imágenes**— `/v1/images/generaciones` con 10 proveedores y más de 20 modelos (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
--**Texto a vídeo**— `/v1/videos/generaciones` — ComfyUI (AnimateDiff, SVD) y SD WebUI
--**Texto a música**— `/v1/music/generaciones` — ComfyUI (Audio estable abierto, MusicGen)
--**Transcripción de audio**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
--**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + proveedores existentes
--**Moderaciones**— `/v1/moderaciones` — Comprobaciones de seguridad del contenido
--**Reclasificación**— `/v1/rerank` — Reclasificación de relevancia del documento
--**API de respuestas**: compatibilidad total con `/v1/responses` para Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction.
-
-🧪 14. "No tengo forma de probar y comparar la calidad entre modelos"
+**How OmniRoute solves it:**
-Los desarrolladores quieren saber qué modelo es mejor para su caso de uso (código, traducción, razonamiento), pero comparar manualmente es lento. No existen herramientas de evaluación integradas.
+- **npm global install** — `npm install -g omniroute && omniroute` — done
+- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi)
+- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw)
+- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode
+- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking)
+- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers
+- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups
-**Cómo lo resuelve OmniRoute:**
+
--**Evaluaciones LLM**: pruebas de conjunto dorado con 10 casos precargados que cubren saludos, matemáticas, geografía, generación de código, cumplimiento de JSON, traducción, rebajas y rechazo de seguridad.
--**4 estrategias de coincidencia**: `exact`, `contains`, `regex`, `custom` (función JS)
--**Translator Playground Test Bench**: pruebas por lotes con múltiples entradas y resultados esperados, comparación entre proveedores
--**Chat Tester**: recorrido completo de ida y vuelta con representación de respuesta visual
--**Live Monitor**: flujo en tiempo real de todas las solicitudes que fluyen a través del proxy
+
+🌍 12. "The interface is English-only and my team doesn't speak English"
-
-📈 15. "Necesito escalar sin perder rendimiento"
+Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors.
-A medida que crece el volumen de solicitudes, sin almacenar en caché las mismas preguntas generan costos duplicados. Sin idempotencia, las solicitudes duplicadas desperdician el procesamiento. Se deben respetar los límites de tarifas por proveedor.
+**How OmniRoute solves it:**
-**Cómo lo resuelve OmniRoute:**
+- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English
+- **RTL Support** — Right-to-left support for Arabic and Hebrew
+- **Multi-Language READMEs** — 30 complete documentation translations
+- **Language Selector** — Globe icon in header for real-time switching
--**Caché semántica**: la caché de dos niveles (firma + semántica) reduce el costo y la latencia
--**Solicitud de idempotencia**: ventana de deduplicación de 5 segundos para solicitudes idénticas
--**Detección de límite de velocidad**: RPM por proveedor, intervalo mínimo y seguimiento simultáneo máximo
--**Límites de velocidad editables**: valores predeterminados configurables en Configuración → Resiliencia con persistencia
--**Caché de validación de clave API**: caché de 3 niveles para rendimiento de producción
--**Panel de estado con telemetría**: latencia p50/p95/p99, estadísticas de caché, tiempo de actividad
+
-
-🤖 16. "Quiero controlar el comportamiento del modelo globalmente"
+
+🔄 13. "I need more than chat — I need embeddings, images, audio"
-Desarrolladores que quieran todas las respuestas en un idioma específico, con un tono específico o quieran limitar los tokens de razonamiento. Configurar esto en cada herramienta/solicitud no es práctico.
+AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format.
-**Cómo lo resuelve OmniRoute:**
+**How OmniRoute solves it:**
--**Inyección de aviso del sistema**: aviso global aplicado a todas las solicitudes
--**Thinking Budget Validation**: control de asignación de tokens de razonamiento por solicitud (transferencia, automática, personalizada, adaptativa)
--**9 estrategias de enrutamiento**: estrategias globales que determinan cómo se distribuyen las solicitudes
--**Enrutador comodín**: los patrones `proveedor/*` se enrutan dinámicamente a cualquier proveedor
--**Activar/desactivar combinación de alternar**: alterna combinaciones directamente desde el panel
--**Alternar proveedor**: activa/desactiva todas las conexiones de un proveedor con un solo clic
--**Proveedores bloqueados**: excluye proveedores específicos de la lista `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models
+- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
+- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI
+- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
+- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
+- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers
+- **Moderations** — `/v1/moderations` — Content safety checks
+- **Reranking** — `/v1/rerank` — Document relevance reranking
+- **Responses API** — Full `/v1/responses` support for Codex
-
-🧰 17. "Necesito herramientas MCP como capacidades de producto de primera clase"
+
-Muchas puertas de enlace de IA exponen MCP solo como un detalle de implementación oculto. Los equipos necesitan una capa operativa visible y manejable.
+
+🧪 14. "I have no way to test and compare quality across models"
-**Cómo lo resuelve OmniRoute:**
+Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist.
-- MCP aparece en la pestaña de navegación del panel y protocolo de punto final
-- Página de gestión de MCP dedicada con procesos, herramientas, alcances y auditoría
-- Inicio rápido integrado para `omniroute --mcp` e incorporación de clientes
+**How OmniRoute solves it:**
-
-🧠 18. "Necesito orquestación A2A con rutas de tareas de sincronización y transmisión"
+- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal
+- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function)
+- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison
+- **Chat Tester** — Full round-trip with visual response rendering
+- **Live Monitor** — Real-time stream of all requests flowing through the proxy
-Los flujos de trabajo de los agentes necesitan respuestas directas y una ejecución continua y continua con control del ciclo de vida.
+
-**Cómo lo resuelve OmniRoute:**
+
+📈 15. "I need to scale without losing performance"
-- Punto final A2A JSON-RPC (`POST /a2a`) con `mensaje/envío` y `mensaje/transmisión`
-- Transmisión SSE con propagación del estado terminal
-- API de ciclo de vida de tareas para `tareas/obtener` y `tareas/cancelar`
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected.
-
-🛰️ 19. "Necesito un estado real del proceso MCP, no un estado adivinado"
+**How OmniRoute solves it:**
-Los equipos operativos necesitan saber si MCP está realmente activo, no solo si se puede acceder a una API.
+- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency
+- **Request Idempotency** — 5s deduplication window for identical requests
+- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking
+- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence
+- **API Key Validation Cache** — 3-tier cache for production performance
+- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime
-**Cómo lo resuelve OmniRoute:**
+
-- Archivo de latidos en tiempo de ejecución con PID, marcas de tiempo, transporte, recuento de herramientas y modo de alcance
-- API de estado de MCP que combina latidos + actividad reciente
-- Tarjetas de estado de la interfaz de usuario para el proceso/tiempo de actividad/actualización de latidos
+
+🤖 16. "I want to control model behavior globally"
-
-📋 20. "Necesito ejecución de herramienta MCP auditable"
+Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical.
-Cuando las herramientas modifican la configuración o desencadenan acciones de operaciones, los equipos necesitan trazabilidad forense.
+**How OmniRoute solves it:**
-**Cómo lo resuelve OmniRoute:**
+- **System Prompt Injection** — Global prompt applied to all requests
+- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive)
+- **9 Routing Strategies** — Global strategies that determine how requests are distributed
+- **Wildcard Router** — `provider/*` patterns route dynamically to any provider
+- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard
+- **Provider Toggle** — Enable/disable all connections for a provider with one click
+- **Blocked Providers** — Exclude specific providers from `/v1/models` listing
-- Registro de auditoría respaldado por SQLite para llamadas a herramientas MCP
-- Filtros por herramienta, éxito/fracaso, clave API y paginación
-- Tabla de auditoría del panel + puntos finales de estadísticas para automatización
+
-
-🔐 21. "Necesito permisos MCP con alcance por integración"
+
+🧰 17. "I need MCP tools as first-class product capabilities"
-Los diferentes clientes deberían tener acceso con privilegios mínimos a las categorías de herramientas.
+Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer.
-**Cómo lo resuelve OmniRoute:**
+**How OmniRoute solves it:**
-- 10 alcances MCP granulares para acceso controlado a herramientas
-- Aplicación del alcance y visibilidad en la interfaz de usuario de gestión de MCP
-- Postura predeterminada segura para herramientas operativas
+- MCP appears in the dashboard navigation and endpoint protocol tab
+- Dedicated MCP management page with process, tools, scopes, and audit
+- Built-in quick-start for `omniroute --mcp` and client onboarding
-
-⚙️ 22. "Necesito controles operativos sin redistribuir"
+
-Los equipos necesitan cambios rápidos en el tiempo de ejecución durante incidentes o eventos de costos.
+
+🧠 18. "I need A2A orchestration with sync + stream task paths"
-**Cómo lo resuelve OmniRoute:**
+Agent workflows need both direct replies and long-running streamed execution with lifecycle control.
-- Cambie la activación combinada directamente desde el panel de MCP
-- Aplicar perfiles de resiliencia de paquetes de políticas predefinidos
-- Restablecer el estado del disyuntor desde el mismo panel de operaciones.
+**How OmniRoute solves it:**
-
-🔄 23. "Necesito visibilidad y cancelación del ciclo de vida de la tarea A2A en vivo"
+- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream`
+- SSE streaming with terminal state propagation
+- Task lifecycle APIs for `tasks/get` and `tasks/cancel`
-Sin visibilidad del ciclo de vida, los incidentes de tareas se vuelven difíciles de clasificar.
+
-**Cómo lo resuelve OmniRoute:**
+
+🛰️ 19. "I need real MCP process health, not guessed status"
-- Listado de tareas/filtrado por estado/habilidad con paginación
-- Profundización en metadatos, eventos y artefactos de tareas
-- Punto final de cancelación de tarea y acción de UI con confirmación
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable.
-
-🌊 24. "Necesito métricas de transmisión activas para la carga A2A"
+**How OmniRoute solves it:**
-Los flujos de trabajo de streaming requieren información operativa sobre la simultaneidad y las conexiones en vivo.
+- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode
+- MCP status API combining heartbeat + recent activity
+- UI status cards for process/uptime/heartbeat freshness
-**Cómo lo resuelve OmniRoute:**
+
-- Contadores de flujo activos integrados en el estado A2A
-- Marca de tiempo de la última tarea y recuentos por estado
-- Tarjetas de tablero A2A para monitoreo de operaciones en tiempo real
+
+📋 20. "I need auditable MCP tool execution"
-
-🪪 25. "Necesito un descubrimiento de agentes estándar para los clientes"
+When tools mutate config or trigger ops actions, teams need forensic traceability.
-Los clientes y orquestadores externos necesitan metadatos legibles por máquina para la incorporación.
+**How OmniRoute solves it:**
-**Cómo lo resuelve OmniRoute:**
+- SQLite-backed audit logging for MCP tool calls
+- Filters by tool, success/failure, API key, and pagination
+- Dashboard audit table + stats endpoints for automation
-- Tarjeta de agente expuesta en `/.well-known/agent.json`
-- Capacidades y habilidades mostradas en la interfaz de usuario de gestión.
-- La API de estado A2A incluye metadatos de descubrimiento para la automatización
+
-
-🧭 26. "Necesito capacidad de descubrimiento del protocolo en la UX del producto"
+
+🔐 21. "I need scoped MCP permissions per integration"
-Si los usuarios no pueden descubrir las superficies de protocolo, la calidad de la adopción y el soporte disminuye.
+Different clients should have least-privilege access to tool categories.
-**Cómo lo resuelve OmniRoute:**
+**How OmniRoute solves it:**
-- Página consolidada de**Puntos finales**con pestañas para Proxy, MCP, A2A y API Endpoints
-- El estado del servicio en línea alterna (en línea/fuera de línea) para MCP y A2A
-- Enlaces desde la descripción general a pestañas de administración dedicadas
+- 10 granular MCP scopes for controlled tool access
+- Scope enforcement and visibility in MCP management UI
+- Safe default posture for operational tooling
-
-🧪 27. "Necesito validación de protocolo de un extremo a otro con clientes reales"
+
-Las pruebas simuladas no son suficientes para validar la compatibilidad del protocolo antes del lanzamiento.
+
+⚙️ 22. "I need operational controls without redeploying"
-**Cómo lo resuelve OmniRoute:**
+Teams need quick runtime changes during incidents or cost events.
-- Suite E2E que inicia la aplicación y utiliza transporte de cliente MCP SDK real
-- Pruebas de cliente A2A para descubrimiento, envío, transmisión, obtención y cancelación de flujos
-- Verificar las afirmaciones con las API de auditoría MCP y tareas A2A.
+**How OmniRoute solves it:**
-
-📡 28. "Necesito observabilidad unificada en todas las interfaces"
+- Switch combo activation directly from MCP dashboard
+- Apply resilience profiles from pre-defined policy packs
+- Reset circuit breaker state from the same operations panel
-Dividir la observabilidad por protocolo crea puntos ciegos y MTTR más largos.
+
-**Cómo lo resuelve OmniRoute:**
+
+🔄 23. "I need live A2A task lifecycle visibility and cancellation"
-- Paneles/registros/análisis unificados en un solo producto
-- Salud + auditoría + solicitud de telemetría en capas OpenAI, MCP y A2A
-- API operativas para estado y automatización.
+Without lifecycle visibility, task incidents become hard to triage.
-
-💼 29. "Necesito un tiempo de ejecución para proxy + herramientas + orquestación de agentes"
+**How OmniRoute solves it:**
-La ejecución de muchos servicios separados aumenta los costos operativos y los modos de falla.
+- Task listing/filtering by state/skill with pagination
+- Drill-down on task metadata, events, and artifacts
+- Task cancellation endpoint and UI action with confirmation
-**Cómo lo resuelve OmniRoute:**
+
-- Proxy compatible con OpenAI, servidor MCP y servidor A2A en una sola pila
-- Autenticación compartida, resiliencia, almacenamiento de datos y observabilidad.
-- Modelo de política consistente en todas las superficies de interacción.
+
+🌊 24. "I need active stream metrics for A2A load"
-
-🚀 30. "Necesito enviar flujos de trabajo agentes sin expansión de códigos adhesivos"
+Streaming workflows require operational insight into concurrency and live connections.
-Los equipos pierden velocidad al unir múltiples scripts y servicios ad hoc.
+**How OmniRoute solves it:**
-**Cómo lo resuelve OmniRoute:**
+- Active stream counters integrated into A2A status
+- Last task timestamp and per-state counts
+- A2A dashboard cards for real-time ops monitoring
-- Estrategia de endpoint unificada para clientes y agentes
-- UI de gestión de protocolos integradas y rutas de validación de humo
-- Fundamentos listos para producción (seguridad, registro, resiliencia, respaldo)
+
+
+
+🪪 25. "I need standard agent discovery for clients"
+
+External clients and orchestrators need machine-readable metadata for onboarding.
+
+**How OmniRoute solves it:**
+
+- Agent Card exposed at `/.well-known/agent.json`
+- Capabilities and skills shown in management UI
+- A2A status API includes discovery metadata for automation
+
+
+
+
+🧭 26. "I need protocol discoverability in the product UX"
+
+If users cannot discover protocol surfaces, adoption and support quality drop.
+
+**How OmniRoute solves it:**
+
+- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints
+- Inline service status toggles (Online/Offline) for MCP and A2A
+- Links from overview to dedicated management tabs
+
+
+
+
+🧪 27. "I need end-to-end protocol validation with real clients"
+
+Mock tests are not enough to validate protocol compatibility before release.
+
+**How OmniRoute solves it:**
+
+- E2E suite that boots app and uses real MCP SDK client transport
+- A2A client tests for discovery, send, stream, get, and cancel flows
+- Cross-check assertions against MCP audit and A2A tasks APIs
+
+
+
+
+📡 28. "I need unified observability across all interfaces"
+
+Splitting observability by protocol creates blind spots and longer MTTR.
+
+**How OmniRoute solves it:**
+
+- Unified dashboards/logs/analytics in one product
+- Health + audit + request telemetry across OpenAI, MCP, and A2A layers
+- Operational APIs for status and automation
+
+
+
+
+💼 29. "I need one runtime for proxy + tools + agent orchestration"
+
+Running many separate services increases operational cost and failure modes.
+
+**How OmniRoute solves it:**
+
+- OpenAI-compatible proxy, MCP server, and A2A server in one stack
+- Shared auth, resilience, data store, and observability
+- Consistent policy model across all interaction surfaces
+
+
+
+
+🚀 30. "I need to ship agentic workflows without glue-code sprawl"
+
+Teams lose velocity when stitching multiple ad-hoc services and scripts.
+
+**How OmniRoute solves it:**
+
+- Unified endpoint strategy for clients and agents
+- Built-in protocol management UIs and smoke validation paths
+- Production-ready foundations (security, logging, resilience, backup)
+
+
### Example Playbooks (Integrated Use Cases)
-**Libro de estrategias A: maximizar la suscripción paga + copia de seguridad económica**```txt
+**Playbook A: Maximize paid subscription + cheap backup**
+
+```txt
Combo: "maximize-claude"
1. cc/claude-opus-4-6
2. glm/glm-4.7
@@ -607,21 +690,23 @@ Combo: "maximize-claude"
Monthly cost: $20 + small backup spend
Outcome: higher quality, near-zero interruption
-````
+```
-**Libro de estrategias B: pila de codificación de costo cero**```txt
+**Playbook B: Zero-cost coding stack**
+
+```txt
Combo: "free-forever"
-
-1. gc/gemini-3-flash
-2. if/kimi-k2-thinking
-3. qw/qwen3-coder-plus
+ 1. gc/gemini-3-flash
+ 2. if/kimi-k2-thinking
+ 3. qw/qwen3-coder-plus
Monthly cost: $0
Outcome: stable free coding workflow
+```
-````
+**Playbook C: 24/7 always-on fallback chain**
-**Libro de estrategias C: cadena alternativa siempre disponible las 24 horas del día, los 7 días de la semana**```txt
+```txt
Combo: "always-on"
1. cc/claude-opus-4-6
2. cx/gpt-5.2-codex
@@ -630,122 +715,134 @@ Combo: "always-on"
5. if/kimi-k2-thinking
Outcome: deep fallback depth for deadline-critical workloads
-````
+```
-**Libro de jugadas D: Operaciones del agente con MCP + A2A**```txt
+**Playbook D: Agent ops with MCP + A2A**
-1. Start MCP transport (`omniroute --mcp`) for tool-driven operations
-2. Run A2A tasks via `message/send` and `message/stream`
-3. Observe via /dashboard/endpoint (MCP and A2A tabs)
-4. Toggle services via inline status controls
-
-````
+```txt
+1) Start MCP transport (`omniroute --mcp`) for tool-driven operations
+2) Run A2A tasks via `message/send` and `message/stream`
+3) Observe via /dashboard/endpoint (MCP and A2A tabs)
+4) Toggle services via inline status controls
+```
---
## 🆓 Start Free — Zero Configuration Cost
-> Configure la codificación AI en minutos a**$0/mes**. Conecte estas cuentas gratuitas y utilice el combo**Free Stack**integrado.
+> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo.
-| Paso | Acción | Proveedores desbloqueados |
+| Step | Action | Providers Unlocked |
| ---- | -------------------------------------------------- | ------------------------------------------------------------------ |
-| 1 | Conectar**Kiro**(ID de AWS Builder OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**ilimitado**|
-| 2 | Conectar**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**ilimitado**|
-| 3 | Conectar**Qwen**(Código de dispositivo) | qwen3-coder-plus, qwen3-coder-flash... —**ilimitado**|
-| 4 | Conectar**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/mes gratis**|
-| 5 | `/dashboard/combos` →**Plantilla de pila gratuita ($0)**| Round-robin todos los proveedores gratuitos automáticamente |
+| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** |
+| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** |
+| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** |
+| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** |
+| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically |
-**Apunte cualquier IDE/CLI a:**`http://localhost:20128/v1` · Clave API: `any-string` · Listo.
+**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done.
->**Cobertura adicional opcional (también gratuita):**Clave API Groq (30 RPM gratis), NVIDIA NIM (40 RPM gratis, más de 70 modelos), Cerebras (1 millón de tok/día), clave API LongCat (¡50 millones de tokens/día!), Cloudflare Workers AI (10 000 neuronas/día, más de 50 modelos).## Inicio Rápido
+> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models).
+
+## Inicio Rápido
### 1) Install and run
```bash
npm install -g omniroute
omniroute
-````
+```
-> **usuarios de pnpm:**Ejecute `pnpm aprobar-builds -g` después de la instalación para habilitar los scripts de compilación nativos requeridos por `better-sqlite3` y `@swc/core`:
+> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`:
>
-> ```golpecito
-> pnpm instalar -g omniruta
-> pnpm aprobar-builds -g # Seleccionar todos los paquetes → aprobar
-> omniruta
+> ```bash
+> pnpm install -g omniroute
+> pnpm approve-builds -g # Select all packages → approve
+> omniroute
> ```
-El panel se abre en `http://localhost:20128` y la URL base de API es `http://localhost:20128/v1`.
+Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`.
-| Comando | Descripción |
-| ------------------------ | --------------------------------------------------------------- |
-| `omniruta` | Iniciar servidor (`PORT=20128`, API y panel en el mismo puerto) |
-| `omniruta --puerto 3000` | Establezca el puerto canónico/API en 3000 |
-| `omniruta --mcp` | Inicie el servidor MCP (transporte stdio) |
-| `omniroute --no-abierto` | No abrir automáticamente el navegador |
-| `omniroute --ayuda` | Mostrar ayuda |
+| Command | Description |
+| ----------------------- | ----------------------------------------------------------- |
+| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) |
+| `omniroute --port 3000` | Set canonical/API port to 3000 |
+| `omniroute --mcp` | Start MCP server (stdio transport) |
+| `omniroute --no-open` | Don't auto-open browser |
+| `omniroute --help` | Show help |
-Modo de puerto dividido opcional:```bash
+Optional split-port mode:
+
+```bash
PORT=20128 DASHBOARD_PORT=20129 omniroute
-
-# API: http://localhost:20128/v1
-
+# API: http://localhost:20128/v1
# Dashboard: http://localhost:20129
-
-````
+```
### Long-Running Streaming Timeouts
-Para la mayoría de las implementaciones, solo necesita:
+For most deployments, you only need:
-| Variables | Predeterminado | Propósito |
-| ------------------------ | ----------------------- | --------------------------------------------------------------------------------------------------------------------- |
-| `REQUEST_TIMEOUT_MS` | `600000` | Línea de base compartida para recuperación ascendente, tiempos de espera de Undici ocultos, solicitudes de huellas digitales TLS y tiempos de espera de proxy/solicitud de puente API |
-| `STREAM_IDLE_TIMEOUT_MS` | hereda `REQUEST_TIMEOUT_MS` | Brecha máxima entre fragmentos de transmisión antes de que OmniRoute cancele la transmisión SSE |
+| Variable | Default | Purpose |
+| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- |
+| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts |
+| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream |
-Se conserva la compatibilidad con versiones anteriores: `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` y otras variables de tiempo de espera por capa aún funcionan y anulan la línea base compartida.
+Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline.
-Las anulaciones avanzadas están disponibles si necesita un control más preciso:| Variables | Predeterminado | Propósito |
-| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------- |
-| `FETCH_TIMEOUT_MS` | hereda `REQUEST_TIMEOUT_MS` | Tiempo de espera total de solicitudes ascendentes utilizado por la señal de aborto de recuperación principal |
-| `FETCH_HEADERS_TIMEOUT_MS` | hereda `FETCH_TIMEOUT_MS` | Límite de tiempo de Undici para recibir encabezados de respuesta ascendentes |
-| `FETCH_BODY_TIMEOUT_MS` | hereda `FETCH_TIMEOUT_MS` | Límite de tiempo undici entre fragmentos de cuerpo ascendentes (`0` lo desactiva) |
-| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Tiempo de espera de conexión TCP de Undici |
-| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici tiempo de espera del socket de mantenimiento activo inactivo |
-| `TLS_CLIENT_TIMEOUT_MS` | hereda `FETCH_TIMEOUT_MS` | Tiempo de espera para solicitudes de huellas digitales TLS realizadas a través de `wreq-js` |
-| `API_BRIDGE_PROXY_TIMEOUT_MS` | hereda `REQUEST_TIMEOUT_MS` o `30000` | Tiempo de espera para el reenvío de proxy `/v1` desde el puerto API al puerto del panel |
-| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Tiempo de espera de solicitud entrante en el servidor puente API |
-| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Tiempo de espera del encabezado entrante en el servidor puente API |
-| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Tiempo de espera de mantenimiento de actividad en el servidor puente API |
-| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Tiempo de espera de inactividad del socket en el servidor puente API (`0` lo deshabilita) |
+Advanced overrides are available if you need finer control:
-Si ejecuta OmniRoute detrás de Nginx, Caddy, Cloudflare u otro proxy inverso, asegúrese de que el proxy
-Los tiempos de espera también son mayores que los tiempos de espera de transmisión/recuperación de OmniRoute.### 2) Connect providers and create your API key
+| Variable | Default | Purpose |
+| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- |
+| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal |
+| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers |
+| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) |
+| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout |
+| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout |
+| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` |
+| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port |
+| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server |
+| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server |
+| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server |
+| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) |
-1. Abra Panel → `Proveedores` y conecte al menos un proveedor (clave OAuth o API).
-2. Abra Panel → `Endpoints` y cree una clave API.
-3. (Opcional) Abra el Panel → `Combos` y configure su cadena alternativa.### 3) Point your coding tool to OmniRoute
+If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy
+timeouts are also higher than your OmniRoute stream/fetch timeouts.
+
+### 2) Connect providers and create your API key
+
+1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key).
+2. Open Dashboard → `Endpoints` and create an API key.
+3. (Optional) Open Dashboard → `Combos` and set your fallback chain.
+
+### 3) Point your coding tool to OmniRoute
```txt
Base URL: http://localhost:20128/v1
API Key: [copy from Endpoint page]
Model: if/kimi-k2-thinking (or any provider/model prefix)
-````
+```
-Funciona con Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode y SDK compatibles con OpenAI.### 4) Enable and validate protocols (v2.0)
+Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs.
-**MCP (para operaciones basadas en herramientas):**```bash
+### 4) Enable and validate protocols (v2.0)
+
+**MCP (for tool-driven operations):**
+
+```bash
omniroute --mcp
+```
-````
+Then connect your MCP client over `stdio` and test tools like:
-Luego conecte su cliente MCP a través de `stdio` y pruebe herramientas como:
+- `omniroute_get_health`
+- `omniroute_list_combos`
--`omniroute_get_health`
--`omniroute_list_combos`
+**A2A (for agent-to-agent workflows):**
-**A2A (para flujos de trabajo de agente a agente):**```bash
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
```bash
curl -X POST http://localhost:20128/a2a \
@@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \
npm run test:protocols:e2e
```
-Esta suite valida flujos de clientes MCP y A2A reales frente a una aplicación en ejecución.### Alternative: run from source
+This suite validates real MCP and A2A client flows against a running app.
+
+### Alternative: run from source
```bash
cp .env.example .env
@@ -767,13 +866,13 @@ npm install
PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev
```
-
-Void Linux (plantilla `xbps-src`)
+
+Void Linux (`xbps-src` template)
-Para los usuarios de Void Linux, pueden crear un paquete nativo usando `xbps-src`. Guarde este bloque como `srcpkgs/omniroute/template`:```bash
+For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`:
+```bash
# Template file for 'omniroute'
-
pkgname=omniroute
version=3.4.1
revision=1
@@ -785,7 +884,7 @@ license="MIT"
homepage="https://github.com/diegosouzapw/OmniRoute"
distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz"
checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b
-system_accounts="\_omniroute"
+system_accounts="_omniroute"
omniroute_homedir="/var/lib/omniroute"
export NODE_ENV=production
export npm_config_engine_strict=false
@@ -793,71 +892,70 @@ export npm_config_loglevel=error
export npm_config_fund=false
export npm_config_audit=false
-do_build() { # Determine target CPU arch for node-gyp
-local \_gyp_arch
-case "$XBPS_TARGET_MACHINE" in
-aarch64*) \_gyp_arch=arm64 ;;
-armv7*|armv6*) \_gyp_arch=arm ;;
-i686*) \_gyp_arch=ia32 ;;
-\*) \_gyp_arch=x64 ;;
-esac
+do_build() {
+ # Determine target CPU arch for node-gyp
+ local _gyp_arch
+ case "$XBPS_TARGET_MACHINE" in
+ aarch64*) _gyp_arch=arm64 ;;
+ armv7*|armv6*) _gyp_arch=arm ;;
+ i686*) _gyp_arch=ia32 ;;
+ *) _gyp_arch=x64 ;;
+ esac
- # 1) Install all deps – skip scripts (no network in do_build, native modules
- # compiled separately below; better-sqlite3 is serverExternalPackage so
- # Next.js does not execute it during next build)
- NODE_ENV=development npm ci --ignore-scripts
+ # 1) Install all deps – skip scripts (no network in do_build, native modules
+ # compiled separately below; better-sqlite3 is serverExternalPackage so
+ # Next.js does not execute it during next build)
+ NODE_ENV=development npm ci --ignore-scripts
- # 2) Build the Next.js standalone bundle
- npm run build
+ # 2) Build the Next.js standalone bundle
+ npm run build
- # 3) Copy static assets into standalone
- cp -r .next/static .next/standalone/.next/static
- [ -d public ] && cp -r public .next/standalone/public || true
+ # 3) Copy static assets into standalone
+ cp -r .next/static .next/standalone/.next/static
+ [ -d public ] && cp -r public .next/standalone/public || true
- # 4) Compile better-sqlite3 native binding for the target architecture.
- # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
- # without npm altering them.
- local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
- (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
+ # 4) Compile better-sqlite3 native binding for the target architecture.
+ # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
+ # without npm altering them.
+ local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
+ (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
- # 5) Place the compiled binding into the standalone bundle
- local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
- mkdir -p "$_bs3_release"
- cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
+ # 5) Place the compiled binding into the standalone bundle
+ local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
+ mkdir -p "$_bs3_release"
+ cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
- # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
- # so sharp is not used at runtime; x64 .so files would break aarch64 strip
- rm -rf .next/standalone/node_modules/@img
-
- # 7) Copy pino runtime deps omitted by Next.js static analysis:
- # pino-abstract-transport – required by pino's worker thread
- # split2 – dep of pino-abstract-transport
- # process-warning – dep of pino itself
- for _mod in pino-abstract-transport split2 process-warning; do
- cp -r "node_modules/$_mod" .next/standalone/node_modules/
- done
+ # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
+ # so sharp is not used at runtime; x64 .so files would break aarch64 strip
+ rm -rf .next/standalone/node_modules/@img
+ # 7) Copy pino runtime deps omitted by Next.js static analysis:
+ # pino-abstract-transport – required by pino's worker thread
+ # split2 – dep of pino-abstract-transport
+ # process-warning – dep of pino itself
+ for _mod in pino-abstract-transport split2 process-warning; do
+ cp -r "node_modules/$_mod" .next/standalone/node_modules/
+ done
}
do_check() {
-npm run test:unit
+ npm run test:unit
}
do_install() {
-vmkdir usr/lib/omniroute/.next
+ vmkdir usr/lib/omniroute/.next
- vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
+ vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
- # Prevent removal of empty Next.js app router dirs by the post-install hook
- for _d in \
- .next/standalone/.next/server/app/dashboard \
- .next/standalone/.next/server/app/dashboard/settings \
- .next/standalone/.next/server/app/dashboard/providers; do
- touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
- done
-
- cat > "${WRKDIR}/omniroute" <<'EOF'
+ # Prevent removal of empty Next.js app router dirs by the post-install hook
+ for _d in \
+ .next/standalone/.next/server/app/dashboard \
+ .next/standalone/.next/server/app/dashboard/settings \
+ .next/standalone/.next/server/app/dashboard/providers; do
+ touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
+ done
+ cat > "${WRKDIR}/omniroute" <<'EOF'
#!/bin/sh
export PORT="${PORT:-20128}"
export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}"
@@ -869,10 +967,9 @@ EOF
}
post_install() {
-vlicense LICENSE
+ vlicense LICENSE
}
-
-````
+```
@@ -880,9 +977,11 @@ vlicense LICENSE
## 🐳 Docker
-OmniRoute está disponible como imagen pública de Docker en [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
+OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
-**Ejecución rápida:**```bash
+**Quick run:**
+
+```bash
docker run -d \
--name omniroute \
--restart unless-stopped \
@@ -890,85 +989,96 @@ docker run -d \
-p 20128:20128 \
-v omniroute-data:/app/data \
diegosouzapw/omniroute:latest
-````
+```
-**Con archivo de entorno:**```bash
+**With environment file:**
+```bash
# Copy and edit .env first
-
cp .env.example .env
docker run -d \
- --name omniroute \
- --restart unless-stopped \
- --stop-timeout 40 \
- --env-file .env \
- -p 20128:20128 \
- -v omniroute-data:/app/data \
- diegosouzapw/omniroute:latest
+ --name omniroute \
+ --restart unless-stopped \
+ --stop-timeout 40 \
+ --env-file .env \
+ -p 20128:20128 \
+ -v omniroute-data:/app/data \
+ diegosouzapw/omniroute:latest
+```
-````
+**Using Docker Compose:**
-**Usando Docker Compose:**```bash
+```bash
# Base profile (no CLI tools)
docker compose --profile base up -d
# CLI profile (Claude Code, Codex, OpenClaw built-in)
docker compose --profile cli up -d
-````
+```
-El soporte del panel para implementaciones de Docker ahora incluye un**Cloudflare Quick Tunnel**con un solo clic en "Panel → Endpoints". La primera habilitación descarga `cloudflared` solo cuando es necesario, inicia un túnel temporal hacia su punto final `/v1` actual y muestra la URL `https://*.trycloudflare.com/v1` generada directamente debajo de su URL pública normal.
+Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL.
-Notas:
+Notes:
-- Las URL de Quick Tunnel son temporales y cambian después de cada reinicio.
-- Los túneles rápidos no se restauran automáticamente después de reiniciar OmniRoute o un contenedor. Vuelva a habilitarlos desde el panel cuando sea necesario.
-- La instalación administrada actualmente es compatible con Linux, macOS y Windows en `x64`/`arm64`.
-- Los túneles rápidos administrados utilizan de forma predeterminada el transporte HTTP/2 para evitar ruidosas advertencias de búfer QUIC UDP en entornos de contenedores restringidos. Configure `CLOUDFLARED_PROTOCOL=quic` o `auto` si desea un transporte diferente.
-- Las imágenes de Docker agrupan las raíces de CA del sistema y las pasan a "cloudflared" administrado, lo que evita fallas de confianza de TLS cuando el túnel se inicia dentro del contenedor.
-- SQLite se ejecuta en modo WAL. Se debe permitir que `docker stop` finalice para que OmniRoute pueda verificar los últimos cambios en `storage.sqlite`.
-- Los archivos Compose incluidos ya establecen un período de gracia de parada de 40 segundos. Si ejecuta la imagen directamente, mantenga `--stop-timeout 40` (o similar) para que las paradas manuales no interrumpan la limpieza del apagado.
-- Configure `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` si desea que OmniRoute use un binario existente en lugar de descargar uno.
+- Quick Tunnel URLs are temporary and change after every restart.
+- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed.
+- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`.
+- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport.
+- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container.
+- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`.
+- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup.
+- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one.
-**Usando Docker Compose con Caddy (HTTPS Auto-TLS):**
+**Using Docker Compose with Caddy (HTTPS Auto-TLS):**
-OmniRoute se puede exponer de forma segura mediante el aprovisionamiento SSL automático de Caddy. Asegúrese de que el registro DNS A de su dominio apunte a la IP de su servidor.```yaml
+OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP.
+
+```yaml
services:
-omniroute:
-image: diegosouzapw/omniroute:latest
-container_name: omniroute
-restart: unless-stopped
-volumes: - omniroute-data:/app/data
-environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com
+ omniroute:
+ image: diegosouzapw/omniroute:latest
+ container_name: omniroute
+ restart: unless-stopped
+ volumes:
+ - omniroute-data:/app/data
+ environment:
+ - PORT=20128
+ - NEXT_PUBLIC_BASE_URL=https://your-domain.com
-caddy:
-image: caddy:latest
-container_name: caddy
-restart: unless-stopped
-ports: - "80:80" - "443:443"
-command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
+ caddy:
+ image: caddy:latest
+ container_name: caddy
+ restart: unless-stopped
+ ports:
+ - "80:80"
+ - "443:443"
+ command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
volumes:
-omniroute-data:
+ omniroute-data:
+```
-````
-
-| Imagen | Etiqueta | Tamaño | Descripción |
+| Image | Tag | Size | Description |
| ------------------------ | -------- | ------ | --------------------- |
-| `diegosouzapw/omniroute` | `último` | ~250MB | Última versión estable |
-| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Versión actual |---
+| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release |
+| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version |
+
+---
## 🖥️ Desktop App — Offline & Always-On
-> 🆕**¡NUEVO!**OmniRoute ahora está disponible como**aplicación de escritorio nativa**para Windows, macOS y Linux.
+> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux.
-Ejecute OmniRoute como una aplicación de escritorio independiente: no se requiere terminal, navegador ni Internet para los modelos locales. La aplicación basada en Electron incluye:
+Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes:
-- 🖥️**Ventana nativa**: ventana de aplicación dedicada con integración de la bandeja del sistema
-- 🔄**Inicio automático**: inicie OmniRoute al iniciar sesión en el sistema
-- 🔔**Notificaciones nativas**: reciba alertas sobre el agotamiento de la cuota o problemas con el proveedor
-- ⚡**Instalación con un clic**: NSIS (Windows), DMG (macOS), AppImage (Linux)
-- 🌐**Modo sin conexión**: funciona completamente sin conexión con el servidor incluido### Inicio Rápido
+- 🖥️ **Native Window** — Dedicated app window with system tray integration
+- 🔄 **Auto-Start** — Launch OmniRoute on system login
+- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues
+- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux)
+- 🌐 **Offline Mode** — Works fully offline with bundled server
+
+### Inicio Rápido
```bash
# Development mode
@@ -979,308 +1089,370 @@ npm run electron:build # Current platform
npm run electron:build:win # Windows (.exe)
npm run electron:build:mac # macOS (.dmg) — x64 & arm64
npm run electron:build:linux # Linux (.AppImage)
-````
+```
### System Tray
-Cuando está minimizado, OmniRoute reside en la bandeja del sistema con acciones rápidas:
+When minimized, OmniRoute lives in your system tray with quick actions:
-- Abrir panel
-- Cambiar puerto del servidor
-- Salir de la aplicación
+- Open dashboard
+- Change server port
+- Quit application
-📖 Documentación completa: [`electron/README.md`](electron/README.md)---
+📖 Full documentation: [`electron/README.md`](electron/README.md)
+
+---
## 💰 Pricing at a Glance
-| Nivel | Proveedor | Costo | Restablecer cuota | Mejor para |
-| ------------------ | --------------------------------------- | -------------------------------------- | ----------------------------- | ---------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| **💳 SUSCRIPCIÓN** | Código Claude (Pro) | $20/mes | 5h + semanales | Ya suscrito |
-| | Códice (Plus/Pro) | $20-200/mes | 5h + semanales | Usuarios de OpenAI |
-| | Géminis CLI | **GRATIS** | 180K/mes + 1K/día | ¡Todos! |
-| | Copiloto de GitHub | $10-19/mes | Mensual | Usuarios de GitHub |
-| **🔑 CLAVE API** | NIM de NVIDIA | **GRATIS**(desarrollador para siempre) | ~40 RPM | Más de 70 modelos abiertos |
-| | Cerebras | **GRATIS**(1 millón de tok/día) | 60.000 TPM / 30 RPM | El más rápido del mundo |
-| | Groq | **GRATIS**(30 RPM) | 14,4K RPD | Llama/Gemma ultrarrápida |
-| | DeepSeek V3.2 | $0,27/$1,10 por 1 millón | Ninguno | Mejor razonamiento precio/calidad |
-| | xAI Grok-4 Rápido | **$0,20/$0,50 por 1M**🆕 | Ninguno | Llamada de herramienta + más rápida, ultrabaja |
-| | xAI Grok-4 (estándar) | $0,20/$1,50 por 1 millón 🆕 | Ninguno | Insignia de razonamiento de xAI |
-| | Mistral | Prueba gratuita + pago | Tarifa limitada | IA europea |
-| | Enrutador abierto | Pago por uso | Ninguno | Más de 100 modelos agregados. |
-| **💰 BARATO** | GLM-5 (vía Z.AI) 🆕 | 0,5 dólares/1 millón | Todos los días a las 10 a. m. | Salida de 128K, el buque insignia más nuevo |
-| | GLM-4.7 | 0,6 dólares/1 millón | Todos los días a las 10 a. m. | Respaldo presupuestario |
-| | MiniMax M2.5 🆕 | 0,3 $/1 millón de entrada | 5 horas rodantes | Razonamiento + tareas agentes |
-| | MiniMax M2.1 | 0,2 dólares/1 millón | 5 horas rodantes | Opción más barata |
-| | Kimi K2.5 (API Moonshot) 🆕 | Pago por uso | Ninguno | Acceso directo a la API Moonshot |
-| | Kimi K2 | $9/mes fijo | 10 millones de tokens/mes | Costo predecible |
-| **🆓 GRATIS** | Qoder | **$0** | Ilimitado | 5 modelos ilimitados |
-| | Qwen | **$0** | Ilimitado | 4 modelos ilimitados |
-| | kiro | **$0** | Ilimitado | Claude Sonnet/Haiku (constructor de AWS) |
-| | LongCat Flash Lite 🆕 | **$0**(50 millones de tok/día 🔥) | 1 RPS | La cuota gratuita más grande del mundo |
-| | Polinizaciones AI 🆕 | **$0**(no se necesita clave) | 1 solicitud/15 s | GPT-5, Claude, DeepSeek, Llama 4 |
-| | IA de los trabajadores de Cloudflare 🆕 | **$0**(10K Neuronas/día) | ~150 resp/día | Más de 50 modelos, ventaja global |
-| | Escala de IA 🆕 | **$0**(1 millón de tokens en total) | Tarifa limitada | UE/RGPD, Qwen3 235B, Llama 70B | > 🆕**Nuevos modelos agregados (marzo de 2026):**Familia Grok-4 Fast a $0,20/$0,50/M (comparado a 1143 ms: 30 % más rápido que Gemini 2.5 Flash), GLM-5 a través de Z.AI con salida de 128 K, razonamiento MiniMax M2.5, precios actualizados de DeepSeek V3.2, Kimi K2.5 a través de API directa Moonshot. |
+| Tier | Provider | Cost | Quota Reset | Best For |
+| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- |
+| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed |
+| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users |
+| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! |
+| | GitHub Copilot | $10-19/mo | Monthly | GitHub users |
+| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models |
+| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest |
+| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma |
+| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning |
+| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow |
+| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI |
+| | Mistral | Free trial + paid | Rate limited | European AI |
+| | OpenRouter | Pay-per-use | None | 100+ models aggr. |
+| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship |
+| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup |
+| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks |
+| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option |
+| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access |
+| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost |
+| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited |
+| | Qwen | **$0** | Unlimited | 4 models unlimited |
+| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) |
+| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth |
+| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 |
+| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge |
+| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B |
-**💡 Pila combinada de $0: la configuración gratuita completa:**```
+> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API.
+**💡 $0 Combo Stack — The Complete Free Setup:**
+
+```
# 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever
+Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
+Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
+Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
+Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
+NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+```
-Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
-Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
-LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
-Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
-Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
-Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
-Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
-Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
-Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
-NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
-Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever.
-````
-
-**Costo cero. Nunca deja de codificar.**Configure esto como un combo OmniRoute y todos los respaldos se realizarán automáticamente, sin cambios manuales.---
+---
---
## 🆓 Free Models — What You Actually Get
-> Todos los modelos a continuación son**100% gratuitos y no se requiere tarjeta de crédito**. OmniRoute realiza rutas automáticas entre ellos cuando se agota una cuota; combínelos todos para obtener una combinación irrompible de $0.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
+> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo.
-| Modelo | Prefijo | Límite | Límite de tarifa |
+### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------- | ------ | ------------- | --------------------- |
-| `claude-soneto-4.5` | `kr/` |**Ilimitado**| No se ha informado de un límite diario |
-| `claude-haiku-4.5` | `kr/` |**Ilimitado**| No se ha informado de un límite diario |
-| `claude-opus-4.6` | `kr/` |**Ilimitado**| Última obra a través de Kiro |### 🟢 QODER MODELS (Free PAT via qodercli)
+| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro |
-| Modelo | Prefijo | Límite | Límite de tarifa |
+### 🟢 QODER MODELS (Free PAT via qodercli)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------ | ------ | ------------- | --------------- |
-| `kimi-k2-pensamiento` | `si/` |**Ilimitado**| No hay límite reportado |
-| `qwen3-codificador-plus` | `si/` |**Ilimitado**| No hay límite reportado |
-| `deepseek-r1` | `si/` |**Ilimitado**| No hay límite reportado |
-| `minimax-m2.1` | `si/` |**Ilimitado**| No hay límite reportado |
-| `kimi-k2` | `si/` |**Ilimitado**| No hay límite reportado |
+| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap |
+| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap |
+| `deepseek-r1` | `if/` | **Unlimited** | No reported cap |
+| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap |
+| `kimi-k2` | `if/` | **Unlimited** | No reported cap |
-> Método de conexión recomendado:**Token de acceso personal + `qodercli`**. El navegador OAuth es
-> experimental y deshabilitado de forma predeterminada a menos que las variables de entorno `QODER_OAUTH_*` estén configuradas.### 🟡 QWEN MODELS (Device Code Auth)
+> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is
+> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured.
-| Modelo | Prefijo | Límite | Límite de tarifa |
+### 🟡 QWEN MODELS (Device Code Auth)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------- | ------ | ------------- | ------------------- |
-| `qwen3-codificador-plus` | `qw/` |**Ilimitado**| No hay límite reportado |
-| `qwen3-codificador-flash` | `qw/` |**Ilimitado**| No hay límite reportado |
-| `qwen3-codificador-siguiente` | `qw/` |**Ilimitado**| No hay límite reportado |
-| `modelo-visión` | `qw/` |**Ilimitado**| Multimodal (imágenes) |### 🟣 GEMINI CLI (Google OAuth)
+| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap |
+| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) |
-| Modelo | Prefijo | Límite | Límite de tarifa |
+### 🟣 GEMINI CLI (Google OAuth)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------------ | ------ | --------------------------- | ------------- |
-| `gemini-3-flash-preview` | `gc/` |**180.000 tok/mes**+ 1.000/día | Reinicio mensual |
-| `géminis-2.5-pro` | `gc/` | 180K/mes (piscina compartida) | Alta calidad |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
+| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset |
+| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality |
-| Nivel | Límite diario | Límite de tarifa | Notas |
+### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
+
+| Tier | Daily Limit | Rate Limit | Notes |
| ---------- | ------------ | ----------- | ------------------------------------------------------ |
-| Gratis (desarrollador) | Sin límite de fichas |**~40 RPM**| Más de 70 modelos; transición a límites de tasa pura a mediados de 2025 |
+| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 |
-Modelos gratuitos populares: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
+Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`
-| Nivel | Límite diario | Límite de tarifa | Notas |
-| ---- | ----------------- | ---------------- | ------------------------------------- |
-| Gratis |**1 millón de tokens/día**| 60.000 TPM / 30 RPM | La inferencia LLM más rápida del mundo; se reinicia diariamente |
+### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
-Disponible gratis: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com)
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---- | ----------------- | ---------------- | ------------------------------------------- |
+| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily |
-| Nivel | Límite diario | Límite de tarifa | Notas |
+Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`
+
+### 🔴 GROQ (Free API Key — console.groq.com)
+
+| Tier | Daily Limit | Rate Limit | Notes |
| ---- | ------------- | ---------------- | ----------------------------------------- |
-| Gratis |**14,4K RPD**| 30 RPM por modelo | Sin tarjeta de crédito; 429 en límite, sin cargo |
+| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged |
-Disponible gratis: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`
-| Modelo | Prefijo | Cuota Diaria Gratuita | Notas |
-| ----------------------- | ------ | ----------------- | ----------------------- |
-| `LongCat-Flash-Lite` | `lc/` |**50 millones de tokens**💥 | La cuota gratuita más grande de la historia |
-| `LongCat-Flash-Chat` | `lc/` | Fichas de 500.000 | Chat multiturno |
-| `LongCat-Flash-Pensamiento` | `lc/` | Fichas de 500.000 | Razonamiento / CoT |
-| `LongCat-Flash-Thinking-2601` | `lc/` | Fichas de 500.000 | Versión de enero de 2026 |
-| `LongCat-Flash-Omni-2603` | `lc/` | Fichas de 500.000 | Multimodal |
+### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
-> 100% gratis mientras estés en la versión beta pública. Regístrese en [longcat.chat](https://longcat.chat) con correo electrónico o teléfono. Se reinicia diariamente a las 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+| Model | Prefix | Daily Free Quota | Notes |
+| ----------------------------- | ------ | ----------------- | ----------------------- |
+| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever |
+| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat |
+| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT |
+| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version |
+| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal |
-| Modelo | Prefijo | Límite de tarifa | Proveedor detrás |
+> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC.
+
+### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+
+| Model | Prefix | Rate Limit | Provider Behind |
| ---------- | ------ | ---------- | ------------------ |
-| `openai` | `pol/` | 1 solicitud/15 s | GPT-5 |
-| `claude` | `pol/` | 1 solicitud/15 s | Claude antrópico |
-| `géminis` | `pol/` | 1 solicitud/15 s | Google Géminis |
-| `búsqueda profunda` | `pol/` | 1 solicitud/15 s | Búsqueda profunda V3 |
-| `llama` | `pol/` | 1 solicitud/15 s | Meta Llama 4 Explorador |
-| `mistral` | `pol/` | 1 solicitud/15 s | Mistral IA |
+| `openai` | `pol/` | 1 req/15s | GPT-5 |
+| `claude` | `pol/` | 1 req/15s | Anthropic Claude |
+| `gemini` | `pol/` | 1 req/15s | Google Gemini |
+| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 |
+| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout |
+| `mistral` | `pol/` | 1 req/15s | Mistral AI |
-> ✨**Cero fricción:**Sin registro, sin clave API. Agregue el proveedor de polinizaciones con un campo clave vacío y funcionará de inmediato.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately.
-| Nivel | Neuronas Diarias | Uso equivalente | Notas |
+### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+
+| Tier | Daily Neurons | Equivalent Usage | Notes |
| ---- | ------------- | --------------------------------------- | ----------------------- |
-| Gratis |**10.000**| ~150 LLM resp / audio 500s / 15K incrustaciones | Ventaja global, más de 50 modelos |
+| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models |
-Modelos gratuitos populares: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (¡audio gratis!), `@cf/qwen/qwen2.5-coder-15b-instruct`
+Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct`
-> Requiere token API + ID de cuenta de [dash.cloudflare.com](https://dash.cloudflare.com). Almacene la identificación de la cuenta en la configuración del proveedor.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings.
-| Nivel | Cuota Gratuita | Ubicación | Notas |
+### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+
+| Tier | Free Quota | Location | Notes |
| ---- | ------------- | ------------ | ----------------------------------- |
-| Gratis |**1 millón de tokens**| 🇫🇷 París, UE | No se necesita tarjeta de crédito dentro de los límites |
+| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits |
-Disponible gratis: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
+Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
-> Cumple con la UE/GDPR. Obtenga la clave API en [console.scaleway.com](https://console.scaleway.com).
+> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com).
->**💡 El paquete gratuito definitivo (11 proveedores, $0 para siempre):**
+> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):**
>
> ```
-> Kiro (kr/) → Claude Soneto/Haiku ILIMITADO
-> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 ILIMITADO
-> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 millones de tokens/día 🔥
-> Polinizaciones (pol/) → GPT-5, Claude, DeepSeek, Llama 4: no se necesita clave
-> Qwen (qw/) → modelos de codificador qwen3 ILIMITADOS
-> Gemini (gemini/) → Gemini 2.5 Flash: 1.500 solicitudes/día gratis
-> Cloudflare AI (cf/) → Más de 50 modelos: 10.000 neuronas/día
-> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 millón de tokens gratis (UE)
-> Groq (groq/) → Llama/Gemma — 14,4K solicitudes/día ultrarrápidas
-> NVIDIA NIM (nvidia/) → Más de 70 modelos abiertos: 40 RPM para siempre
-> Cerebras (cerebras/) → Llama/Qwen más rápido del mundo: 1 millón de tok/día
-> ```## 🎙️ Free Transcription Combo
+> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+> Qwen (qw/) → qwen3-coder models UNLIMITED
+> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free
+> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day
+> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast
+> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+> ```
-> Transcribe cualquier audio/video por**$0**: Deepgram ofrece $200 gratis, un respaldo de $50 para AssemblyAI y Groq Whisper como respaldo de emergencia ilimitado.
+## 🎙️ Free Transcription Combo
-| Proveedor | Créditos gratis | Mejor modelo | Límite de tarifa |
+> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup.
+
+| Provider | Free Credits | Best Model | Rate Limit |
| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- |
-| 🟢**Deepgrama**|**$200 gratis**(registro) | `nova-3`: máxima precisión, más de 30 idiomas | Sin límite de RPM en créditos gratis |
-| 🔵**AsambleaAI**|**$50 gratis**(registro) | `universal-3-pro` — capítulos, sentimiento, PII | Sin límite de RPM en créditos gratis |
-| 🔴**Groq**|**Gratis para siempre**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (velocidad limitada) |
+| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits |
+| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits |
+| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) |
-**Combo sugerido en `/dashboard/combos`:**```
+**Suggested combo in `/dashboard/combos`:**
+
+```
Name: free-transcription
Strategy: Priority
Nodes:
[1] deepgram/nova-3 → uses $200 free first
[2] assemblyai/universal-3-pro → fallback when Deepgram credits run out
[3] groq/whisper-large-v3 → free forever, emergency fallback
-````
+```
-Luego, en `/dashboard/media` → pestaña**Transcripción**: cargue cualquier archivo de audio o video → seleccione su punto final combinado → obtenga la transcripción en formatos compatibles.## 💡 Key Features
+Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats.
-OmniRoute v2.0 está diseñado como una plataforma operativa, no solo como un proxy de retransmisión.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
+## 💡 Key Features
-| Característica | Qué hace |
-| ----------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ |
-| ⚡**Familia rápida Grok-4** | Modelos xAI a $0,20/$0,50/M: comparado con 1143 ms (30% más rápido que Gemini 2.5 Flash) |
-| 🧠**GLM-5 vía Z.AI** | Contexto de salida de 128.000 dólares, 0,5 dólares/1 millón: el buque insignia más nuevo de la familia GLM |
-| 🔮**MiniMax M2.5** | Razonamiento + tareas de agente a 0,30 USD/1 millón: mejora significativa desde M2.1 |
-| 🎯**marcador de llamadas de herramientas por modelo** | `toolCalling: verdadero/falso` por modelo en el registro: AutoCombo omite los modelos que no son compatibles con herramientas |
-| 🌍**Detección de intención multilingüe** | Palabras clave PT/ZH/ES/AR en la puntuación AutoCombo: mejor selección de modelos para contenido que no está en inglés |
-| 📊**Retrocesos impulsados por los índices de referencia** | Latencia p95 real de solicitudes en vivo alimenta puntuación combinada: AutoCombo aprende de datos reales |
-| 🔁**Solicitar deduplicación** | Ventana de deduplicación basada en hash de contenido: segura para múltiples agentes, evita cargos duplicados |
-| 🔌**Estrategia de enrutador conectable** | Interfaz extensible `RouterStrategy`: agregue lógica de enrutamiento personalizada como complementos | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP |
+OmniRoute v3.5 is built as an operational platform, not just a relay proxy.
-| Característica | Qué hace |
-| ------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |
-| 🎮**Patio de juegos modelo** | Página de panel para probar cualquier modelo directamente: selectores de proveedor/modelo/punto final, editor Monaco, transmisión, cancelación, sincronización |
-| 🔏**Coincidencia de huellas dactilares CLI** | Orden de encabezado/cuerpo por proveedor para que coincida con las firmas CLI nativas: alterne por proveedor en Configuración > Seguridad.**Se conserva la IP de tu proxy** |
-| 🤝**Soporte ACP (Protocolo cliente-agente)** | Descubrimiento de agentes CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw y 9 más), generador de procesos, punto final `/api/acp/agents` |
-| 🤖**Panel de agentes de ACP** | Página Depurar › Agentes: cuadrícula de 14 agentes con estado de instalación, versión y formulario de agente personalizado para cualquier herramienta CLI. Los usuarios de**OpenCode**obtienen un botón "Descargar opencode.json" que genera automáticamente una configuración lista para usar con todos los modelos disponibles. |
-| 🔧**Enrutamiento del modelo personalizado `apiFormat`** | Los modelos personalizados con `apiFormat: "responses"` ahora se enrutan correctamente al traductor de la API de Respuestas |
-| 🏢**Aislamiento del espacio de trabajo del Codex** | Múltiples espacios de trabajo de Codex por correo electrónico: OAuth separa correctamente las conexiones por ID del espacio de trabajo |
-| 🔄**Actualización automática electrónica** | La aplicación de escritorio busca actualizaciones + instalación automática al reiniciar | ### 🤖 Agent & Protocol Operations (v2.0) |
+### 🆕 New — v3.5.5 Highlights (Apr 2026)
-| Característica | Qué hace |
-| --------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- |
-| 🔧**Servidor MCP (25 herramientas)** | Herramientas IDE/agente a través de 3 transportes: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 núcleos + 3 memorias + 4 herramientas de habilidades |
-| 🤝**Servidor A2A (JSON-RPC + SSE)** | Ejecución de tareas de agente a agente con flujos de sincronización y streaming |
-| 🧭**Página de puntos finales consolidados** | Página de administración con pestañas con pestañas Endpoint Proxy, MCP, A2A y API Endpoints |
-| 🎚️**Activación/desactivación de servicio** | Interruptores ON/OFF para MCP y A2A con persistencia de configuración (predeterminado: OFF) |
-| 🛰️**Latido del tiempo de ejecución de MCP** | Estado real del proceso (pid, tiempo de actividad, antigüedad del latido, transporte, modo de alcance) |
-| 📋**Pista de auditoría de MCP** | Registros de auditoría filtrables con éxito/fracaso y atribución de claves |
-| 🔐**Cumplimiento del alcance del MCP** | 10 permisos de alcance granular para acceso controlado a herramientas |
-| 📡**Gestión del ciclo de vida de tareas A2A** | Enumerar/filtrar tareas, inspeccionar eventos/artefactos, cancelar tareas en ejecución |
-| 📋**Descubrimiento de tarjeta de agente** | `/.well-known/agent.json` para el descubrimiento automático de clientes |
-| 🧪**Arnés de prueba del protocolo E2E** | El cliente real MCP SDK + A2A fluye en `test:protocols:e2e` |
-| ⚙️**Controles operativos** | Cambie el combo, aplique perfiles de resiliencia, reinicie los disyuntores desde una superficie de control | ### 🧠 Routing & Intelligence |
+| Feature | What It Does |
+| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation |
+| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs |
+| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions |
+| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection |
+| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts |
-| Característica | Qué hace |
-| ----------------------------------------------- | ----------------------------------------------------------------------------------------------- | ----------------------- |
-| 🎯**Retroceso inteligente de 4 niveles** | Ruta automática: Suscripción → Clave API → Barato → Gratis |
-| 📊**Seguimiento de cuotas en tiempo real** | Recuento de tokens en vivo + reinicio de cuenta regresiva por proveedor |
-| 🔄**Traducción de formato** | OpenAI ↔ Claude ↔ Gemini ↔ Respuestas con conversiones seguras para esquemas |
-| 👥**Soporte multicuenta** | Múltiples cuentas por proveedor con selección inteligente |
-| 🔄**Actualización automática de tokens** | Los tokens de OAuth se actualizan automáticamente con un reintento |
-| 🎨**Combinaciones personalizadas** | 9 estrategias de equilibrio + control de la cadena alternativa |
-| 🌐**Enrutador comodín** | `proveedor/*` enrutamiento dinámico |
-| 🧠**Pensando en los controles presupuestarios** | Límites de razonamiento de transferencia, automático, personalizado y adaptativo |
-| 🔀**Alias de modelo** | Seguridad de migración y alias de modelo integrado y personalizado |
-| ⚡**Degradación del fondo** | Dirija tareas en segundo plano de baja prioridad a modelos más baratos |
-| 🧪**Enrutamiento inteligente basado en tareas** | Modelo de selección automática por tipo de contenido (codificación/visión/análisis/resumen) |
-| 🔄**Flujos de trabajo del agente A2A** | Orquestador FSM determinista para ejecuciones de agentes de varios pasos con estado |
-| 🔀**Enrutamiento adaptativo** | Anulación de estrategia dinámica basada en el volumen de tokens y la complejidad del aviso |
-| 🎲**Diversidad de proveedores** | Puntuación de entropía de Shannon que equilibra la distribución del tráfico de combo automático |
-| 💬**Inyección de indicación del sistema** | Controles de comportamiento global aplicados consistentemente |
-| 📄**Compatibilidad API de respuestas** | Soporte completo `/v1/responses` para Codex y flujos de trabajo agentes avanzados | ### 🎵 Multi-Modal APIs |
+### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
-| Característica | Qué hace |
-| ---------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- |
-| 🖼️**Generación de imágenes** | `/v1/images/generaciones` con backends locales y en la nube |
-| 📐**Incrustaciones** | `/v1/embeddings` para búsqueda y canales RAG |
-| 🎤**Transcripción de audio** | `/v1/audio/transcriptions` — 7 proveedores (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), detección automática de idioma, compatibilidad con MP4/MP3/WAV |
-| 🔊**Texto a voz** | `/v1/audio/speech` — 10 proveedores (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) con mensajes de error correctos |
-| 🎬**Generación de vídeo** | `/v1/videos/generaciones` (flujos de trabajo ComfyUI + SD WebUI) |
-| 🎵**Generación Musical** | `/v1/music/generaciones` (flujos de trabajo de ComfyUI) |
-| 🛡️**Moderaciones** | `/v1/moderaciones` controles de seguridad |
-| 🔀**Reclasificación** | `/v1/rerank` para puntuación de relevancia |
-| 🔍**Búsqueda web**🆕 | `/v1/search` — 5 proveedores (Serper, Brave, Perplexity, Exa, Tavily), más de 6500 gratis/mes, conmutación por error automática, caché | ### 🛡️ Resilience, Security & Governance |
+| Feature | What It Does |
+| ------------------------------------ | ------------------------------------------------------------------------------------------- |
+| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) |
+| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family |
+| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 |
+| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models |
+| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content |
+| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data |
+| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges |
+| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins |
-| Característica | Qué hace |
-| ---------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------- | -------------------------------- |
-| 🔌**Disyuntores** | Viaje/recuperación por modelo con controles de umbral |
-| 🎯**Modelos compatibles con endpoints** | Los modelos personalizados declaran puntos finales compatibles + formato API |
-| 🛡️**Rebaño Anti-Truenos** | Mutex + protecciones de semáforo en eventos de reintento/tasa |
-| 🧠**Caché semántico + firma** | Reducción de costos/latencia con dos capas de caché |
-| ⚡**Solicitar Idempotencia** | Ventana de protección duplicada |
-| 🔒**Suplantación de huellas dactilares TLS** | Huella digital TLS similar a la de un navegador:**reduce la detección de bots y el marcado de cuentas** |
-| 🔏**Coincidencia de huellas dactilares CLI** | Coincide con las firmas de solicitudes CLI nativas:**reduce el riesgo de prohibición y al mismo tiempo preserva la IP del proxy** |
-| 🌐**Filtrado de IP** | Control de lista blanca/lista negra para implementaciones expuestas |
-| 📊**Límites de tarifas editables** | Límites globales/a nivel de proveedor configurables con persistencia |
-| 📉**Degradación elegante** | Respaldos de capacidad multicapa que protegen las operaciones centrales de la puerta de enlace |
-| 📜**Pista de auditoría de configuración** | Seguimiento de cambios basado en diferencias que evita la deriva operativa con reversiones simples |
-| ⏳**Sincronización de salud del proveedor** | Monitoreo proactivo de vencimiento de tokens que activa alertas antes de fallas de autorización |
-| 🚪**Desactivación automática de cuentas prohibidas** | Disyuntor operativo que sella automáticamente cuentas simbólicas bloqueadas permanentemente |
-| 🔑**Administración de claves API + Alcance** | Emisión/rotación de claves segura y controles de modelo/proveedor |
-| 👁️**Revelación de clave API con alcance**🆕 | Recuperación voluntaria de claves API a través de `ALLOW_API_KEY_REVEAL` |
-| 🛡️**Protegido `/modelos`** | Puerta de autenticación opcional y ocultación de proveedores para el catálogo de modelos | ### 📊 Observability & Analytics |
+### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP
-| Característica | Qué hace |
-| ----------------------------------------- | ---------------------------------------------------------------------------------- | ---------------------------- |
-| 📝**Solicitud + Registro de proxy** | Solicitud/respuesta completa y registro de proxy |
-| 📉**Registros detallados transmitidos**🆕 | Reconstruye secuencias de carga útil SSE limpiamente en la interfaz de usuario |
-| 📋**Panel de registros unificado** | Vistas de solicitud, proxy, auditoría y consola en una sola página |
-| 🔍**Solicitar telemetría** | Latencia p50/p95/p99 y seguimiento de solicitudes |
-| 🏥**Panel de salud** | Tiempo de actividad, estados de los interruptores, bloqueos, estadísticas de caché |
-| 💰**Seguimiento de costos** | Controles de presupuesto y visibilidad de precios por modelo |
-| 📈**Visualizaciones analíticas** | Información sobre el uso de modelos/proveedores y vistas de tendencias |
-| 🧪**Marco de evaluación** | Prueba de set dorado con estrategias de partido configurables |
-| 📡**Diagnóstico en vivo**🆕 | Omisión de caché semántica para pruebas combinadas en vivo precisas | ### ☁️ Deployment & Platform |
+| Feature | What It Does |
+| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing |
+| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** |
+| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint |
+| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. |
+| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator |
+| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID |
+| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart |
-| Característica | Qué hace |
-| --------------------------------------- | ------------------------------------------------------------------------------------- | --------------------- |
-| 🌐**Implementar en cualquier lugar** | Localhost, VPS, Docker, entornos Cloud |
-| 🚇**Túnel Cloudflare**🆕 | Integración de Quick Tunnel con un clic desde el panel |
-| 🔑**Filtrado de modelo de clave API** | Respuesta nativa /v1/models filtrada mediante roles de contexto de portador asignados |
-| ⚡**Omisión de caché inteligente** | Heurísticas TTL configurables y controles de recuperación forzada |
-| 🔄**Copia de seguridad/Restaurar** | Flujos de exportación/importación y recuperación ante desastres |
-| 🧙**Asistente de incorporación** | Configuración guiada de primera ejecución |
-| 🔧**Panel de herramientas CLI** | Configuración con un clic para herramientas de codificación populares |
-| 🎮**Patio de juegos modelo** | Pruebe cualquier proveedor/modelo/punto final desde el panel |
-| 🔏**Alternar huella digital CLI** | Coincidencia de huellas dactilares por proveedor en Configuración > Seguridad |
-| 🌐**i18n (30 idiomas)** | Panel completo + compatibilidad con idiomas de documentos con cobertura RTL |
-| 🧹**Borrar todos los modelos** | Borrado de la lista de modelos con un solo clic en los detalles del proveedor |
-| 👁️**Controles de la barra lateral**🆕 | Ocultar componentes e integraciones desde Configuración de apariencia |
-| 📋**Plantillas de problemas** | Plantillas de GitHub estandarizadas para errores y funciones |
-| 📂**Directorio de datos personalizado** | Anulación de `DATA_DIR` para la ubicación de almacenamiento | ### Feature Deep Dive |
+### 🤖 Agent & Protocol Operations (v2.0)
+
+| Feature | What It Does |
+| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
+| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools |
+| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows |
+| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs |
+| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) |
+| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) |
+| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution |
+| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access |
+| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks |
+| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery |
+| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` |
+| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface |
+
+### 🧠 Routing & Intelligence
+
+| Feature | What It Does |
+| ---------------------------------- | ------------------------------------------------------------------------ |
+| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free |
+| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider |
+| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions |
+| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection |
+| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry |
+| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control |
+| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session |
+| 🌐 **Wildcard Router** | `provider/*` dynamic routing |
+| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits |
+| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety |
+| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models |
+| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) |
+| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions |
+| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity |
+| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution |
+| 💬 **System Prompt Injection** | Global behavior controls applied consistently |
+| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows |
+
+### 🎵 Multi-Modal APIs
+
+| Feature | What It Does |
+| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends |
+| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines |
+| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support |
+| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages |
+| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) |
+| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) |
+| 🛡️ **Moderations** | `/v1/moderations` safety checks |
+| 🔀 **Reranking** | `/v1/rerank` for relevance scoring |
+| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache |
+
+### 🛡️ Resilience, Security & Governance
+
+| Feature | What It Does |
+| ----------------------------------- | -------------------------------------------------------------------------------------- |
+| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls |
+| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format |
+| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events |
+| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers |
+| ⚡ **Request Idempotency** | Duplicate protection window |
+| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** |
+| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** |
+| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments |
+| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence |
+| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations |
+| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks |
+| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures |
+| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically |
+| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls |
+| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` |
+| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog |
+
+### 📊 Observability & Analytics
+
+| Feature | What It Does |
+| -------------------------------- | ----------------------------------------------------- |
+| 📝 **Request + Proxy Logging** | Full request/response and proxy logging |
+| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI |
+| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page |
+| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing |
+| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats |
+| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility |
+| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views |
+| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies |
+| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing |
+
+### ☁️ Deployment & Platform
+
+| Feature | What It Does |
+| ------------------------------ | --------------------------------------------------------------------- |
+| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments |
+| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard |
+| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles |
+| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls |
+| 🔄 **Backup/Restore** | Export/import and disaster recovery flows |
+| 🧙 **Onboarding Wizard** | First-run guided setup |
+| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools |
+| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard |
+| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security |
+| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage |
+| 🧹 **Clear All Models** | One-click model list clearing in provider details |
+| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings |
+| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features |
+| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location |
+
+### Feature Deep Dive
#### Smart fallback with practical cost control
@@ -1292,103 +1464,132 @@ Combo: "my-coding-stack"
4. if/kimi-k2-thinking
```
-Cuando falla la cuota, la tasa o el estado, OmniRoute pasa automáticamente al siguiente candidato sin necesidad de cambiar manualmente.#### Protocol management that is visible and operable
+When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching.
-- MCP + A2A se pueden descubrir en la interfaz de usuario y en los documentos (no están ocultos)
-- Las API de estado del protocolo exponen datos operativos en vivo (`/api/mcp/*`, `/api/a2a/*`)
-- Los paneles incluyen acciones para las operaciones del día 2 (cambio de combo, reinicio de interruptores, cancelación de tareas)#### Translator + validation workflow
+#### Protocol management that is visible and operable
-El área de Traductor incluye:
+- MCP + A2A are discoverable in UI and docs (not hidden)
+- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`)
+- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation)
--**Parque infantil**: solicitar comprobaciones de transformación -**Chat Tester**: solicitud/respuesta completa de ida y vuelta -**Banco de pruebas**: varios casos en una ejecución -**Live Monitor**: vista del tráfico en tiempo real
+#### Translator + validation workflow
-Además de validación de protocolo con clientes reales a través de `npm run test:protocols:e2e`.
+The Translator area includes:
-> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Referencia de herramientas, configuraciones IDE y ejemplos de clientes
+- **Playground**: request transformation checks
+- **Chat Tester**: full request/response round-trip
+- **Test Bench**: multiple cases in one run
+- **Live Monitor**: real-time traffic view
+
+Plus protocol validation with real clients via `npm run test:protocols:e2e`.
+
+> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples
>
-> 📖**[README del servidor A2A](src/lib/a2a/README.md)**— Habilidades, métodos JSON-RPC, transmisión y ciclo de vida de las tareas## 🧪 Evaluations (Evals)
+> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle
-OmniRoute incluye un marco de evaluación integrado para probar la calidad de la respuesta de LLM frente a un conjunto de referencia. Acceda a él a través de**Análisis → Evaluaciones**en el panel.### Built-in Golden Set
+## 🧪 Evaluations (Evals)
-El "OmniRoute Golden Set" precargado contiene casos de prueba para:
+OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard.
-- Saludos, matemáticas, geografía, generación de código.
-- Cumplimiento del formato JSON, traducción, generación de rebajas.
-- Rechazo de seguridad (contenido nocivo), conteo, lógica booleana### Evaluation Strategies
+### Built-in Golden Set
-| Estrategia | Descripción | Ejemplo |
-| ------------------- | ---------------------------------------------------------------------------------- | ---------------------------------- | --- |
-| `exacto` | La salida debe coincidir exactamente | `"4"` |
-| `contiene` | La salida debe contener una subcadena (no distingue entre mayúsculas y minúsculas) | `"París"` |
-| `expresión regular` | La salida debe coincidir con el patrón de expresiones regulares | `"1.*2.*3"` |
-| `personalizado` | La función JS personalizada devuelve verdadero/falso | `(salida) => salida.longitud > 10` | --- |
+The pre-loaded "OmniRoute Golden Set" contains test cases for:
+
+- Greetings, math, geography, code generation
+- JSON format compliance, translation, markdown generation
+- Safety refusal (harmful content), counting, boolean logic
+
+### Evaluation Strategies
+
+| Strategy | Description | Example |
+| ---------- | ------------------------------------------------ | -------------------------------- |
+| `exact` | Output must match exactly | `"4"` |
+| `contains` | Output must contain substring (case-insensitive) | `"Paris"` |
+| `regex` | Output must match regex pattern | `"1.*2.*3"` |
+| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` |
+
+---
## 📖 Setup Guide
### Protocol Setup (MCP + A2A)
-
-🧩 Configuración de MCP (Protocolo de contexto del modelo)
+
+🧩 MCP Setup (Model Context Protocol)
-Inicie el transporte MCP en modo stdio:```bash
+Start MCP transport in stdio mode:
+
+```bash
omniroute --mcp
+```
-````
+Recommended validation flow:
-Flujo de validación recomendado:
+1. Connect your MCP client over stdio.
+2. Run `omniroute_get_health`.
+3. Run `omniroute_list_combos`.
+4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit.
-1. Conecte su cliente MCP a través de stdio.
-2. Ejecute `omniroute_get_health`.
-3. Ejecute `omniroute_list_combos`.
-4. Abra `/dashboard/mcp` para confirmar el latido, la actividad y la auditoría.
+Useful APIs for automation:
-API útiles para la automatización:
+- `GET /api/mcp/status`
+- `GET /api/mcp/tools`
+- `GET /api/mcp/audit`
+- `GET /api/mcp/audit/stats`
-- `OBTENER /api/mcp/status`
-- `OBTENER /api/mcp/tools`
-- `OBTENER /api/mcp/auditoría`
-- `OBTENER /api/mcp/audit/stats`
+
-
-🤝 Configuración de A2A (Agent2Agent)
+
+🤝 A2A Setup (Agent2Agent)
-Descubra el agente:```bash
+Discover the agent:
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
-Enviar una tarea:```bash
+Send a task:
+
+```bash
curl -X POST http://localhost:20128/a2a \
- -H 'content-type: application/json' \
- -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+ -H 'content-type: application/json' \
+ -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+```
-````
+Manage lifecycle:
-Gestionar el ciclo de vida:
+- `GET /api/a2a/status`
+- `GET /api/a2a/tasks`
+- `GET /api/a2a/tasks/:id`
+- `POST /api/a2a/tasks/:id/cancel`
-- `OBTENER /api/a2a/status`
-- `OBTENER /api/a2a/tareas`
-- `OBTENER /api/a2a/tasks/:id`
-- `POST /api/a2a/tasks/:id/cancelar`
+Operational UI:
-Interfaz de usuario operativa:
+- `/dashboard/a2a` for task/state/stream observability and smoke actions
-- `/dashboard/a2a` para observabilidad de tarea/estado/corriente y acciones de humo
+
-
-🧪 Validación de protocolo de un extremo a otro
+
+🧪 End-to-end protocol validation
-Validar ambos protocolos con clientes reales:```bash
+Validate both protocols with real clients:
+
+```bash
npm run test:protocols:e2e
-````
+```
-Esto verifica:
+This verifies:
-- Conexión/lista/llamada del cliente MCP SDK
-- Descubrimiento A2A/enviar/transmitir/obtener/cancelar
-- Verificación cruzada de datos en auditoría MCP y API de administración de tareas A2A
+- MCP SDK client connect/list/call
+- A2A discovery/send/stream/get/cancel
+- Cross-check data in MCP audit and A2A task management APIs
-
-💳 Proveedores de suscripción### Claude Code (Pro/Max)
+
+
+
+💳 Subscription Providers
+
+### Claude Code (Pro/Max)
```bash
Dashboard → Providers → Connect Claude Code
@@ -1401,7 +1602,9 @@ Models:
cc/claude-haiku-4-5-20251001
```
-**Consejo profesional:**Utilice Opus para tareas complejas y Sonnet para mayor velocidad. ¡OmniRoute realiza un seguimiento de la cuota por modelo!### OpenAI Codex (Plus/Pro)
+**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model!
+
+### OpenAI Codex (Plus/Pro)
```bash
Dashboard → Providers → Connect Codex
@@ -1415,20 +1618,22 @@ Models:
#### Codex Account Limit Management (5h + Weekly)
-Cada cuenta de Codex ahora tiene políticas para alternar en `Panel -> Proveedores`:
+Each Codex account now has policy toggles in `Dashboard -> Providers`:
-- `5h` (ON/OFF): aplica la política de umbral de ventana de 5 horas.
-- `Semanal` (ON/OFF): aplica la política de umbral de ventana semanal.
-- Comportamiento de umbral: cuando una ventana habilitada alcanza >=90% de uso, esa cuenta se omite.
-- Comportamiento de rotación: OmniRoute dirige automáticamente a la siguiente cuenta elegible del Codex.
-- Comportamiento de reinicio: cuando pasa el tiempo `resetAt` del proveedor, la cuenta vuelve a ser elegible automáticamente.
+- `5h` (ON/OFF): enforce the 5-hour window threshold policy.
+- `Weekly` (ON/OFF): enforce the weekly window threshold policy.
+- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped.
+- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically.
+- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically.
-Escenarios:
+Scenarios:
-- `5h ON` + `Weekly ON`: la cuenta se omite cuando cualquiera de las ventanas alcanza el umbral.
-- `5h OFF` + `Weekly ON`: solo el uso semanal puede bloquear la cuenta.
-- `5h ON` + `Weekly OFF`: solo el uso de 5 horas puede bloquear la cuenta.
-- `resetAt` pasó: la cuenta vuelve a ingresar a la rotación automáticamente (no se puede volver a habilitar manualmente).### Gemini CLI (FREE 180K/month!)
+- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold.
+- `5h OFF` + `Weekly ON`: only weekly usage can block the account.
+- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account.
+- `resetAt` passed: account re-enters rotation automatically (no manual re-enable).
+
+### Gemini CLI (FREE 180K/month!)
```bash
Dashboard → Providers → Connect Gemini CLI
@@ -1440,7 +1645,9 @@ Models:
gc/gemini-2.5-pro
```
-**Mejor valor:**¡Enorme nivel gratuito! Utilice esto antes de los niveles pagos.### GitHub Copilot
+**Best Value:** Huge free tier! Use this before paid tiers.
+
+### GitHub Copilot
```bash
Dashboard → Providers → Connect GitHub
@@ -1455,71 +1662,91 @@ Models:
-
-🔑 Proveedores de claves API### NVIDIA NIM (FREE developer access — 70+ models)
+
+🔑 API Key Providers
-1. Regístrate: [build.nvidia.com](https://build.nvidia.com)
-2. Obtenga una clave API gratuita (1000 créditos de inferencia incluidos)
-3. Panel de control → Agregar proveedor → NVIDIA NIM:
- - Clave API: `nvapi-tu-clave`
+### NVIDIA NIM (FREE developer access — 70+ models)
-**Modelos:**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` y más de 50
+1. Sign up: [build.nvidia.com](https://build.nvidia.com)
+2. Get free API key (1000 inference credits included)
+3. Dashboard → Add Provider → NVIDIA NIM:
+ - API Key: `nvapi-your-key`
-**Consejo profesional:**API compatible con OpenAI: ¡funciona perfectamente con la traducción de formatos de OmniRoute!### DeepSeek
+**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more
-1. Regístrate: [platform.deepseek.com](https://platform.deepseek.com)
-2. Obtenga la clave API
-3. Panel de control → Agregar proveedor → DeepSeek
+**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation!
-**Modelos:**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!)
+### DeepSeek
-1. Regístrese: [console.groq.com](https://console.groq.com)
-2. Obtenga la clave API (nivel gratuito incluido)
-3. Panel de control → Agregar proveedor → Groq
+1. Sign up: [platform.deepseek.com](https://platform.deepseek.com)
+2. Get API key
+3. Dashboard → Add Provider → DeepSeek
-**Modelos:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b`
+**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder`
-**Consejo profesional:**Inferencia ultrarrápida: ¡lo mejor para codificación en tiempo real!### OpenRouter (100+ Models)
+### Groq (Free Tier Available!)
-1. Regístrate: [openrouter.ai](https://openrouter.ai)
-2. Obtenga la clave API
-3. Panel de control → Agregar proveedor → OpenRouter
+1. Sign up: [console.groq.com](https://console.groq.com)
+2. Get API key (free tier included)
+3. Dashboard → Add Provider → Groq
-**Modelos:**Acceda a más de 100 modelos de los principales proveedores a través de una única clave API.
+**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b`
-**Comportamiento del panel:**Los modelos OpenRouter se administran desde**Modelos disponibles**. La adición manual, la importación y la sincronización automática actualizan la misma lista.
+**Pro Tip:** Ultra-fast inference — best for real-time coding!
-
-💰 Proveedores baratos (copia de seguridad)### GLM-4.7 (Daily reset, $0.6/1M)
+### OpenRouter (100+ Models)
-1. Regístrate: [Zhipu AI](https://open.bigmodel.cn/)
-2. Obtenga la clave API del plan de codificación
-3. Panel de control → Agregar clave API:
- - Proveedor: `glm`
- - Clave API: `tu-clave`
+1. Sign up: [openrouter.ai](https://openrouter.ai)
+2. Get API key
+3. Dashboard → Add Provider → OpenRouter
-**Uso:**`glm/glm-4.7`
+**Models:** Access 100+ models from all major providers through a single API key.
-**Consejo profesional:**¡El plan de codificación ofrece una cuota triple a un costo de 1/7! Reiniciar diariamente a las 10:00 a.m.### MiniMax M2.1 (5h reset, $0.20/1M)
+**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list.
-1. Regístrate: [MiniMax](https://www.minimax.io/)
-2. Obtenga la clave API
-3. Panel de control → Agregar clave API
+
-**Uso:**`minimax/MiniMax-M2.1`
+
+💰 Cheap Providers (Backup)
-**Consejo profesional:**¡La opción más barata para contexto largo (1 millón de tokens)!### Kimi K2 ($9/month flat)
+### GLM-4.7 (Daily reset, $0.6/1M)
-1. Suscríbete: [Moonshot AI](https://platform.moonshot.ai/)
-2. Obtenga la clave API
-3. Panel de control → Agregar clave API
+1. Sign up: [Zhipu AI](https://open.bigmodel.cn/)
+2. Get API key from Coding Plan
+3. Dashboard → Add API Key:
+ - Provider: `glm`
+ - API Key: `your-key`
-**Uso:**`kimi/kimi-latest`
+**Use:** `glm/glm-4.7`
-**Consejo profesional:**¡Fijo $9/mes por 10 millones de tokens = $0,90/1 millón de costo efectivo!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM.
-
-🆓 Proveedores GRATUITOS (respaldo de emergencia)### Qoder (5 FREE models via OAuth)
+### MiniMax M2.1 (5h reset, $0.20/1M)
+
+1. Sign up: [MiniMax](https://www.minimax.io/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `minimax/MiniMax-M2.1`
+
+**Pro Tip:** Cheapest option for long context (1M tokens)!
+
+### Kimi K2 ($9/month flat)
+
+1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `kimi/kimi-latest`
+
+**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost!
+
+
+
+
+🆓 FREE Providers (Emergency Backup)
+
+### Qoder (5 FREE models via OAuth)
```bash
Dashboard → Connect Qoder
@@ -1560,8 +1787,10 @@ Models:
-
-🎨 Crear combos### Example 1: Maximize Subscription → Cheap Backup
+
+🎨 Create Combos
+
+### Example 1: Maximize Subscription → Cheap Backup
```
Dashboard → Combos → Create New
@@ -1589,8 +1818,10 @@ Cost: $0 forever!
-
-🔧 Integración CLI### Cursor IDE
+
+🔧 CLI Integration
+
+### Cursor IDE
```
Settings → Models → Advanced:
@@ -1601,7 +1832,9 @@ Settings → Models → Advanced:
### Claude Code
-Utilice la página**Herramientas CLI**en el panel para realizar la configuración con un solo clic o edite `~/.claude/settings.json` manualmente.### Codex CLI
+Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually.
+
+### Codex CLI
```bash
export OPENAI_BASE_URL="http://localhost:20128"
@@ -1612,12 +1845,15 @@ codex "your prompt"
### OpenClaw
-**Opción 1: Panel de control (recomendado):**```
+**Option 1 — Dashboard (recommended):**
+
+```
Dashboard → CLI Tools → OpenClaw → Select Model → Apply
+```
-````
+**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`:
-**Opción 2 — Manual:**Editar `~/.openclaw/openclaw.json`:```json
+```json
{
"models": {
"providers": {
@@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply
}
}
}
-````
+```
-> **Nota:**OpenClaw solo funciona con OmniRoute local. Utilice `127.0.0.1` en lugar de `localhost` para evitar problemas de resolución de IPv6.### Cline / Continue / RooCode
+> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues.
+
+### Cline / Continue / RooCode
```
Settings → API Configuration:
@@ -1643,15 +1881,17 @@ Settings → API Configuration:
### OpenCode
-**Paso 1:**Agregue OmniRoute como proveedor personalizado:```bash
+**Step 1:** Add OmniRoute as a custom provider:
+
+```bash
opencode
/connect
-
# Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key
+```
-````
+**Step 2:** Create/edit `opencode.json` in your project root:
-**Paso 2:**Crea/edita `opencode.json` en la raíz de tu proyecto:```json
+```json
{
"$schema": "https://opencode.ai/config.json",
"provider": {
@@ -1669,117 +1909,130 @@ opencode
}
}
}
-````
+```
-**Paso 3:**Selecciona el modelo en OpenCode:```bash
+**Step 3:** Select the model in OpenCode:
+
+```bash
/models
-
# Select any OmniRoute model from the list
+```
-````
+> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard.
->**Consejo:**Agregue cualquier modelo disponible en su terminal `/v1/models` de OmniRoute a la sección `modelos`. Utilice el formato `proveedor/modelo-id` desde su panel de OmniRoute.
+
---
## Solución de Problemas
-
-Haga clic para expandir la guía de solución de problemas
+
+Click to expand troubleshooting guide
-**"El modelo de idioma no proporcionó mensajes"**
+**"Language model did not provide messages"**
-- Cuota de proveedor agotada → Verifique el rastreador de cuotas del panel
-- Solución: utilice el combo alternativo o cambie a un nivel más económico
+- Provider quota exhausted → Check dashboard quota tracker
+- Solution: Use combo fallback or switch to cheaper tier
-**Limitación de tasa**
+**Rate limiting**
-- Cuota de suscripción agotada → Alternativa a GLM/MiniMax
-- Agregar combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Subscription quota out → Fallback to GLM/MiniMax
+- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-**El token de OAuth expiró**
+**OAuth token expired**
-- Actualizado automáticamente por OmniRoute
-- Si los problemas persisten: Panel → Proveedor → Volver a conectar
+- Auto-refreshed by OmniRoute
+- If issues persist: Dashboard → Provider → Reconnect
-**Altos costos**
+**High costs**
-- Verifique las estadísticas de uso en Panel → Costos
-- Cambiar el modelo principal a GLM/MiniMax
-- Utilice el nivel gratuito (Gemini CLI, Qoder) para tareas no críticas
+- Check usage stats in Dashboard → Costs
+- Switch primary model to GLM/MiniMax
+- Use free tier (Gemini CLI, Qoder) for non-critical tasks
-**Los puertos del panel/API están incorrectos**
+**Dashboard/API ports are wrong**
-- `PORT` es el puerto base canónico (y el puerto API por defecto)
-- `API_PORT` anula sólo el detector de API compatible con OpenAI
-- `DASHBOARD_PORT` anula solo el panel de control/escucha Next.js
-- Configure `NEXT_PUBLIC_BASE_URL` en su panel/URL pública (para devoluciones de llamada de OAuth)
+- `PORT` is the canonical base port (and API port by default)
+- `API_PORT` overrides only OpenAI-compatible API listener
+- `DASHBOARD_PORT` overrides only dashboard/Next.js listener
+- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks)
-**Errores de sincronización en la nube**
+**Cloud sync errors**
-- Verifique que `BASE_URL` apunte a su instancia en ejecución
-- Verifique que `CLOUD_URL` apunte al punto final de nube esperado
-- Mantenga los valores `NEXT_PUBLIC_*` alineados con los valores del lado del servidor
+- Verify `BASE_URL` points to your running instance
+- Verify `CLOUD_URL` points to your expected cloud endpoint
+- Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**El primer inicio de sesión no funciona**
+**First login not working**
-- Marque `INITIAL_PASSWORD` en `.env`
-- Si no está configurada, la contraseña alternativa es `123456`
+- Check `INITIAL_PASSWORD` in `.env`
+- If unset, fallback password is `123456`
-**No hay registros de solicitudes**
+**No request logs**
-- Los artefactos de solicitud se escriben en `DATA_DIR/call_logs/` como un archivo JSON por solicitud
-- Habilite la captura de canalización desde Panel → Registros → Solicitar registros si necesita cargas útiles detalladas por etapa
-- Configure `APP_LOG_TO_FILE=true` si también desea que la consola de la aplicación registre `logs/application/app.log`
-- Ajuste `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` y `CALL_LOG_MAX_ENTRIES` según sea necesario
+- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request
+- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads
+- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log`
+- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed
-**La prueba de conexión muestra "No válido" para proveedores compatibles con OpenAI**
+**Connection test shows "Invalid" for OpenAI-compatible providers**
-- Muchos proveedores no exponen un punto final `/models`
-- OmniRoute v1.0.6+ incluye validación alternativa mediante la finalización del chat
-- Asegúrese de que la URL base incluya el sufijo `/v1`### 🔐 OAuth on a Remote Server
+- Many providers don't expose a `/models` endpoint
+- OmniRoute v1.0.6+ includes fallback validation via chat completions
+- Ensure base URL includes `/v1` suffix
-
+### 🔐 OAuth on a Remote Server
+
+
->**⚠️ Importante para los usuarios que ejecutan OmniRoute en un VPS, Docker o cualquier servidor remoto**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
+> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server**
-Los proveedores**Antigravity**y**Gemini CLI**utilizan**Google OAuth 2.0**. Google requiere que `redirect_uri` en el flujo de OAuth coincida exactamente con uno de los URI registrados previamente en Google Cloud Console de la aplicación.
+#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
-Las credenciales de OAuth incluidas en OmniRoute están registradas**solo para `localhost`**. Cuando accede a OmniRoute en un servidor remoto (por ejemplo, `https://omniroute.myserver.com`), Google rechaza la autenticación con:```
+The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console.
+
+The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solution: Configure your own OAuth credentials
-Debes crear un**ID de cliente de OAuth 2.0**en Google Cloud Console con el URI de tu servidor.#### Step-by-step
+You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI.
-**1. Abra la consola de Google Cloud**
+#### Step-by-step
-Vaya a: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
+**1. Open Google Cloud Console**
-**2. Cree un nuevo ID de cliente OAuth 2.0**
+Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-- Haga clic en**"+ Crear credenciales"**→**"ID de cliente OAuth"**
-- Tipo de aplicación:**"Aplicación web"**
-- Nombre: lo que quieras (por ejemplo, `OmniRoute Remote`)
+**2. Create a new OAuth 2.0 Client ID**
-**3. Agregar URI de redireccionamiento autorizado**
+- Click **"+ Create Credentials"** → **"OAuth client ID"**
+- Application type: **"Web application"**
+- Name: anything you like (e.g. `OmniRoute Remote`)
-En el campo**"URI de redireccionamiento autorizado"**, agregue:```
+**3. Add Authorized Redirect URIs**
+
+In the **"Authorized redirect URIs"** field, add:
+
+```
https://your-server.com/callback
+```
-````
+> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`).
-> Reemplace `your-server.com` con el dominio o IP de su servidor (incluya el puerto si es necesario, por ejemplo, `http://45.33.32.156:20128/callback`).
+**4. Save and copy the credentials**
-**4. Guarde y copie las credenciales**
+After creating, Google will show the **Client ID** and **Client Secret**.
-Después de la creación, Google mostrará el**ID de cliente**y el**Secreto de cliente**.
+**5. Set environment variables**
-**5. Establecer variables de entorno**
+In your `.env` (or Docker environment variables):
-En su `.env` (o variables de entorno de Docker):```bash
+```bash
# For Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
@@ -1788,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
-````
+```
-**6. Reiniciar OmniRoute**```bash
+**6. Restart OmniRoute**
+```bash
# npm:
-
npm run dev
# Docker:
-
docker restart omniroute
+```
-````
+**7. Try connecting again**
-**7. Intente conectarse nuevamente**
+Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth
-Panel → Proveedores → Antigravity (o Gemini CLI) → OAuth
+Google will now redirect correctly to `https://your-server.com/callback`.
-Google ahora redirigirá correctamente a `https://your-server.com/callback`.---
+---
#### Temporary workaround (without custom credentials)
-Si no desea configurar sus propias credenciales en este momento, aún puede usar el**flujo de URL manual**:
+If you don't want to set up your own credentials right now, you can still use the **manual URL flow**:
-1. OmniRoute abre la URL de autorización de Google.
-2. Después de autorizar, Google intenta redirigir a `localhost` (que falla en el servidor remoto)
-3.**Copia la URL completa**de la barra de direcciones de tu navegador (incluso si la página no se carga)
-4. Pegue esa URL en el campo que se muestra en el modo de conexión de OmniRoute.
-5. Haga clic en**"Conectar"**
+1. OmniRoute opens the Google authorization URL
+2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server)
+3. **Copy the full URL** from your browser's address bar (even if the page doesn't load)
+4. Paste that URL into the field shown in the OmniRoute connection modal
+5. Click **"Connect"**
-> Esto funciona porque el código de autorización en la URL es válido independientemente de si se cargó la página de redireccionamiento.---
+> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded.
-
-🇧🇷 Versión en portugués
#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+---
-Los proveedores**Antigravity**y**Gemini CLI**usan**Google OAuth 2.0**para autenticar. O Google exige que un `redirect_uri` usado sin flujo OAuth seja**exatamente**uma das URI pre-cadastradas en Google Cloud Console de la aplicación.
+
+🇧🇷 Versão em Português
-Como credenciales OAuth embutidas no OmniRoute están catastradas**apenas para `localhost`**. Cuando accede a OmniRoute en un servidor remoto (por ejemplo: `https://omniroute.meuservidor.com`), o Google envía una autenticación con:```
+#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+
+Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo.
+
+As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solução: Configure suas próprias credenciais OAuth
-Debe crear un**ID de cliente OAuth 2.0**en Google Cloud Console con un URI en su servidor.#### Passo a passo
+Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor.
-**1. Acceso a Google Cloud Console**
+#### Passo a passo
+
+**1. Acesse o Google Cloud Console**
Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-**2. Llame a un nuevo ID de cliente OAuth 2.0**
+**2. Crie um novo OAuth 2.0 Client ID**
-- Haga clic en**"+ Crear credenciales"**→**"ID de cliente OAuth"**
-- Tipo de aplicación:**"Aplicación web"**
-- Nombre: escolha qualquer nome (por ejemplo: `OmniRoute Remote`)
+- Clique em **"+ Create Credentials"** → **"OAuth client ID"**
+- Tipo de aplicativo: **"Web application"**
+- Nome: escolha qualquer nome (ex: `OmniRoute Remote`)
-**3. Agregar como URI de redireccionamiento autorizado**
+**3. Adicione as Authorized Redirect URIs**
-No hay campo**"URI de redireccionamiento autorizado"**, además:```
+No campo **"Authorized redirect URIs"**, adicione:
+
+```
https://seu-servidor.com/callback
+```
-````
+> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`).
-> Sustituye `seu-servidor.com` por el dominio o IP de tu servidor (incluye una porta si es necesaria, por ejemplo: `http://45.33.32.156:20128/callback`).
+**4. Salve e copie as credenciais**
-**4. Salve y copie como credencial**
+Após criar, o Google mostrará o **Client ID** e o **Client Secret**.
-Después de abrir, Google mostrará**ID de cliente**y**Secreto de cliente**.
+**5. Configure as variáveis de ambiente**
-**5. Configurar como variáveis de ambiente**
+No seu `.env` (ou nas variáveis de ambiente do Docker):
-No seu `.env` (o las variaciones de ambiente de Docker):```bash
+```bash
# Para Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
@@ -1867,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
-````
+```
-**6. Reiniciar OmniRoute**```bash
+**6. Reinicie o OmniRoute**
+```bash
# Se usando npm:
-
npm run dev
# Se usando Docker:
-
docker restart omniroute
-
-````
+```
**7. Tente conectar novamente**
-Panel → Proveedores → Antigravity (o Gemini CLI) → OAuth
+Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth
-Agora o Google redirigirá correctamente para `https://seu-servidor.com/callback` y autenticação funcionará.---
+Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará.
+
+---
#### Workaround temporário (sem configurar credenciais próprias)
-Si no quieres crear credenciales propias ahora, aún puedes usar el flujo**manual de URL**:
+Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**:
-1. El OmniRoute abrirá una URL de autorización de Google
-2. Después de autorizar, Google intentará redirigir a `localhost` (que no tiene servidor remoto)
-3.**Copia una URL completa**de la barra de envío de tu navegador (también que a página no carregue)
-4. Cole esa URL en el campo que aparece en el modo de conexión de OmniRoute
-5. Haz clic en**"Conectar"**
+1. O OmniRoute abrirá a URL de autorização do Google
+2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto)
+3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue)
+4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute
+5. Clique em **"Connect"**
-> Esta solución funciona porque el código de autorización de la URL es válido independiente de la redirección ter cargada o no.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+
+
---
@@ -1905,64 +2171,73 @@ Si no quieres crear credenciales propias ahora, aún puedes usar el flujo**manua
## 🛠️ Tech Stack
-
-Haga clic para ampliar los detalles de la pila tecnológica
+
+Click to expand tech stack details
--**Tiempo de ejecución**: Node.js 18–22 LTS (⚠️ Node.js 24+**no es compatible**; los archivos binarios nativos `better-sqlite3` son incompatibles)
--**Idioma**: TypeScript 5.9 —**100% TypeScript**en `src/` y `open-sse/` (cero `any` en los módulos principales desde v2.0)
--**Marco**: Next.js 16 + React 19 + Tailwind CSS 4
--**Base de datos**: LowDB (JSON) + SQLite (estado de dominio + registros de proxy + auditoría de MCP + decisiones de enrutamiento)
--**Esquemas**: Zod (validación de E/S de herramienta MCP, contratos API)
--**Protocolos**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
--**Transmisión**: Eventos enviados por el servidor (SSE)
--**Auth**: OAuth 2.0 (PKCE) + JWT + Claves API + Autorización con alcance MCP
--**Pruebas**: Ejecutor de pruebas de Node.js + Vitest (más de 900 pruebas que incluyen unidad, integración, E2E)
--**CI/CD**: Acciones de GitHub (publicación automática de npm + Docker Hub en el lanzamiento)
--**Sitio web**: [omniroute.online](https://omniroute.online)
--**Paquete**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
--**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
--**Resiliencia**: disyuntor, retroceso exponencial, rebaño anti-truenos, suplantación de TLS, autocuración combinada automática
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible)
+- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0)
+- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4
+- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions)
+- **Schemas**: Zod (MCP tool I/O validation, API contracts)
+- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
+- **Streaming**: Server-Sent Events (SSE)
+- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization
+- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E)
+- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release)
+- **Website**: [omniroute.online](https://omniroute.online)
+- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
+- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
+- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing
+
+
---
## Documentación
-| Documento | Descripción |
+| Document | Description |
| ---------------------------------------------- | --------------------------------------------------- |
-| [Guía del usuario](docs/USER_GUIDE.md) | Proveedores, combos, integración CLI, implementación |
-| [Referencia de API](docs/API_REFERENCE.md) | Todos los puntos finales con ejemplos |
-| [Servidor MCP](open-sse/mcp-server/README.md) | 16 herramientas MCP, configuraciones IDE, clientes Python/TS/Go |
-| [Servidor A2A](src/lib/a2a/README.md) | Protocolo JSON-RPC 2.0, habilidades, streaming, gestión de tareas |
-| [Motor de combinación automática](docs/auto-combo.md) | Puntuación de 6 factores, paquetes de modos, autocuración |
-| [Solución de problemas](docs/TROUBLESHOOTING.md) | Problemas comunes y soluciones |
-| [Arquitectura](docs/ARCHITECTURE.md) | Arquitectura del sistema e partes internas |
-| [Contribuyendo](CONTRIBUYENDO.md) | Configuración y pautas de desarrollo |
-| [Especificación de OpenAPI](docs/openapi.yaml) | Especificación OpenAPI 3.0 |
-| [Política de seguridad](SECURITY.md) | Informes de vulnerabilidad y prácticas de seguridad |
-| [Implementación de VM](docs/VM_DEPLOYMENT_GUIDE.md) | Guía completa: configuración de VM + nginx + Cloudflare |
-| [Galería de funciones](docs/FEATURES.md) | Recorrido visual por el panel con capturas de pantalla |
-| [Lista de verificación de lanzamiento](docs/RELEASE_CHECKLIST.md) | Pasos de validación previa al lanzamiento |---
+| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment |
+| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples |
+| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients |
+| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt |
+| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing |
+| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation |
+| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions |
+| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals |
+| [Contributing](CONTRIBUTING.md) | Development setup and guidelines |
+| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification |
+| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices |
+| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup |
+| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots |
+| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps |
+
+---
## 🗺️ Roadmap
-OmniRoute tiene**más de 210 funciones planificadas**en múltiples fases de desarrollo. Estas son las áreas clave:
+OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas:
-| Categoría | Funciones planificadas | Aspectos destacados |
-| ----------------------- | ---------------- | -------------------------------------------------------------------------------------- |
-| 🧠**Enrutamiento e inteligencia**| 25+ | Enrutamiento de latencia más baja, enrutamiento basado en etiquetas, verificación previa de cuotas, selección de cuentas P2C |
-| 🔒**Seguridad y cumplimiento**| 20+ | Refuerzo SSRF, encubrimiento de credenciales, límite de velocidad por punto final, alcance de claves de administración |
-| 📊**Observabilidad**| 15+ | Integración de OpenTelemetry, monitoreo de cuotas en tiempo real, seguimiento de costos por modelo |
-| 🔄**Integraciones de proveedores**| 20+ | Registro de modelo dinámico, tiempos de reutilización de proveedores, Codex multicuenta, análisis de cuotas de Copilot |
-| ⚡**Rendimiento**| 15+ | Capa de caché dual, caché de avisos, caché de respuestas, transmisión keepalive, API por lotes |
-| 🌐**Ecosistema**| 10+ | API WebSocket, recarga en caliente de configuración, almacén de configuración distribuido, modo comercial |### 🔜 Coming Soon
+| Category | Planned Features | Highlights |
+| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- |
+| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection |
+| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping |
+| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model |
+| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing |
+| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API |
+| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode |
-- 🔗**Integración OpenCode**: soporte de proveedor nativo para el IDE de codificación OpenCode AI
-- 🔗**Integración TRAE**: soporte total para el marco de desarrollo de IA de TRAE
-- 📦**API por lotes**: procesamiento por lotes asíncrono para solicitudes masivas
-- 🎯**Enrutamiento basado en etiquetas**: enruta solicitudes basadas en etiquetas y metadatos personalizados
-- 💰**Estrategia de menor costo**: seleccione automáticamente el proveedor más barato disponible
+### 🔜 Coming Soon
-> 📝 Especificaciones completas de funciones disponibles en [`docs/new-features/`](docs/new-features/) (217 especificaciones detalladas)---
+- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE
+- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework
+- 📦 **Batch API** — Asynchronous batch processing for bulk requests
+- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata
+- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider
+
+> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs)
+
+---
## 👥 Contributors
@@ -1970,18 +2245,20 @@ OmniRoute tiene**más de 210 funciones planificadas**en múltiples fases de desa
### How to Contribute
-1. Bifurcar el repositorio
-2. Crea tu rama de funciones (`git checkout -b feature/amazing-feature`)
-3. Confirme sus cambios (`git commit -m 'Agregar característica sorprendente'`)
-4. Empuje a la rama (`git push origin feature/amazing-feature`)
-5. Abra una solicitud de extracción
+1. Fork the repository
+2. Create your feature branch (`git checkout -b feature/amazing-feature`)
+3. Commit your changes (`git commit -m 'Add amazing feature'`)
+4. Push to the branch (`git push origin feature/amazing-feature`)
+5. Open a Pull Request
-Consulte [CONTRIBUTING.md](CONTRIBUTING.md) para obtener pautas detalladas.### Releasing a New Version
+See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines.
+
+### Releasing a New Version
```bash
# Create a release — npm publish happens automatically
gh release create v2.0.0 --title "v2.0.0" --generate-notes
-````
+```
---
@@ -1993,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes
## 🙏 Acknowledgments
-Un agradecimiento especial a**[9router](https://github.com/decolua/9router)**de**[decolua](https://github.com/decolua)**, el proyecto original que inspiró esta bifurcación. OmniRoute se basa en esa increíble base con funciones adicionales, API multimodales y una reescritura completa de TypeScript.
+Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite.
-Un agradecimiento especial a**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**: la implementación original de Go que inspiró este puerto de JavaScript.---
+Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port.
+
+---
## Licencia
-Licencia MIT: consulte [LICENCIA](LICENCIA) para obtener más detalles.---
+MIT License - see [LICENSE](LICENSE) for details.
+
+---
Built with ❤️ for developers who code 24/7
diff --git a/docs/i18n/es/docs/ARCHITECTURE.md b/docs/i18n/es/docs/ARCHITECTURE.md
index 5b58707141..1783363207 100644
--- a/docs/i18n/es/docs/ARCHITECTURE.md
+++ b/docs/i18n/es/docs/ARCHITECTURE.md
@@ -4,80 +4,93 @@
---
-_Última actualización: 2026-03-28_## Executive Summary
-OmniRoute es un panel y una puerta de enlace de enrutamiento de IA local creado en Next.js.
-Proporciona un único punto final compatible con OpenAI (`/v1/*`) y enruta el tráfico a través de múltiples proveedores ascendentes con traducción, respaldo, actualización de tokens y seguimiento de uso.
-Capacidades principales:
+_Last updated: 2026-03-28_
-- Superficie API compatible con OpenAI para CLI/herramientas (28 proveedores)
-- Traducción de solicitudes/respuestas entre formatos de proveedores.
-- Modelo combinado de respaldo (secuencia multimodelo)
-- Respaldo a nivel de cuenta (varias cuentas por proveedor)
-- Gestión de conexión de proveedor de claves OAuth + API
-- Generación de incrustaciones mediante `/v1/embeddings` (6 proveedores, 9 modelos)
-- Generación de imágenes a través de `/v1/images/generaciones` (4 proveedores, 9 modelos)
-- Piense en el análisis de etiquetas (`
...`) para modelos de razonamiento
-- Saneamiento de respuesta para una estricta compatibilidad con OpenAI SDK
-- Normalización de roles (desarrollador → sistema, sistema → usuario) para compatibilidad entre proveedores
-- Conversión de salida estructurada (json_schema → Gemini ResponseSchema)
-- Persistencia local para proveedores, claves, alias, combos, configuraciones, precios.
-- Seguimiento de uso/costos y registro de solicitudes
-- Sincronización en la nube opcional para sincronización multidispositivo/estado
-- Lista de IP permitidas/lista de bloqueo para control de acceso a API
-- Pensando en la gestión del presupuesto (transferencia/automática/personalizada/adaptativa)
-- Inyección rápida del sistema global
-- Seguimiento de sesiones y toma de huellas digitales
-- Limitación de tarifas mejorada por cuenta con perfiles específicos del proveedor
-- Patrón de disyuntor para la resiliencia del proveedor
-- Protección de rebaño anti-truenos con bloqueo mutex
-- Caché de deduplicación de solicitudes basado en firmas
-- Capa de dominio: disponibilidad del modelo, reglas de costos, política de respaldo, política de bloqueo
-- Persistencia del estado del dominio (caché de escritura SQLite para respaldos, presupuestos, bloqueos, disyuntores)
-- Motor de políticas para la evaluación centralizada de solicitudes (bloqueo → presupuesto → respaldo)
-- Solicitar telemetría con agregación de latencia p50/p95/p99
-- ID de correlación (X-Request-Id) para seguimiento de un extremo a otro
-- Registro de auditoría de cumplimiento con opción de exclusión por clave API
-- Marco de evaluación para el aseguramiento de la calidad del LLM.
-- Panel de interfaz de usuario de resiliencia con estado del disyuntor en tiempo real
-- Proveedores modulares de OAuth (12 módulos individuales en `src/lib/oauth/providers/`)
+## Executive Summary
-Modelo de tiempo de ejecución principal:
+OmniRoute is a local AI routing gateway and dashboard built on Next.js.
+It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking.
-- Las rutas de la aplicación Next.js en `src/app/api/*` implementan tanto las API del panel como las API de compatibilidad.
-- Un núcleo de enrutamiento/SSE compartido en `src/sse/*` + `open-sse/*` maneja la ejecución, traducción, transmisión, respaldo y uso del proveedor.## Scope and Boundaries
+Core capabilities:
+
+- OpenAI-compatible API surface for CLI/tools (28 providers)
+- Request/response translation across provider formats
+- Model combo fallback (multi-model sequence)
+- Account-level fallback (multi-account per provider)
+- OAuth + API-key provider connection management
+- Embedding generation via `/v1/embeddings` (6 providers, 9 models)
+- Image generation via `/v1/images/generations` (4 providers, 9 models)
+- Think tag parsing (`
...`) for reasoning models
+- Response sanitization for strict OpenAI SDK compatibility
+- Role normalization (developer→system, system→user) for cross-provider compatibility
+- Structured output conversion (json_schema → Gemini responseSchema)
+- Local persistence for providers, keys, aliases, combos, settings, pricing
+- Usage/cost tracking and request logging
+- Optional cloud sync for multi-device/state sync
+- IP allowlist/blocklist for API access control
+- Thinking budget management (passthrough/auto/custom/adaptive)
+- Global system prompt injection
+- Session tracking and fingerprinting
+- Per-account enhanced rate limiting with provider-specific profiles
+- Circuit breaker pattern for provider resilience
+- Anti-thundering herd protection with mutex locking
+- Signature-based request deduplication cache
+- Domain layer: model availability, cost rules, fallback policy, lockout policy
+- Context Relay: session handoff summaries for account rotation continuity
+- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers)
+- Policy engine for centralized request evaluation (lockout → budget → fallback)
+- Request telemetry with p50/p95/p99 latency aggregation
+- Correlation ID (X-Request-Id) for end-to-end tracing
+- Compliance audit logging with opt-out per API key
+- Eval framework for LLM quality assurance
+- Resilience UI dashboard with real-time circuit breaker status
+- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`)
+
+Primary runtime model:
+
+- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs
+- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage
+
+## Scope and Boundaries
### In Scope
-- Tiempo de ejecución de la puerta de enlace local
-- API de gestión de paneles
-- Autenticación de proveedor y actualización de token
-- Solicitar traducción y transmisión SSE
-- Estado local + persistencia de uso.
-- Orquestación de sincronización en la nube opcional### Out of Scope
+- Local gateway runtime
+- Dashboard management APIs
+- Provider authentication and token refresh
+- Request translation and SSE streaming
+- Local state + usage persistence
+- Optional cloud sync orchestration
-- Implementación del servicio en la nube detrás de `NEXT_PUBLIC_CLOUD_URL`
-- Proveedor SLA/plano de control fuera del proceso local
-- Los propios binarios CLI externos (Claude CLI, Codex CLI, etc.)## Dashboard Surface (Current)
+### Out of Scope
-Páginas principales en `src/app/(dashboard)/dashboard/`:
+- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL`
+- Provider SLA/control plane outside local process
+- External CLI binaries themselves (Claude CLI, Codex CLI, etc.)
-- `/dashboard` — inicio rápido + descripción general del proveedor
-- `/dashboard/endpoint` — proxy de punto final + MCP + A2A + pestañas de punto final API
-- `/dashboard/providers` — conexiones y credenciales de proveedores
-- `/dashboard/combos` — estrategias combinadas, plantillas, reglas de enrutamiento de modelos
-- `/dashboard/costs` — agregación de costos y visibilidad de precios
-- `/dashboard/analytics` — análisis y evaluaciones de uso
-- `/dashboard/limits` — controles de cuota/tasa
-- `/dashboard/cli-tools`: incorporación de CLI, detección de tiempo de ejecución, generación de configuración
-- `/dashboard/agents` — agentes ACP detectados + registro de agente personalizado
-- `/dashboard/media` — área de juegos de imágenes/videos/música
-- `/dashboard/search-tools` — historial y pruebas del proveedor de búsqueda
-- `/dashboard/health`: tiempo de actividad, disyuntores, límites de velocidad
-- `/dashboard/logs` — registros de solicitud/proxy/auditoría/consola
-- `/dashboard/settings`: pestañas de configuración del sistema (general, enrutamiento, valores predeterminados combinados, etc.)
-- `/dashboard/api-manager` — Ciclo de vida de la clave API y permisos del modelo## High-Level System Context
+## Dashboard Surface (Current)
+
+Main pages under `src/app/(dashboard)/dashboard/`:
+
+- `/dashboard` — quick start + provider overview
+- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs
+- `/dashboard/providers` — provider connections and credentials
+- `/dashboard/combos` — combo strategies, templates, model routing rules
+- `/dashboard/costs` — cost aggregation and pricing visibility
+- `/dashboard/analytics` — usage analytics and evaluations
+- `/dashboard/limits` — quota/rate controls
+- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation
+- `/dashboard/agents` — detected ACP agents + custom agent registration
+- `/dashboard/media` — image/video/music playground
+- `/dashboard/search-tools` — search provider testing and history
+- `/dashboard/health` — uptime, circuit breakers, rate limits
+- `/dashboard/logs` — request/proxy/audit/console logs
+- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.)
+- `/dashboard/api-manager` — API key lifecycle and model permissions
+
+## High-Level System Context
```mermaid
flowchart LR
@@ -181,89 +194,99 @@ Management domains:
## 2) SSE + Translation Core
-Módulos de flujo principales:
+Main flow modules:
-- Entrada: `src/sse/handlers/chat.ts`
-- Orquestación central: `open-sse/handlers/chatCore.ts`
-- Adaptadores de ejecución del proveedor: `open-sse/executors/*`
-- Detección de formato/configuración del proveedor: `open-sse/services/provider.ts`
-- Análisis/resolución del modelo: `src/sse/services/model.ts`, `open-sse/services/model.ts`
-- Lógica alternativa de cuenta: `open-sse/services/accountFallback.ts`
-- Registro de traducción: `open-sse/translator/index.ts`
-- Transformaciones de flujo: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
-- Extracción/normalización de uso: `open-sse/utils/usageTracking.ts`
-- Analizador de etiquetas Think: `open-sse/utils/thinkTagParser.ts`
-- Controlador de incrustación: `open-sse/handlers/embeddings.ts`
-- Registro de proveedores de incrustación: `open-sse/config/embeddingRegistry.ts`
-- Manejador de generación de imágenes: `open-sse/handlers/imageGeneration.ts`
-- Registro del proveedor de imágenes: `open-sse/config/imageRegistry.ts`
-- Sanitización de respuestas: `open-sse/handlers/responseSanitizer.ts`
-- Normalización de roles: `open-sse/services/roleNormalizer.ts`
+- Entry: `src/sse/handlers/chat.ts`
+- Core orchestration: `open-sse/handlers/chatCore.ts`
+- Provider execution adapters: `open-sse/executors/*`
+- Format detection/provider config: `open-sse/services/provider.ts`
+- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts`
+- Account fallback logic: `open-sse/services/accountFallback.ts`
+- Translation registry: `open-sse/translator/index.ts`
+- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
+- Usage extraction/normalization: `open-sse/utils/usageTracking.ts`
+- Think tag parser: `open-sse/utils/thinkTagParser.ts`
+- Embedding handler: `open-sse/handlers/embeddings.ts`
+- Embedding provider registry: `open-sse/config/embeddingRegistry.ts`
+- Image generation handler: `open-sse/handlers/imageGeneration.ts`
+- Image provider registry: `open-sse/config/imageRegistry.ts`
+- Response sanitization: `open-sse/handlers/responseSanitizer.ts`
+- Role normalization: `open-sse/services/roleNormalizer.ts`
-Servicios (lógica de negocios):
+Services (business logic):
-- Selección/puntuación de cuenta: `open-sse/services/accountSelector.ts`
-- Gestión del ciclo de vida del contexto: `open-sse/services/contextManager.ts`
-- Aplicación del filtro IP: `open-sse/services/ipFilter.ts`
-- Seguimiento de sesión: `open-sse/services/sessionManager.ts`
-- Solicitar deduplicación: `open-sse/services/signatureCache.ts`
-- Inyección de aviso del sistema: `open-sse/services/systemPrompt.ts`
-- Pensando en la gestión del presupuesto: `open-sse/services/thinkingBudget.ts`
-- Enrutamiento del modelo comodín: `open-sse/services/wildcardRouter.ts`
-- Gestión de límites de tarifas: `open-sse/services/rateLimitManager.ts`
-- Disyuntor: `open-sse/services/circuitBreaker.ts`
+- Account selection/scoring: `open-sse/services/accountSelector.ts`
+- Context lifecycle management: `open-sse/services/contextManager.ts`
+- IP filter enforcement: `open-sse/services/ipFilter.ts`
+- Session tracking: `open-sse/services/sessionManager.ts`
+- Request deduplication: `open-sse/services/signatureCache.ts`
+- System prompt injection: `open-sse/services/systemPrompt.ts`
+- Thinking budget management: `open-sse/services/thinkingBudget.ts`
+- Wildcard model routing: `open-sse/services/wildcardRouter.ts`
+- Rate limit management: `open-sse/services/rateLimitManager.ts`
+- Circuit breaker: `open-sse/services/circuitBreaker.ts`
+- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy
+- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions
-Módulos de capa de dominio:
+Domain layer modules:
-- Disponibilidad del modelo: `src/lib/domain/modelAvailability.ts`
-- Reglas de costos/presupuestos: `src/lib/domain/costRules.ts`
-- Política alternativa: `src/lib/domain/fallbackPolicy.ts`
-- Resolución combinada: `src/lib/domain/comboResolver.ts`
-- Política de bloqueo: `src/lib/domain/lockoutPolicy.ts`
-- Motor de políticas: `src/domain/policyEngine.ts` — bloqueo centralizado → presupuesto → evaluación alternativa
-- Catálogo de códigos de error: `src/lib/domain/errorCodes.ts`
-- ID de solicitud: `src/lib/domain/requestId.ts`
-- Recuperar tiempo de espera: `src/lib/domain/fetchTimeout.ts`
-- Solicitar telemetría: `src/lib/domain/requestTelemetry.ts`
-- Cumplimiento/auditoría: `src/lib/domain/compliance/index.ts`
-- Corredor de evaluación: `src/lib/domain/evalRunner.ts`
-- Persistencia del estado del dominio: `src/lib/db/domainState.ts` — SQLite CRUD para cadenas de respaldo, presupuestos, historial de costos, estado de bloqueo, disyuntores
+- Model availability: `src/lib/domain/modelAvailability.ts`
+- Cost rules/budgets: `src/lib/domain/costRules.ts`
+- Fallback policy: `src/lib/domain/fallbackPolicy.ts`
+- Combo resolver: `src/lib/domain/comboResolver.ts`
+- Lockout policy: `src/lib/domain/lockoutPolicy.ts`
+- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation
+- Error codes catalog: `src/lib/domain/errorCodes.ts`
+- Request ID: `src/lib/domain/requestId.ts`
+- Fetch timeout: `src/lib/domain/fetchTimeout.ts`
+- Request telemetry: `src/lib/domain/requestTelemetry.ts`
+- Compliance/audit: `src/lib/domain/compliance/index.ts`
+- Eval runner: `src/lib/domain/evalRunner.ts`
+- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers
-Módulos de proveedor de OAuth (12 archivos individuales en `src/lib/oauth/providers/`):
+OAuth provider modules (12 individual files under `src/lib/oauth/providers/`):
-- Índice de registro: `src/lib/oauth/providers/index.ts`
-- Proveedores individuales: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
-- Contenedor delgado: `src/lib/oauth/providers.ts` — reexportaciones desde módulos individuales## 3) Persistence Layer
+- Registry index: `src/lib/oauth/providers/index.ts`
+- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
+- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules
-Base de datos de estado primario (SQLite):
+## 3) Persistence Layer
-- Infraestructura principal: `src/lib/db/core.ts` (better-sqlite3, migraciones, WAL)
-- Reexportación de fachada: `src/lib/localDb.ts` (capa delgada de compatibilidad para quienes llaman)
-- archivo: `${DATA_DIR}/storage.sqlite` (o `$XDG_CONFIG_HOME/omniroute/storage.sqlite` cuando está configurado, en caso contrario `~/.omniroute/storage.sqlite`)
-- entidades (tablas + espacios de nombres KV): conexiones de proveedor, nodos de proveedor, alias de modelo, combos, claves de API, configuración, precios,**modelos personalizados**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt**
+Primary state DB (SQLite):
-Persistencia de uso:
+- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL)
+- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers)
+- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`)
+- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt**
-- fachada: `src/lib/usageDb.ts` (módulos descompuestos en `src/lib/usage/*`)
-- Tablas SQLite en `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs`
-- Los artefactos de archivos opcionales permanecen para compatibilidad/depuración (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `
/logs/...`)
-- Los archivos JSON heredados se migran a SQLite mediante migraciones de inicio cuando están presentes
+Usage persistence:
-Base de datos de estado de dominio (SQLite):
+- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`)
+- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs`
+- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`)
+- legacy JSON files are migrated to SQLite by startup migrations when present
-- `src/lib/db/domainState.ts` — Operaciones CRUD para el estado del dominio
-- Tablas (creadas en `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
-- Patrón de caché de escritura simultánea: los mapas en memoria tienen autoridad en tiempo de ejecución; las mutaciones se escriben sincrónicamente en SQLite; El estado se restaura desde la base de datos en el arranque en frío.## 4) Auth + Security Surfaces
+Domain State DB (SQLite):
-- Autenticación de cookies del panel: `src/proxy.ts`, `src/app/api/auth/login/route.ts`
-- Generación/verificación de clave API: `src/shared/utils/apiKey.ts`
-- Los secretos del proveedor persistieron en las entradas de `providerConnections`
-- Soporte de proxy saliente a través de `open-sse/utils/proxyFetch.ts` (env vars) y `open-sse/utils/networkProxy.ts` (configurable por proveedor o global)## 5) Cloud Sync
+- `src/lib/db/domainState.ts` — CRUD operations for domain state
+- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
+- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start
-- Inicio del programador: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
-- Tarea periódica: `src/shared/services/cloudSyncScheduler.ts`
-- Tarea periódica: `src/shared/services/modelSyncScheduler.ts`
-- Ruta de control: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`)
+## 4) Auth + Security Surfaces
+
+- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts`
+- API key generation/verification: `src/shared/utils/apiKey.ts`
+- Provider secrets persisted in `providerConnections` entries
+- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global)
+
+## 5) Cloud Sync
+
+- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
+- Periodic task: `src/shared/services/cloudSyncScheduler.ts`
+- Periodic task: `src/shared/services/modelSyncScheduler.ts`
+- Control route: `src/app/api/sync/cloud/route.ts`
+
+## Request Lifecycle (`/v1/chat/completions`)
```mermaid
sequenceDiagram
@@ -340,7 +363,9 @@ flowchart TD
Q -- No --> R[Return all unavailable]
```
-Las decisiones de respaldo están impulsadas por `open-sse/services/accountFallback.ts` utilizando códigos de estado y heurísticas de mensajes de error. El enrutamiento combinado agrega una protección adicional: los 400 con alcance del proveedor, como los errores de validación de roles y bloques de contenido ascendentes, se tratan como errores del modelo local, por lo que los destinos combinados posteriores aún pueden ejecutarse.## OAuth Onboarding and Token Refresh Lifecycle
+Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run.
+
+## OAuth Onboarding and Token Refresh Lifecycle
```mermaid
sequenceDiagram
@@ -370,7 +395,9 @@ sequenceDiagram
Test-->>UI: validation result
```
-La actualización durante el tráfico en vivo se ejecuta dentro de `open-sse/handlers/chatCore.ts` a través del ejecutor `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable)
+Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`.
+
+## Cloud Sync Lifecycle (Enable / Sync / Disable)
```mermaid
sequenceDiagram
@@ -402,7 +429,9 @@ sequenceDiagram
Sync-->>UI: disabled
```
-La sincronización periódica la activa "CloudSyncScheduler" cuando la nube está habilitada.## Data Model and Storage Map
+Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled.
+
+## Data Model and Storage Map
```mermaid
erDiagram
@@ -503,12 +532,14 @@ erDiagram
}
```
-Archivos de almacenamiento físico:
+Physical storage files:
-- Base de datos de ejecución principal: `${DATA_DIR}/storage.sqlite`
-- solicitar líneas de registro: `${DATA_DIR}/log.txt` (artefacto de compatibilidad/depuración)
-- archivos de carga útil de llamadas estructuradas: `${DATA_DIR}/call_logs/`
-- traductor opcional/solicitar sesiones de depuración: `/logs/...`## Deployment Topology
+- primary runtime DB: `${DATA_DIR}/storage.sqlite`
+- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact)
+- structured call payload archives: `${DATA_DIR}/call_logs/`
+- optional translator/request debug sessions: `/logs/...`
+
+## Deployment Topology
```mermaid
flowchart LR
@@ -543,205 +574,249 @@ flowchart LR
### Route and API Modules
-- `src/app/api/v1/*`, `src/app/api/v1beta/*`: API de compatibilidad
-- `src/app/api/v1/providers/[provider]/*`: rutas dedicadas por proveedor (chat, incrustaciones, imágenes)
-- `src/app/api/providers*`: proveedor CRUD, validación, pruebas
-- `src/app/api/provider-nodes*`: gestión personalizada de nodos compatibles
-- `src/app/api/provider-models`: gestión de modelos personalizados (CRUD)
-- `src/app/api/models/route.ts`: API de catálogo de modelos (alias + modelos personalizados)
-- `src/app/api/oauth/*`: OAuth/flujos de código de dispositivo
-- `src/app/api/keys*`: ciclo de vida de la clave API local
-- `src/app/api/models/alias`: gestión de alias
-- `src/app/api/combos*`: gestión de combos alternativos
-- `src/app/api/pricing`: anulación de precios para el cálculo de costos
-- `src/app/api/settings/proxy`: configuración del proxy (GET/PUT/DELETE)
-- `src/app/api/settings/proxy/test`: prueba de conectividad de proxy saliente (POST)
-- `src/app/api/usage/*`: API de uso y registros
-- `src/app/api/sync/*` + `src/app/api/cloud/*`: sincronización en la nube y ayudantes orientados a la nube
-- `src/app/api/cli-tools/*`: escritores/comprobadores de configuración CLI local
-- `src/app/api/settings/ip-filter`: lista de IP permitidas/lista de bloqueo (GET/PUT)
-- `src/app/api/settings/thinking-budget`: configuración del presupuesto del token de pensamiento (GET/PUT)
-- `src/app/api/settings/system-prompt`: indicador global del sistema (GET/PUT)
-- `src/app/api/sessions`: listado de sesiones activas (GET)
-- `src/app/api/rate-limits`: estado del límite de tasa por cuenta (GET)### Routing and Execution Core
+- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs
+- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images)
+- `src/app/api/providers*`: provider CRUD, validation, testing
+- `src/app/api/provider-nodes*`: custom compatible node management
+- `src/app/api/provider-models`: custom model management (CRUD)
+- `src/app/api/models/route.ts`: model catalog API (aliases + custom models)
+- `src/app/api/oauth/*`: OAuth/device-code flows
+- `src/app/api/keys*`: local API key lifecycle
+- `src/app/api/models/alias`: alias management
+- `src/app/api/combos*`: fallback combo management
+- `src/app/api/pricing`: pricing overrides for cost calculation
+- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE)
+- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST)
+- `src/app/api/usage/*`: usage and logs APIs
+- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers
+- `src/app/api/cli-tools/*`: local CLI config writers/checkers
+- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT)
+- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT)
+- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT)
+- `src/app/api/sessions`: active session listing (GET)
+- `src/app/api/rate-limits`: per-account rate limit status (GET)
-- `src/sse/handlers/chat.ts`: análisis de solicitudes, manejo combinado, bucle de selección de cuentas
-- `open-sse/handlers/chatCore.ts`: traducción, envío del ejecutor, reintento/actualización, configuración de flujo
-- `open-sse/executors/*`: comportamiento de formato y red específico del proveedor### Translation Registry and Format Converters
+### Routing and Execution Core
-- `open-sse/translator/index.ts`: registro y orquestación de traductores
-- Solicitar traductores: `open-sse/translator/request/*`
-- Traductores de respuesta: `open-sse/translator/response/*`
-- Constantes de formato: `open-sse/translator/formats.ts`### Persistence
+- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop
+- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup
+- `open-sse/executors/*`: provider-specific network and format behavior
-- `src/lib/db/*`: configuración/estado persistente y persistencia de dominio en SQLite
-- `src/lib/localDb.ts`: reexportación de compatibilidad para módulos DB
-- `src/lib/usageDb.ts`: fachada de historial de uso/registros de llamadas encima de las tablas SQLite## Provider Executor Coverage (Strategy Pattern)
+### Translation Registry and Format Converters
-Cada proveedor tiene un ejecutor especializado que extiende `BaseExecutor` (en `open-sse/executors/base.ts`), que proporciona creación de URL, construcción de encabezados, reintentos con retroceso exponencial, enlaces de actualización de credenciales y el método de orquestación `execute()`.
+- `open-sse/translator/index.ts`: translator registry and orchestration
+- Request translators: `open-sse/translator/request/*`
+- Response translators: `open-sse/translator/response/*`
+- Format constants: `open-sse/translator/formats.ts`
-| Ejecutor | Proveedor(es) | Manejo Especial |
-| ------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | --------------------------------------------------------------------------------------------- |
-| `Ejecutor predeterminado` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Configuración dinámica de URL/encabezado por proveedor |
-| `AntigravityExecutor` | Antigravedad de Google | ID personalizados de proyecto/sesión, reintento después del análisis |
-| `CodexExecutor` | Códice OpenAI | Inyecta instrucciones del sistema, fuerza el esfuerzo de razonamiento |
-| `CursorEjecutor` | Cursor IDE | Protocolo ConnectRPC, codificación Protobuf, solicitud de firma mediante suma de comprobación |
-| `GithubExecutor` | Copiloto de GitHub | Actualización del token Copilot, encabezados que imitan VSCode |
-| `KiroExecutor` | AWS CodeWhisperer/Kiro | Formato binario de AWS EventStream → Conversión SSE |
-| `GeminiCLIExecutor` | Géminis CLI | Ciclo de actualización del token OAuth de Google |
+### Persistence
-Todos los demás proveedores (incluidos los nodos compatibles personalizados) utilizan `DefaultExecutor`.## Provider Compatibility Matrix
+- `src/lib/db/*`: persistent config/state and domain persistence on SQLite
+- `src/lib/localDb.ts`: compatibility re-export for DB modules
+- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables
-| Proveedor | Formato | Autenticación | Corriente | Sin transmisión | Actualización de token | API de uso |
-| ---------------------- | ----------------- | ---------------------------------- | --------------------------- | --------------- | ---------------------- | ------------------------- | ------------------------------ |
-| Claudio | claudio | Clave API/OAuth | ✅ | ✅ | ✅ | ⚠️ Solo administrador |
-| Géminis | géminis | Clave API/OAuth | ✅ | ✅ | ✅ | ⚠️ Consola en la nube |
-| Géminis CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Consola en la nube |
-| Antigravedad | antigravedad | OAuth | ✅ | ✅ | ✅ | ✅ API de cuota completa |
-| Abierta AI | abierto | Clave API | ✅ | ✅ | ❌ | ❌ |
-| Códice | respuestas-openai | OAuth | ✅ forzado | ❌ | ✅ | ✅ Límites de tarifas |
-| Copiloto de GitHub | abierto | OAuth + Token de copiloto | ✅ | ✅ | ✅ | ✅ Instantáneas de cuotas |
-| Cursores | cursor | Suma de comprobación personalizada | ✅ | ✅ | ❌ | ❌ |
-| kiro | kiro | AWS SSO OIDC | ✅ (Transmisión de eventos) | ❌ | ✅ | ✅ Límites de uso |
-| Qwen | abierto | OAuth | ✅ | ✅ | ✅ | ⚠️ Por solicitud |
-| Qoder | abierto | OAuth (básico) | ✅ | ✅ | ✅ | ⚠️ Por solicitud |
-| Enrutador abierto | abierto | Clave API | ✅ | ✅ | ❌ | ❌ |
-| GLM/Kimi/MiniMax | claudio | Clave API | ✅ | ✅ | ❌ | ❌ |
-| Búsqueda profunda | abierto | Clave API | ✅ | ✅ | ❌ | ❌ |
-| Groq | abierto | Clave API | ✅ | ✅ | ❌ | ❌ |
-| xAI (Grok) | abierto | Clave API | ✅ | ✅ | ❌ | ❌ |
-| Mistral | abierto | Clave API | ✅ | ✅ | ❌ | ❌ |
-| Perplejidad | abierto | Clave API | ✅ | ✅ | ❌ | ❌ |
-| Juntos IA | abierto | Clave API | ✅ | ✅ | ❌ | ❌ |
-| Fuegos artificiales AI | abierto | Clave API | ✅ | ✅ | ❌ | ❌ |
-| Cerebras | abierto | Clave API | ✅ | ✅ | ❌ | ❌ |
-| Coherir | abierto | Clave API | ✅ | ✅ | ❌ | ❌ |
-| NIM de NVIDIA | abierto | Clave API | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage |
+## Provider Executor Coverage (Strategy Pattern)
-Los formatos de origen detectados incluyen:
+Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method.
+
+| Executor | Provider(s) | Special Handling |
+| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- |
+| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider |
+| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing |
+| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort |
+| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum |
+| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers |
+| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion |
+| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle |
+
+All other providers (including custom compatible nodes) use the `DefaultExecutor`.
+
+## Provider Compatibility Matrix
+
+| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API |
+| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ |
+| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only |
+| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API |
+| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits |
+| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots |
+| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ |
+| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits |
+| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request |
+| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request |
+| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ |
+| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+
+## Format Translation Coverage
+
+Detected source formats include:
- `openai`
-- `respuestas openai`
-- `claudio`
-- `géminis`
+- `openai-responses`
+- `claude`
+- `gemini`
-Los formatos de destino incluyen:
+Target formats include:
-- Chat/Respuestas de OpenAI
-- Claudio
-- Géminis/Gemini-CLI/sobre antigravedad
- -Kiro
-- Cursores
+- OpenAI chat/Responses
+- Claude
+- Gemini/Gemini-CLI/Antigravity envelope
+- Kiro
+- Cursor
-Las traducciones utilizan**OpenAI como formato central**; todas las conversiones pasan por OpenAI como formato intermedio:```
+Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate:
+
+```
Source Format → OpenAI (hub) → Target Format
+```
-````
+Translations are selected dynamically based on source payload shape and provider target format.
-Las traducciones se seleccionan dinámicamente según la forma de la carga útil de origen y el formato de destino del proveedor.
+Additional processing layers in the translation pipeline:
-Capas de procesamiento adicionales en el proceso de traducción:
+- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance
+- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE)
+- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field
+- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema`
--**Desinfección de respuestas**: elimina los campos no estándar de las respuestas en formato OpenAI (tanto en streaming como sin streaming) para garantizar el estricto cumplimiento del SDK.
--**Normalización de roles**: convierte `desarrollador` → `sistema` para objetivos que no son OpenAI; fusiona `sistema` → `usuario` para modelos que rechazan el rol del sistema (GLM, ERNIE)
--**Extracción de etiquetas Think**: analiza los bloques `...` del contenido en el campo `reasoning_content`
--**Salida estructurada**: convierte OpenAI `response_format.json_schema` en `responseMimeType` + `responseSchema` de Gemini.## Supported API Endpoints
+## Supported API Endpoints
-| Punto final | Formato | Manejador |
+| Endpoint | Format | Handler |
| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- |
-| `POST /v1/chat/compleciones` | Chat abierto de IA | `src/sse/handlers/chat.ts` |
-| `POST /v1/mensajes` | Mensajes de Claude | Mismo controlador (detectado automáticamente) |
-| `POST /v1/respuestas` | Respuestas de OpenAI | `open-sse/handlers/responsesHandler.ts` |
-| `POST /v1/incrustaciones` | Incrustaciones de OpenAI | `open-sse/handlers/embeddings.ts` |
-| `GET /v1/incrustaciones` | Listado de modelos | Ruta API |
-| `POST /v1/imagenes/generaciones` | Imágenes de OpenAI | `open-sse/handlers/imageGeneration.ts` |
-| `GET /v1/images/generaciones` | Listado de modelos | Ruta API |
-| `POST /v1/proveedores/{proveedor}/chat/completions` | Chat abierto de IA | Dedicado por proveedor con validación de modelo |
-| `POST /v1/proveedores/{proveedor}/incrustaciones` | Incrustaciones de OpenAI | Dedicado por proveedor con validación de modelo |
-| `POST /v1/proveedores/{proveedor}/images/generaciones` | Imágenes de OpenAI | Dedicado por proveedor con validación de modelo |
-| `POST /v1/mensajes/count_tokens` | Recuento de fichas de Claude | Ruta API |
-| `OBTENER /v1/modelos` | Lista de modelos OpenAI | Ruta API (chat + incrustación + imagen + modelos personalizados) |
-| `OBTENER /api/modelos/catalog` | Catálogo | Todos los modelos agrupados por proveedor + tipo |
-| `POST /v1beta/models/*:streamGenerateContent` | Nativo de Géminis | Ruta API |
-| `OBTENER/PONER/BORRAR /api/settings/proxy` | Configuración de proxy | Configuración del proxy de red |
-| `POST /api/configuración/proxy/prueba` | Conectividad de proxy | Punto final de prueba de conectividad/estado del proxy |
-| `GET/POST/DELETE /api/provider-models` | Modelos de proveedores | Metadatos del modelo de proveedor que respaldan los modelos disponibles personalizados y administrados |## Bypass Handler
+| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` |
+| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) |
+| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` |
+| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` |
+| `GET /v1/embeddings` | Model listing | API route |
+| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` |
+| `GET /v1/images/generations` | Model listing | API route |
+| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation |
+| `POST /v1/messages/count_tokens` | Claude Token Count | API route |
+| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) |
+| `GET /api/models/catalog` | Catalog | All models grouped by provider + type |
+| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route |
+| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration |
+| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint |
+| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models |
-El controlador de omisión (`open-sse/utils/bypassHandler.ts`) intercepta solicitudes "desechables" conocidas de Claude CLI (pings de preparación, extracciones de títulos y recuentos de tokens) y devuelve una**respuesta falsa**sin consumir tokens del proveedor ascendente. Esto se activa solo cuando "User-Agent" contiene "claude-cli".## Request Logger Pipeline
+## Bypass Handler
-El registrador de solicitudes (`open-sse/utils/requestLogger.ts`) proporciona un canal de registro de depuración de 7 etapas, deshabilitado de forma predeterminada, habilitado a través de `ENABLE_REQUEST_LOGS=true`:```
+The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`.
+
+## Request Logger Pipeline
+
+The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`:
+
+```
1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json
→ 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt
-````
+```
-Los archivos se escriben en `/logs//` para cada sesión de solicitud.## Failure Modes and Resilience
+Files are written to `/logs//` for each request session.
+
+## Failure Modes and Resilience
## 1) Account/Provider Availability
-- tiempo de reutilización de la cuenta del proveedor en errores transitorios/de tasa/autenticación
-- respaldo de la cuenta antes de fallar la solicitud
-- retroceso del modelo combinado cuando se agota la ruta del modelo/proveedor actual## 2) Token Expiry
+- provider account cooldown on transient/rate/auth errors
+- account fallback before failing request
+- combo model fallback when current model/provider path is exhausted
-- verificación previa y actualización con reintento para proveedores actualizables
-- Reintento 401/403 después de un intento de actualización en la ruta principal## 3) Stream Safety
+## 2) Token Expiry
-- controlador de flujo con reconocimiento de desconexión
-- flujo de traducción con descarga de final de flujo y manejo `[DONE]`
-- reserva de estimación de uso cuando faltan metadatos de uso del proveedor## 4) Cloud Sync Degradation
+- pre-check and refresh with retry for refreshable providers
+- 401/403 retry after refresh attempt in core path
-- Aparecen errores de sincronización pero el tiempo de ejecución local continúa
-- El programador tiene una lógica con capacidad de reintento, pero la ejecución periódica actualmente llama a la sincronización de un solo intento de forma predeterminada.## 5) Data Integrity
+## 3) Stream Safety
-- Migraciones de esquema SQLite y enlaces de actualización automática al inicio
-- JSON heredado → ruta de compatibilidad de migración SQLite## Observability and Operational Signals
+- disconnect-aware stream controller
+- translation stream with end-of-stream flush and `[DONE]` handling
+- usage estimation fallback when provider usage metadata is missing
-Fuentes de visibilidad en tiempo de ejecución:
+## 4) Cloud Sync Degradation
-- registros de consola desde `src/sse/utils/logger.ts`
-- agregados de uso por solicitud en SQLite (`usage_history`, `call_logs`, `proxy_logs`)
-- capturas de carga útil detalladas en cuatro etapas en SQLite (`request_detail_logs`) cuando `settings.detailed_logs_enabled=true`
-- registro de estado de solicitud textual en `log.txt` (opcional/compatible)
-- registros de traducción/solicitud profunda opcionales en `logs/` cuando `ENABLE_REQUEST_LOGS=true`
-- puntos finales de uso del panel (`/api/usage/*`) para el consumo de UI
+- sync errors are surfaced but local runtime continues
+- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default
-La captura de carga útil de solicitud detallada almacena hasta cuatro etapas de carga útil JSON por llamada enrutada:
+## 5) Data Integrity
-- solicitud sin procesar recibida del cliente
-- solicitud traducida realmente enviada en sentido ascendente
-- respuesta del proveedor reconstruida como JSON; las respuestas transmitidas se compactan en el resumen final más los metadatos de la transmisión
-- respuesta final del cliente devuelta por OmniRoute; las respuestas transmitidas se almacenan en el mismo formulario de resumen compacto## Security-Sensitive Boundaries
+- SQLite schema migrations and auto-upgrade hooks at startup
+- legacy JSON → SQLite migration compatibility path
-- JWT secret (`JWT_SECRET`) protege la verificación/firma de cookies de sesión del panel
-- El arranque de contraseña inicial (`INITIAL_PASSWORD`) debe configurarse explícitamente para el aprovisionamiento de primera ejecución.
-- La clave API HMAC secreta (`API_KEY_SECRET`) protege el formato de clave API local generado
-- Los secretos del proveedor (claves/tokens de API) se conservan en la base de datos local y deben protegerse a nivel del sistema de archivos.
-- Los puntos finales de sincronización en la nube se basan en la semántica de autenticación de clave API + ID de máquina## Environment and Runtime Matrix
+## Observability and Operational Signals
-Variables de entorno utilizadas activamente por el código:
+Runtime visibility sources:
-- Aplicación/autenticación: `JWT_SECRET`, `INITIAL_PASSWORD`
-- Almacenamiento: `DATA_DIR`
-- Comportamiento de nodo compatible: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
-- Anulación de la base de almacenamiento opcional (Linux/macOS cuando `DATA_DIR` no está configurado): `XDG_CONFIG_HOME`
-- Hashing de seguridad: `API_KEY_SECRET`, `MACHINE_ID_SALT`
-- Registro: `ENABLE_REQUEST_LOGS`
-- Sincronización/URL en la nube: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
-- Proxy saliente: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` y variantes en minúsculas
-- Marcas de características de SOCKS5: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
-- Ayudantes de plataforma/tiempo de ejecución (no configuración específica de la aplicación): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes
+- console logs from `src/sse/utils/logger.ts`
+- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`)
+- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true`
+- textual request status log in `log.txt` (optional/compat)
+- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true`
+- dashboard usage endpoints (`/api/usage/*`) for UI consumption
-1. `usageDb` y `localDb` comparten la misma política de directorio base (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) con la migración de archivos heredados.
-2. `/api/v1/route.ts` delega al mismo generador de catálogo unificado utilizado por `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) para evitar la deriva semántica.
-3. El registrador de solicitudes escribe encabezados/cuerpo completo cuando está habilitado; trate el directorio de registro como confidencial.
-4. El comportamiento de la nube depende de la `NEXT_PUBLIC_BASE_URL` correcta y de la accesibilidad del punto final de la nube.
-5. El directorio `open-sse/` se publica como `@omniroute/open-sse`**paquete de espacio de trabajo npm**. El código fuente lo importa a través de `@omniroute/open-sse/...` (resuelto por Next.js `transpilePackages`). Las rutas de archivo en este documento todavía usan el nombre de directorio `open-sse/` para mantener la coherencia.
-6. Los gráficos en el panel utilizan**Recharts**(basados en SVG) para visualizaciones analíticas interactivas y accesibles (gráficos de barras de uso de modelos, tablas de desglose de proveedores con tasas de éxito).
-7. Las pruebas E2E utilizan**Playwright**(`tests/e2e/`), se ejecutan mediante `npm run test:e2e`. Las pruebas unitarias utilizan**ejecutor de pruebas Node.js**(`tests/unit/`), se ejecutan a través de `npm run test:unit`. El código fuente bajo `src/` es**TypeScript**(`.ts`/`.tsx`); el espacio de trabajo `open-sse/` sigue siendo JavaScript (`.js`).
-8. La página de configuración está organizada en 5 pestañas: Seguridad, Enrutamiento (6 estrategias globales: completar primero, por turnos, p2c, aleatorio, menos utilizado, de costo optimizado), Resiliencia (límites de velocidad editables, disyuntor, políticas), IA (presupuesto pensado, aviso del sistema, caché de avisos), Avanzado (proxy).## Operational Verification Checklist
+Detailed request payload capture stores up to four JSON payload stages per routed call:
-- Compilación desde la fuente: `npm run build`
-- Crear imagen de Docker: `docker build -t omniroute.`
-- Iniciar el servicio y verificar:
-- `OBTENER /api/configuración`
-- `OBTENER /api/v1/modelos`
-- La URL base de destino de CLI debe ser `http://:20128/v1` cuando `PORT=20128`
+- raw request received from the client
+- translated request actually sent upstream
+- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata
+- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form
+
+## Security-Sensitive Boundaries
+
+- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing
+- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning
+- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format
+- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level
+- Cloud sync endpoints rely on API key auth + machine id semantics
+
+## Environment and Runtime Matrix
+
+Environment variables actively used by code:
+
+- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD`
+- Storage: `DATA_DIR`
+- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
+- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME`
+- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT`
+- Logging: `ENABLE_REQUEST_LOGS`
+- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
+- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants
+- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
+- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`
+
+## Known Architectural Notes
+
+1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration.
+2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift.
+3. Request logger writes full headers/body when enabled; treat log directory as sensitive.
+4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability.
+5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency.
+6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates).
+7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`).
+8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy).
+9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed.
+10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22.
+11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible.
+
+## Operational Verification Checklist
+
+- Build from source: `npm run build`
+- Build Docker image: `docker build -t omniroute .`
+- Start service and verify:
+- `GET /api/settings`
+- `GET /api/v1/models`
+- CLI target base URL should be `http://:20128/v1` when `PORT=20128`
diff --git a/docs/i18n/es/docs/FEATURES.md b/docs/i18n/es/docs/FEATURES.md
index 6a8945c517..ca348861f0 100644
--- a/docs/i18n/es/docs/FEATURES.md
+++ b/docs/i18n/es/docs/FEATURES.md
@@ -4,102 +4,168 @@
---
-Guía visual de cada sección del panel de OmniRoute.---
+
+
+Visual guide to every section of the OmniRoute dashboard.
+
+---
## 🔌 Providers
-Administre las conexiones de proveedores de IA: proveedores de OAuth (Claude Code, Codex, Gemini CLI), proveedores de claves API (Groq, DeepSeek, OpenRouter) y proveedores gratuitos (Qoder, Qwen, Kiro). Las cuentas Kiro incluyen seguimiento del saldo de crédito: créditos restantes, asignación total y fecha de renovación visibles en Panel → Uso.
+Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage.
+
+
---
## 🎨 Combos
-Cree combinaciones de enrutamiento de modelos con 6 estrategias: prioridad, ponderada, por turnos, aleatoria, menos utilizada y de costo optimizado. Cada combo encadena múltiples modelos con respaldo automático e incluye plantillas rápidas y comprobaciones de preparación.
+Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks.
+
+
---
## 📊 Analytics
-Análisis de uso integral con consumo de tokens, estimaciones de costos, mapas de actividad, gráficos de distribución semanal y desgloses por proveedor.
+Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns.
+
+
---
## 🏥 System Health
-Monitoreo en tiempo real: tiempo de actividad, memoria, versión, percentiles de latencia (p50/p95/p99), estadísticas de caché y estados de los disyuntores del proveedor.
+Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states.
+
+
---
## 🔧 Translator Playground
-Cuatro modos para depurar traducciones de API:**Playground**(convertidor de formato),**Chat Tester**(solicitudes en vivo),**Test Bench**(pruebas por lotes) y**Live Monitor**(transmisión en tiempo real).
+Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream).
+
+
---
## 🎮 Model Playground _(v2.0.9+)_
-Pruebe cualquier modelo directamente desde el tablero. Seleccione proveedor, modelo y punto final, escriba mensajes con Monaco Editor, transmita respuestas en tiempo real, cancele la transmisión a mitad de camino y vea métricas de tiempo.---
+Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics.
+
+---
## 🎨 Themes _(v2.0.5+)_
-Temas de colores personalizables para todo el tablero. Elija entre 7 colores preestablecidos (coral, azul, rojo, verde, violeta, naranja, cian) o cree un tema personalizado eligiendo cualquier color hexadecimal. Admite modo claro, oscuro y de sistema.---
+Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode.
+
+---
## ⚙️ Settings
-Panel de configuración completo con pestañas:
+Comprehensive settings panel with tabs:
--**General**— Almacenamiento del sistema, gestión de copias de seguridad (exportación/importación de base de datos) -**Apariencia**: selector de tema (oscuro/claro/sistema), ajustes preestablecidos de tema de color y colores personalizados, visibilidad del registro de estado, controles de visibilidad de elementos de la barra lateral -**Seguridad**: protección API de endpoints, bloqueo de proveedores personalizado, filtrado de IP, información de sesión -**Enrutamiento**: alias de modelo, degradación de tareas en segundo plano -**Resiliencia**: persistencia del límite de velocidad, ajuste de disyuntores, desactivación automática de cuentas prohibidas, monitoreo de vencimiento del proveedor -**Avanzado**: anulaciones de configuración, seguimiento de auditoría de configuración, modo de degradación alternativa
+- **General** — System storage, backup management (export/import database)
+- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls
+- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info
+- **Routing** — Model aliases, background task degradation
+- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration
+- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode
+
+
---
## 🔧 CLI Tools
-Configuración con un clic para herramientas de codificación de IA: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continuar, Cursor y Factory Droid. Incluye aplicación/restablecimiento de configuración automatizada, perfiles de conexión y mapeo de modelos.
+One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping.
+
+
---
## 🤖 CLI Agents _(v2.0.11+)_
-Panel para descubrir y administrar agentes CLI. Muestra una cuadrícula de 14 agentes integrados (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) con:
+Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with:
--**Estado de instalación**: Instalado/No encontrado con detección de versión -**Insignias de protocolo**: stdio, HTTP, etc. -**Agentes personalizados**: registre cualquier herramienta CLI a través del formulario (nombre, binario, comando de versión, argumentos de generación) -**CLI Fingerprint Matching**: alternancia por proveedor para hacer coincidir las firmas de solicitud CLI nativas, lo que reduce el riesgo de prohibición y preserva la IP del proxy.---
+- **Installation status** — Installed / Not Found with version detection
+- **Protocol badges** — stdio, HTTP, etc.
+- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args)
+- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP
+
+---
+
+## 🔗 Context Relay _(v3.5.5+)_
+
+A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context.
+
+Configurable via combo-level or global settings:
+- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%)
+- **Max Messages For Summary** — How much recent history to condense
+- **Summary Model** — Optional override model for generating the handoff summary
+
+Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md).
+
+---
+
+## 🛡️ Proxy Hardening _(v3.5.5+)_
+
+Comprehensive proxy configuration enforcement across the entire request pipeline:
+
+- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments
+- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings
+- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22
+- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS
+
+---
## 🖼️ Media _(v2.0.3+)_
-Genere imágenes, videos y música desde el tablero. Admite OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open y MusicGen.---
+Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen.
+
+---
## 📝 Request Logs
-Registro de solicitudes en tiempo real con filtrado por proveedor, modelo, cuenta y clave API. Muestra códigos de estado, uso de token, latencia y detalles de respuesta.
+Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details.
+
+
---
## 🌐 API Endpoint
-Su punto final API unificado con desglose de capacidades: finalización de chat, API de respuestas, incrustaciones, generación de imágenes, reclasificación, transcripción de audio, texto a voz, moderaciones y claves API registradas. Integración de Cloudflare Quick Tunnel y soporte de proxy en la nube para acceso remoto.
+Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access.
+
+
---
## 🔑 API Key Management
-Cree, alcance y revoque claves API. Cada clave se puede restringir a modelos/proveedores específicos con acceso completo o permisos de solo lectura. Gestión visual de claves con seguimiento de uso.---
+Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking.
+
+---
## 📋 Audit Log
-Seguimiento de acciones administrativas con filtrado por tipo de acción, actor, objetivo, dirección IP y marca de tiempo. Historial completo de eventos de seguridad.---
+Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history.
+
+---
## 🖥️ Desktop Application
-Aplicación de escritorio Native Electron para Windows, macOS y Linux. Ejecute OmniRoute como una aplicación independiente con integración en la bandeja del sistema, soporte sin conexión, actualización automática e instalación con un solo clic.
+Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install.
-Características clave:
+Key features:
-- Sondeo de preparación del servidor (no hay pantalla en blanco durante el arranque en frío)
-- Bandeja del sistema con gestión de puertos.
-- Política de seguridad de contenidos
-- Cerradura de instancia única
-- Actualización automática al reiniciar
-- UI condicionada a la plataforma (semáforos de macOS, barra de título predeterminada de Windows/Linux)
-- Paquete de compilación de Electron reforzado: los `node_modules' vinculados simbólicamente en el paquete independiente se detectan y rechazan antes del empaquetado, lo que evita la dependencia del tiempo de ejecución en la máquina de compilación (v2.5.5+)
+- Server readiness polling (no blank screen on cold start)
+- System tray with port management
+- Content Security Policy
+- Single-instance lock
+- Auto-update on restart
+- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar)
+- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+)
-📖 Consulte [`electron/README.md`](../electron/README.md) para obtener la documentación completa.
+📖 See [`electron/README.md`](../electron/README.md) for full documentation.
diff --git a/docs/i18n/es/docs/TROUBLESHOOTING.md b/docs/i18n/es/docs/TROUBLESHOOTING.md
index 02011c3e25..4e7d35330d 100644
--- a/docs/i18n/es/docs/TROUBLESHOOTING.md
+++ b/docs/i18n/es/docs/TROUBLESHOOTING.md
@@ -4,68 +4,142 @@
---
-Problemas comunes y soluciones para OmniRoute.---
+
+
+Common problems and solutions for OmniRoute.
+
+---
## Quick Fixes
-| Problema | Solución |
-| ------------------------------------------ | ---------------------------------------------------------------------------------------------- | --- |
-| El primer inicio de sesión no funciona | Establezca `INITIAL_PASSWORD` en `.env` (sin valor predeterminado codificado) |
-| El panel se abre en el puerto incorrecto | Establezca `PORT=20128` y `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
-| No hay registros de solicitudes en `logs/` | Establezca `ENABLE_REQUEST_LOGS = verdadero` |
-| EACCES: permiso denegado | Establezca `DATA_DIR=/path/to/writable/dir` para anular `~/.omniroute` |
-| La estrategia de enrutamiento no se guarda | Actualización a v1.4.11+ (corrección del esquema Zod para la persistencia de la configuración) | --- |
+| Problem | Solution |
+| ----------------------------- | ------------------------------------------------------------------ |
+| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) |
+| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
+| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` |
+| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` |
+| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) |
+| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below |
+| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below |
+
+---
+
+## Node.js Compatibility
+
+
+
+### Login page crashes or shows "Module self-registration" error
+
+**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database.
+
+**Symptoms:**
+- Login page shows a blank screen or a server error
+- Console shows `Error: Module did not self-register` or similar native binding errors
+- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected
+
+**Fix:**
+
+1. Install Node.js 22 LTS (recommended):
+ ```bash
+ nvm install 22
+ nvm use 22
+ ```
+2. Verify your version: `node --version` should show `v22.x.x`
+3. Reinstall OmniRoute: `npm install -g omniroute`
+4. Restart: `omniroute`
+
+> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**.
+
+---
+
+## Proxy Issues
+
+
+
+### Provider validation shows "fetch failed"
+
+**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing.
+
+**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically.
+
+### Token health check fails with "fetch failed"
+
+**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection.
+
+**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+.
+
+### SOCKS5 proxy returns "invalid onRequestStart method"
+
+**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation.
+
+**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+.
+
+---
## Provider Issues
### "Language model did not provide messages"
-**Causa:**Cuota de proveedor agotada.
+**Cause:** Provider quota exhausted.
-**Arreglo:**
+**Fix:**
-1. Verifique el rastreador de cuotas del panel
-2. Utilice un combo con niveles alternativos
-3. Cambiar al nivel más barato/gratuito### Rate Limiting
+1. Check dashboard quota tracker
+2. Use a combo with fallback tiers
+3. Switch to cheaper/free tier
-**Causa:**Cuota de suscripción agotada.
+### Rate Limiting
-**Arreglo:**
+**Cause:** Subscription quota exhausted.
-- Agregar respaldo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-- Utilice GLM/MiniMax como copia de seguridad económica### OAuth Token Expired
+**Fix:**
-OmniRoute actualiza automáticamente los tokens. Si los problemas persisten:
+- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Use GLM/MiniMax as cheap backup
-1. Panel de control → Proveedor → Reconectar
-2. Eliminar y volver a agregar la conexión del proveedor.---
+### OAuth Token Expired
+
+OmniRoute auto-refreshes tokens. If issues persist:
+
+1. Dashboard → Provider → Reconnect
+2. Delete and re-add the provider connection
+
+---
## Cloud Issues
### Cloud Sync Errors
-1. Verifique que `BASE_URL` apunte a su instancia en ejecución (por ejemplo, `http://localhost:20128`)
-2. Verifique que `CLOUD_URL` apunte a su punto final en la nube (por ejemplo, `https://omniroute.dev`).
-3. Mantenga los valores `NEXT_PUBLIC_*` alineados con los valores del lado del servidor### Cloud `stream=false` Returns 500
+1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`)
+2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`)
+3. Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**Síntoma:**`Token inesperado 'd'...` en el punto final de la nube para llamadas que no son de transmisión.
+### Cloud `stream=false` Returns 500
-**Causa:**Upstream devuelve la carga útil SSE mientras que el cliente espera JSON.
+**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls.
-**Solución alternativa:**Utilice `stream=true` para llamadas directas en la nube. El tiempo de ejecución local incluye el respaldo SSE → JSON.### Cloud Says Connected but "Invalid API key"
+**Cause:** Upstream returns SSE payload while client expects JSON.
-1. Cree una clave nueva desde el panel local (`/api/keys`)
-2. Ejecute la sincronización en la nube: Habilitar nube → Sincronizar ahora
-3. Las claves antiguas/no sincronizadas aún pueden devolver "401" en la nube---
+**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback.
+
+### Cloud Says Connected but "Invalid API key"
+
+1. Create a fresh key from local dashboard (`/api/keys`)
+2. Run cloud sync: Enable Cloud → Sync Now
+3. Old/non-synced keys can still return `401` on cloud
+
+---
## Docker Issues
### CLI Tool Shows Not Installed
-1. Verifique los campos de tiempo de ejecución: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
-2. Para el modo portátil: use el destino de imagen `runner-cli` (CLI incluidas)
-3. Para el modo de montaje del host: configure `CLI_EXTRA_PATHS` y monte el directorio bin del host como de solo lectura
-4. Si `installed=true` y `runnable=false`: se encontró el binario pero falló la verificación de estado### Quick Runtime Validation
+1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
+2. For portable mode: use image target `runner-cli` (bundled CLIs)
+3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only
+4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck
+
+### Quick Runtime Validation
```bash
curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
@@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,
### High Costs
-1. Verifique las estadísticas de uso en Panel → Uso
-2. Cambie el modelo principal a GLM/MiniMax
-3. Utilice el nivel gratuito (Gemini CLI, Qoder) para tareas no críticas
-4. Establezca presupuestos de costos por clave API: Panel → Claves API → Presupuesto---
+1. Check usage stats in Dashboard → Usage
+2. Switch primary model to GLM/MiniMax
+3. Use free tier (Gemini CLI, Qoder) for non-critical tasks
+4. Set cost budgets per API key: Dashboard → API Keys → Budget
+
+---
## Debugging
### Enable Request Logs
-Establezca `ENABLE_REQUEST_LOGS=true` en su archivo `.env`. Los registros aparecen en el directorio `logs/`.### Check Provider Health
+Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory.
+
+### Check Provider Health
```bash
# Health dashboard
@@ -100,67 +178,95 @@ curl http://localhost:20128/api/monitoring/health
### Runtime Storage
-- Estado principal: `${DATA_DIR}/storage.sqlite` (proveedores, combos, alias, claves, configuraciones)
-- Uso: tablas SQLite en `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + opcional `${DATA_DIR}/log.txt` y `${DATA_DIR}/call_logs/`
-- Solicitar registros: `/logs/...` (cuando `ENABLE_REQUEST_LOGS=true`)---
+- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings)
+- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/`
+- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`)
+
+---
## Circuit Breaker Issues
### Provider stuck in OPEN state
-Cuando el disyuntor de un proveedor está ABIERTO, las solicitudes se bloquean hasta que expire el tiempo de reutilización.
+When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires.
-**Arreglo:**
+**Fix:**
-1. Vaya a**Panel → Configuración → Resiliencia**
-2. Verifique la tarjeta del disyuntor del proveedor afectado.
-3. Haga clic en**Restablecer todo**para borrar todos los interruptores o espere a que expire el tiempo de reutilización.
-4. Verifique que el proveedor esté realmente disponible antes de restablecer### Provider keeps tripping the circuit breaker
+1. Go to **Dashboard → Settings → Resilience**
+2. Check the circuit breaker card for the affected provider
+3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire
+4. Verify the provider is actually available before resetting
-Si un proveedor ingresa repetidamente al estado ABIERTO:
+### Provider keeps tripping the circuit breaker
-1. Marque**Panel → Estado → Estado del proveedor**para ver el patrón de error.
-2. Vaya a**Configuración → Resiliencia → Perfiles de proveedores**y aumente el umbral de falla.
-3. Verifique si el proveedor ha cambiado los límites de API o requiere una nueva autenticación.
-4. Revise la telemetría de latencia: una latencia alta puede causar fallas basadas en el tiempo de espera---
+If a provider repeatedly enters OPEN state:
+
+1. Check **Dashboard → Health → Provider Health** for the failure pattern
+2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold
+3. Check if the provider has changed API limits or requires re-authentication
+4. Review latency telemetry — high latency may cause timeout-based failures
+
+---
## Audio Transcription Issues
### "Unsupported model" error
-- Asegúrate de estar usando el prefijo correcto: `deepgram/nova-3` o `assemblyai/best`
-- Verifique que el proveedor esté conectado en**Panel → Proveedores**### Transcription returns empty or fails
+- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best`
+- Verify the provider is connected in **Dashboard → Providers**
-- Verifique los formatos de audio admitidos: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm`
-- Verifique que el tamaño del archivo esté dentro de los límites del proveedor (normalmente < 25 MB)
-- Verifique la validez de la clave API del proveedor en la tarjeta del proveedor---
+### Transcription returns empty or fails
+
+- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm`
+- Verify file size is within provider limits (typically < 25MB)
+- Check provider API key validity in the provider card
+
+---
## Translator Debugging
-Utilice**Panel → Traductor**para depurar problemas de traducción de formato:
+Use **Dashboard → Translator** to debug format translation issues:
-| Modo | Cuándo utilizar |
-| -------------------------- | ------------------------------------------------------------------------------------------------------------- | ------------------------ |
-| **Parque infantil** | Compare formatos de entrada/salida uno al lado del otro: pegue una solicitud fallida para ver cómo se traduce |
-| **Probador de chat** | Envíe mensajes en vivo e inspeccione la carga útil completa de solicitud/respuesta, incluidos los encabezados |
-| **Banco de pruebas** | Ejecute pruebas por lotes en combinaciones de formatos para encontrar qué traducciones no funcionan |
-| **Monitorización en vivo** | Observe el flujo de solicitudes en tiempo real para detectar problemas de traducción intermitentes | ### Common format issues |
+| Mode | When to Use |
+| ---------------- | -------------------------------------------------------------------------------------------- |
+| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates |
+| **Chat Tester** | Send live messages and inspect the full request/response payload including headers |
+| **Test Bench** | Run batch tests across format combinations to find which translations are broken |
+| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues |
--**Las etiquetas de pensamiento no aparecen**: compruebe si el proveedor objetivo apoya el pensamiento y la configuración del presupuesto de pensamiento. -**Caídas de llamadas a herramientas**: algunas traducciones de formatos pueden eliminar campos no admitidos; verificar en modo Patio de Juegos -**Falta el mensaje del sistema**: Claude y Gemini manejan los mensajes del sistema de manera diferente; comprobar la salida de la traducción -**El SDK devuelve una cadena sin formato en lugar de un objeto**— Corregido en v1.1.0: el desinfectante de respuesta ahora elimina los campos no estándar (`x_groq`, `usage_breakdown`, etc.) que causan fallas de validación de Pydantic en el SDK de OpenAI -**GLM/ERNIE rechaza la función `sistema`**— Corregido en v1.1.0: el normalizador de funciones fusiona automáticamente mensajes del sistema con mensajes de usuario para modelos incompatibles -**Rol de "desarrollador" no reconocido**- Corregido en v1.1.0: convertido automáticamente a "sistema" para proveedores que no son OpenAI -**`json_schema` no funciona con Gemini**— Corregido en v1.1.0: `response_format` ahora se convierte a `responseMimeType` + `responseSchema` de Gemini---
+### Common format issues
+
+- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting
+- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode
+- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output
+- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures
+- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models
+- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers
+- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema`
+
+---
## Resilience Settings
### Auto rate-limit not triggering
-- El límite de velocidad automático solo se aplica a los proveedores de claves API (no a OAuth/suscripción)
-- Verifique que**Configuración → Resiliencia → Perfiles de proveedores**tenga habilitado el límite de tasa automática
-- Verifique si el proveedor devuelve códigos de estado `429` o encabezados `Reintentar después`### Tuning exponential backoff
+- Auto rate-limit only applies to API key providers (not OAuth/subscription)
+- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled
+- Check if the provider returns `429` status codes or `Retry-After` headers
-Los perfiles de proveedor admiten estas configuraciones:
+### Tuning exponential backoff
--**Retraso base**: tiempo de espera inicial después del primer fallo (predeterminado: 1 s) -**Retraso máximo**: límite máximo de tiempo de espera (predeterminado: 30 segundos) -**Multiplicador**: cuánto aumentar el retraso por falla consecutiva (predeterminado: 2x)### Anti-thundering herd
+Provider profiles support these settings:
-Cuando muchas solicitudes simultáneas llegan a un proveedor de velocidad limitada, OmniRoute utiliza mutex + limitación de velocidad automática para serializar solicitudes y evitar fallas en cascada. Esto es automático para los proveedores de claves API.---
+- **Base delay** — Initial wait time after first failure (default: 1s)
+- **Max delay** — Maximum wait time cap (default: 30s)
+- **Multiplier** — How much to increase delay per consecutive failure (default: 2x)
+
+### Anti-thundering herd
+
+When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers.
+
+---
## Optional RAG / LLM failure taxonomy (16 problems)
@@ -199,4 +305,8 @@ You can ignore this section if you do not run RAG or agent pipelines behind Omni
## Still Stuck?
--**Problemas de GitHub**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Arquitectura**: consulte [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) para obtener detalles internos -**Referencia de API**: consulte [`docs/API_REFERENCE.md`](API_REFERENCE.md) para todos los puntos finales -**Panel de estado**: marque**Panel → Salud**para ver el estado del sistema en tiempo real -**Traductor**: use**Panel → Traductor**para depurar problemas de formato
+- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details
+- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints
+- **Health Dashboard**: Check **Dashboard → Health** for real-time system status
+- **Translator**: Use **Dashboard → Translator** to debug format issues
diff --git a/docs/i18n/es/llm.txt b/docs/i18n/es/llm.txt
new file mode 100644
index 0000000000..d07f7bfa97
--- /dev/null
+++ b/docs/i18n/es/llm.txt
@@ -0,0 +1,476 @@
+# OmniRoute (Español)
+
+🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt)
+
+---
+
+
+> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app.
+
+## Resumen
+
+OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free.
+
+**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost.
+
+**Current version:** 3.5.5
+
+## Tech Stack
+
+- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`)
+- **Framework:** Next.js 16 (App Router) with TypeScript 5.9
+- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations)
+- **State management:** Zustand (client), SQLite (server persistence)
+- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons
+- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth
+- **Schemas:** Zod v4 for all API / MCP input validation
+- **Background jobs:** Custom token health check scheduler, 24h model auto-sync
+- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses
+- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine
+- **i18n:** next-intl with 30 languages
+- **Desktop:** Electron (cross-platform: Windows, macOS, Linux)
+- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`)
+
+## Project Structure
+
+```
+/
+├── src/ # Main application source
+│ ├── app/ # Next.js App Router pages and API routes
+│ │ ├── (dashboard)/ # Dashboard UI pages
+│ │ │ └── dashboard/
+│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents)
+│ │ │ ├── analytics/ # Usage analytics and charts
+│ │ │ ├── api-manager/ # API key management
+│ │ │ ├── audit/ # Audit logs
+│ │ │ ├── auto-combo/ # Auto-combo engine dashboard
+│ │ │ ├── cache/ # Cache dashboard (semantic cache stats)
+│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.)
+│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates)
+│ │ │ ├── costs/ # Cost tracking per provider/model
+│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs
+│ │ │ ├── health/ # System health (uptime, circuit breakers, latency)
+│ │ │ ├── limits/ # Rate limits dashboard
+│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed)
+│ │ │ ├── media/ # Image/video/music generation + transcription
+│ │ │ ├── memory/ # Memory system dashboard
+│ │ │ ├── onboarding/ # Onboarding wizard
+│ │ │ ├── playground/ # Model playground (Monaco editor, streaming)
+│ │ │ ├── providers/ # Provider management (OAuth + API key + free)
+│ │ │ ├── search-tools/ # Search tools configuration
+│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced)
+│ │ │ ├── skills/ # Skills system dashboard
+│ │ │ ├── translator/ # Format translator + debug tools
+│ │ │ └── usage/ # Usage history
+│ │ ├── api/ # REST API endpoints (51 route directories)
+│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings,
+│ │ │ │ # images, audio, videos, music, moderations, rerank, search,
+│ │ │ │ # responses, messages, registered-keys, quotas, accounts)
+│ │ │ ├── v1beta/ # Gemini-compatible API
+│ │ │ ├── a2a/ # A2A agent management API
+│ │ │ ├── acp/ # ACP agent management API
+│ │ │ ├── oauth/ # OAuth flows per provider
+│ │ │ ├── providers/ # Provider CRUD and batch testing
+│ │ │ ├── models/ # Dashboard model listing and aliases
+│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains)
+│ │ │ ├── memory/ # Memory system API
+│ │ │ ├── skills/ # Skills system API
+│ │ │ ├── evals/ # Eval runner API
+│ │ │ ├── mcp/ # MCP HTTP transport API
+│ │ │ ├── search/ # Search provider API
+│ │ │ ├── webhooks/ # Webhook management
+│ │ │ ├── tunnels/ # Cloudflare tunnel management
+│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.)
+│ │ ├── landing/ # Landing page
+│ │ ├── login/ # Login page
+│ │ ├── forgot-password/ # Password recovery
+│ │ ├── status/ # Status page
+│ │ └── docs/ # In-app documentation
+│ ├── domain/ # Domain types and policy engine
+│ │ ├── policyEngine.ts # Central policy engine
+│ │ ├── comboResolver.ts # Combo resolution logic
+│ │ ├── costRules.ts # Cost calculation rules
+│ │ ├── degradation.ts # Graceful degradation
+│ │ ├── fallbackPolicy.ts # Fallback behavior
+│ │ ├── lockoutPolicy.ts # Account lockout logic
+│ │ ├── modelAvailability.ts # Model availability checks
+│ │ ├── providerExpiration.ts # Provider credential expiration
+│ │ ├── quotaCache.ts # Quota caching layer
+│ │ ├── configAudit.ts # Configuration auditing
+│ │ └── responses.ts # Domain response types
+│ ├── i18n/ # Internationalization
+│ │ └── messages/ # 30 language JSON files
+│ ├── lib/ # Core libraries
+│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server
+│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting)
+│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup
+│ │ │ └── streaming.ts # SSE streaming for A2A
+│ │ ├── acp/ # Agent Communication Protocol registry and manager
+│ │ ├── compliance/ # Compliance policy engine
+│ │ ├── db/ # SQLite database layer (21 modules + migrations)
+│ │ │ ├── core.ts # Database initialization, connection, schema
+│ │ │ ├── providers.ts # Provider connection CRUD
+│ │ │ ├── models.ts # Model catalog management
+│ │ │ ├── combos.ts # Combo configuration
+│ │ │ ├── apiKeys.ts # API key management
+│ │ │ ├── settings.ts # Settings persistence
+│ │ │ ├── backup.ts # Database backup/restore
+│ │ │ ├── proxies.ts # Proxy registry
+│ │ │ ├── prompts.ts # Prompt templates
+│ │ │ ├── webhooks.ts # Webhook subscriptions
+│ │ │ ├── detailedLogs.ts # Detailed request logging
+│ │ │ ├── domainState.ts # Domain state persistence
+│ │ │ ├── registeredKeys.ts # Registered API keys with quotas
+│ │ │ ├── quotaSnapshots.ts # Quota snapshot history
+│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings
+│ │ │ ├── cliToolState.ts # CLI tool state tracking
+│ │ │ ├── encryption.ts # Data encryption
+│ │ │ ├── readCache.ts # Read-through cache layer
+│ │ │ ├── secrets.ts # Secrets management
+│ │ │ ├── stateReset.ts # State reset utilities
+│ │ │ ├── migrationRunner.ts # Schema migration runner
+│ │ │ └── migrations/ # 16 SQL migration files
+│ │ ├── evals/ # Eval runner and scheduler
+│ │ ├── memory/ # Persistent conversational memory
+│ │ │ ├── extraction.ts # Memory extraction from conversations
+│ │ │ ├── injection.ts # Memory injection into context
+│ │ │ ├── retrieval.ts # Memory retrieval/search
+│ │ │ ├── store.ts # Memory persistence layer
+│ │ │ └── summarization.ts # Memory summarization
+│ │ ├── oauth/ # OAuth providers, services, and utilities
+│ │ │ ├── constants/ # Default OAuth credentials (overridable via env)
+│ │ │ ├── providers/ # Provider-specific OAuth configs
+│ │ │ ├── services/ # Provider-specific token exchange logic
+│ │ │ └── utils/ # PKCE, callback server, token helpers
+│ │ ├── plugins/ # Plugin system
+│ │ ├── skills/ # Extensible skill framework
+│ │ │ ├── registry.ts # Skill registration
+│ │ │ ├── executor.ts # Skill execution engine
+│ │ │ ├── sandbox.ts # Skill sandbox environment
+│ │ │ ├── builtin/ # Built-in skills
+│ │ │ ├── interception.ts # Skill request interception
+│ │ │ └── injection.ts # Skill context injection
+│ │ ├── usage/ # Usage tracking system
+│ │ │ ├── callLogs.ts # Call log persistence
+│ │ │ ├── costCalculator.ts # Cost calculation engine
+│ │ │ └── usageHistory.ts # Usage history queries
+│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers
+│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management
+│ │ ├── pricingSync.ts # LiteLLM pricing data sync
+│ │ ├── semanticCache.ts # Semantic caching layer
+│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler
+│ │ ├── webhookDispatcher.ts # Webhook event dispatcher
+│ │ └── localDb.ts # Unified re-export layer for all DB modules
+│ ├── middleware/ # Request middleware
+│ │ └── promptInjectionGuard.ts # Prompt injection detection
+│ ├── mitm/ # MITM proxy capability
+│ │ ├── cert/ # Certificate management
+│ │ ├── dns/ # DNS handling
+│ │ ├── targets/ # Target routing
+│ │ └── manager.ts # MITM proxy manager
+│ ├── shared/ # Shared utilities, components, and constants
+│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.)
+│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes
+│ │ ├── contracts/ # Shared API contracts
+│ │ ├── hooks/ # React hooks
+│ │ ├── middleware/ # Shared middleware utilities
+│ │ ├── schemas/ # Shared Zod schemas
+│ │ ├── services/ # Shared services
+│ │ ├── types/ # Shared TypeScript types
+│ │ ├── validation/ # Zod schemas (settings, providers, routes)
+│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID)
+│ ├── sse/ # SSE proxy pipeline
+│ │ ├── services/ # Auth resolution, format translation, response handling
+│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency
+│ ├── store/ # Zustand client-side stores (theme, providers, etc.)
+│ └── types/ # TypeScript type definitions
+├── open-sse/ # Standalone SSE server (npm workspace)
+│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video,
+│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models)
+│ ├── executors/ # Provider-specific request executors (14 executors)
+│ │ ├── base.ts # Base executor with shared logic
+│ │ ├── default.ts # Default OpenAI-compatible executor
+│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum)
+│ │ ├── codex.ts # OpenAI Codex CLI
+│ │ ├── antigravity.ts # Antigravity IDE
+│ │ ├── github.ts # GitHub Copilot
+│ │ ├── gemini-cli.ts # Gemini CLI
+│ │ ├── kiro.ts # Kiro AI
+│ │ ├── qoder.ts # Qoder AI
+│ │ ├── vertex.ts # Vertex AI (Service Account JSON)
+│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI
+│ │ ├── opencode.ts # OpenCode Zen/Go
+│ │ ├── pollinations.ts # Pollinations AI
+│ │ └── puter.ts # Puter AI
+│ ├── handlers/ # Request handlers per API type (11 handlers)
+│ │ ├── chatCore.ts # Main chat completions handler
+│ │ ├── responsesHandler.ts # OpenAI Responses API handler
+│ │ ├── embeddings.ts # Embedding generation
+│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.)
+│ │ ├── videoGeneration.ts # Video generation
+│ │ ├── musicGeneration.ts # Music generation
+│ │ ├── audioSpeech.ts # Text-to-speech
+│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI)
+│ │ ├── moderations.ts # Content moderation
+│ │ ├── rerank.ts # Reranking API
+│ │ └── search.ts # Web search API
+│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP)
+│ │ ├── server.ts # MCP server core (tool registration, scope enforcement)
+│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools)
+│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a)
+│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes)
+│ │ ├── audit.ts # Tool call audit logging
+│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat
+│ │ └── httpTransport.ts # HTTP transport handler
+│ ├── services/ # 36+ service modules
+│ │ ├── combo.ts # Core routing engine
+│ │ ├── usage.ts # Usage tracking
+│ │ ├── tokenRefresh.ts # OAuth token refresh
+│ │ ├── rateLimitManager.ts # Rate limit management
+│ │ ├── accountFallback.ts # Multi-account fallback
+│ │ ├── sessionManager.ts # Session management
+│ │ ├── wildcardRouter.ts # Wildcard model routing
+│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration)
+│ │ ├── intentClassifier.ts # Request intent classification
+│ │ ├── taskAwareRouter.ts # Task-aware routing
+│ │ ├── thinkingBudget.ts # Thinking budget management
+│ │ ├── contextManager.ts # Context window management
+│ │ ├── modelDeprecation.ts # Model deprecation handling
+│ │ ├── modelFamilyFallback.ts # Intra-family model fallback
+│ │ ├── emergencyFallback.ts # Emergency fallback
+│ │ ├── workflowFSM.ts # Workflow state machine
+│ │ ├── backgroundTaskDetector.ts # Background task detection
+│ │ ├── ipFilter.ts # IP-based access control
+│ │ ├── signatureCache.ts # CLI signature caching
+│ │ ├── volumeDetector.ts # Request volume detection
+│ │ ├── contextHandoff.ts # Context relay handoff generation and injection
+│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay
+│ │ └── ... # Additional services (14 more modules)
+│ ├── transformer/ # Responses API transformer
+│ │ └── responsesTransformer.ts
+│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek)
+│ │ ├── request/ # Request translators per provider
+│ │ ├── response/ # Response translators per provider
+│ │ ├── helpers/ # Translation helpers
+│ │ └── image/ # Image format translation
+│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.)
+├── electron/ # Electron desktop app (cross-platform)
+│ ├── main.js # Electron main process
+│ ├── preload.js # Preload script (IPC bridge)
+│ └── assets/ # App icons and assets
+├── tests/ # Test suites
+│ ├── unit/ # 122 unit test files
+│ ├── integration/ # Integration tests
+│ ├── e2e/ # Playwright E2E tests
+│ ├── security/ # Security tests
+│ ├── translator/ # Translator-specific tests
+│ └── load/ # Load tests
+├── docs/ # Documentation
+│ ├── i18n/ # 30-language translated docs
+│ ├── ARCHITECTURE.md # Full architecture documentation
+│ ├── API_REFERENCE.md # API reference
+│ ├── USER_GUIDE.md # User guide
+│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview
+│ ├── CLI-TOOLS.md # CLI tools integration guide
+│ ├── A2A-SERVER.md # A2A agent protocol documentation
+│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring)
+│ ├── MCP-SERVER.md # MCP server (25 tools)
+│ ├── TROUBLESHOOTING.md # Troubleshooting guide
+│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide
+│ ├── openapi.yaml # OpenAPI specification
+│ └── screenshots/ # Dashboard screenshots
+├── bin/ # CLI entry points (omniroute, reset-password)
+├── scripts/ # Build and utility scripts
+└── .env.example # Environment variable template
+```
+
+## Key Features (v3.5.5)
+
+### Core Proxy
+- **60+ AI providers** with automatic format translation
+- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible)
+- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay
+- **4-tier fallback**: Subscription → API Key → Cheap → Free
+- **Context Relay strategy**: Session handoff summaries on account rotation for continuity
+- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown
+- **Semantic caching** with cache hit/miss headers
+- **Idempotency** with configurable dedup window
+- **Circuit breaker** per provider with configurable thresholds
+- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback
+- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers
+- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement
+- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization
+- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills
+- **Prompt Injection Guard**: Middleware-level prompt injection detection
+- **MITM Proxy**: Certificate management, DNS handling, and target routing
+- **Cloudflare Tunnels**: Managed tunnel creation for remote access
+- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches)
+
+### Seguridad
+- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs)
+- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()`
+- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()`
+- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection
+- **CLI Fingerprint Matching** — Per-provider request signature matching
+- **Prompt injection guard** — Request middleware detection
+- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`)
+- **PII sanitizer** — Sensitive data scrubbing in logs
+
+### Dashboard Pages (23 sections)
+- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons
+- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies
+- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics
+- **Analytics** — Token consumption, cost, heatmaps, distributions
+- **Health** — Uptime, memory, latency percentiles, circuit breakers
+- **Logs** — Request, Proxy, Audit, Console (tabbed)
+- **Audit** — Audit trail and compliance logging
+- **Costs** — Cost tracking per provider/model
+- **Limits** — Rate limit monitoring
+- **Cache** — Semantic cache statistics and management
+- **CLI Tools** — One-click configuration for 10+ AI CLI tools
+- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration
+- **Playground** — Test any model with Monaco editor, streaming responses
+- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files)
+- **Search Tools** — Search provider configuration and testing
+- **Memory** — Memory system management and visualization
+- **Skills** — Skills framework management and execution
+- **Translator** — Format debugging: playground, chat tester, test bench, live monitor
+- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced
+- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed)
+- **Onboarding** — Setup wizard for new users
+- **Usage** — Usage history and analytics
+- **API Manager** — API key management with scoped permissions
+
+### Protocol Support
+- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations`
+- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens`
+- **OpenAI Responses** — `/v1/responses`
+- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}`
+- **Ollama** — `/v1/api/chat`, `/api/tags`
+- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily)
+- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP)
+- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills)
+- **ACP** — Agent Communication Protocol registry and manager
+
+### MCP Server (25 Tools)
+| Category | Tools |
+|-----------|-------|
+| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` |
+| Memory (3) | `memory_search`, `memory_add`, `memory_clear` |
+| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` |
+
+**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience`
+
+### Provider Categories
+
+**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI
+
+**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline
+
+**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan
+
+**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs
+
+### Internationalization
+- 30 languages for UI (all dashboard pages)
+- 30 translated documentation sets in docs/i18n/
+- Language switcher in documentation
+
+## Key Architectural Decisions
+
+1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints.
+
+2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently.
+
+3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation.
+
+4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity.
+
+5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client.
+
+6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes.
+
+7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`).
+
+8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages.
+
+9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations.
+
+10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`.
+
+11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation.
+
+12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline.
+
+13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast.
+
+## Main Flows
+
+### Proxy Request Flow
+1. Client sends OpenAI-format request to `/v1/chat/completions`
+2. API key validation
+3. Model resolution: direct model or combo lookup
+4. For combos: iterate through models with selected strategy
+5. Auth resolution: get credentials for the target provider
+6. Format translation: OpenAI → provider native format
+7. CLI fingerprint matching (if enabled for provider)
+8. Upstream request with circuit breaker and rate limiting
+9. Response translation: provider → OpenAI format
+10. omniModel tag sanitization (strip internal tags)
+11. SSE streaming back to client
+12. Memory extraction (if memory system enabled)
+13. Usage logging and cost calculation
+
+### OAuth Flow
+1. Dashboard initiates `/api/oauth/[provider]/authorize`
+2. User completes OAuth login in browser
+3. Callback hits `/api/oauth/[provider]/exchange`
+4. Tokens stored as a provider connection in SQLite
+5. Background job refreshes tokens before expiry
+
+## Important Notes for LLMs
+
+1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only).
+
+2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`).
+
+3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services.
+
+4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`.
+
+5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module.
+
+6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`).
+
+7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes.
+
+8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB.
+
+9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown.
+
+10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130).
+
+11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux).
+
+12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint.
+
+13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API.
+
+14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API.
+
+15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time.
+
+16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`.
+
+17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22.
+
+18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false.
+
+## Links
+
+- Repository: https://github.com/diegosouzapw/OmniRoute
+- Website: https://omniroute.online
+- npm: https://www.npmjs.com/package/omniroute
+- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute
+- Documentation: See `/docs/` directory
diff --git a/docs/i18n/fi/README.md b/docs/i18n/fi/README.md
index a2da873826..411a4e1a61 100644
--- a/docs/i18n/fi/README.md
+++ b/docs/i18n/fi/README.md
@@ -4,11 +4,14 @@
---
+
### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback.
-_Universaali API-välityspalvelin – yksi päätepiste, yli 60 palveluntarjoajaa, nolla seisokkiaikaa. Nyt mukana**MCP-palvelin (25 työkalua)**,**A2A-protokolla**,**muisti/taitojärjestelmät**ja**electron-työpöytäsovellus**._
+_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._
-**Pikaviestien loppuun saattaminen • upotukset • kuvien luonti • video • musiikki • ääni • uudelleensijoitus •**verkkohaku**• MCP-palvelin • A2A-protokolla • 100 % TypeScript**---
+**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript**
+
+---
@@ -39,9 +42,13 @@ _Universaali API-välityspalvelin – yksi päätepiste, yli 60 palveluntarjoaja
[](https://omniroute.online)
[](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-[🌐 Verkkosivusto](https://omniroute.online) • [🚀 Pika-aloitus](#-pika-aloitus) • [💡 Ominaisuudet](#-avainominaisuudet) • [📖 Docs](#-dokumentaatio) • [💰 Hinnoittelu](#-hinnoittelu-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-🌐**Saatavilla:**🇺🇸 [Englanti](README.md) | 🇧🇷 [Português (Brasilia)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Tanska](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugali)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)---
+
+
+🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)
+
+---
## 🖼️ Main Dashboard
@@ -53,553 +60,629 @@ _Universaali API-välityspalvelin – yksi päätepiste, yli 60 palveluntarjoaja
## 📸 Dashboard Preview
-
-Napsauta nähdäksesi hallintapaneelin kuvakaappaukset
+
+Click to see dashboard screenshots
-| Sivu | Kuvakaappaus |
-| ---------------- | -------------------------------------------------- | ---------- |
-| **Tarjoajat** |  |
-| **Yhdistelmät** |  |
-| **Analytics** |  |
-| **Terveys** |  |
-| **Kääntäjä** |  |
-| **Asetukset** |  |
-| **CLI-työkalut** |  |
-| **Käyttölokit** |  |
-| **Päätepisteet** |  | |
+| Page | Screenshot |
+| -------------- | ------------------------------------------------- |
+| **Providers** |  |
+| **Combos** |  |
+| **Analytics** |  |
+| **Health** |  |
+| **Translator** |  |
+| **Settings** |  |
+| **CLI Tools** |  |
+| **Usage Logs** |  |
+| **Endpoints** |  |
+
+
---
### 🤖 Free AI Provider for your favorite coding agents
-_Yhdistä mikä tahansa tekoälyllä toimiva IDE- tai CLI-työkalu OmniRouten kautta – ilmainen API-yhdyskäytävä rajoittamattomaan koodaukseen._
+_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._
-
+
-📡 Kaikki agentit muodostavat yhteyden http://localhost:20128/v1 tai http://cloud.omniroute.online/v1 kautta – yksi kokoonpano, rajattomasti malleja ja kiintiö---
+📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota
+
+---
## 🤔 Why OmniRoute?
-**Lopeta rahan tuhlaaminen ja rajojen ylittäminen:**
+**Stop wasting money and hitting limits:**
--
Tilauskiintiö vanhenee käyttämättä joka kuukausi
--
Hintarajoitukset estävät sinua koodaamasta puolivälissä
--
kalliita sovellusliittymiä (20-50 $/kk per tarjoaja)
--
Manuaalinen vaihtaminen palveluntarjoajien välillä
+-
Subscription quota expires unused every month
+-
Rate limits stop you mid-coding
+-
Expensive APIs ($20-50/month per provider)
+-
Manual switching between providers
-**OmniRoute ratkaisee tämän:**
+**OmniRoute solves this:**
-- ✅**Maksimoi tilaukset**- Seuraa kiintiötä, käytä jokainen bitti ennen nollausta
-- ✅**Automaattinen palautus**- Tilaus → API-avain → Halpa → Ilmainen, nolla seisonta-aikaa
-- ✅**Moni tili**- Pyöreä haku tilien välillä per palveluntarjoaja
-- ✅**Universaali**- Toimii Claude Coden, Codexin, Gemini CLI:n, Cursorin, Clinen, OpenClawin ja minkä tahansa CLI-työkalun kanssa---
+- ✅ **Maximize subscriptions** - Track quota, use every bit before reset
+- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime
+- ✅ **Multi-account** - Round-robin between accounts per provider
+- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool
+
+---
## 📧 Support
-> 💬**Liity yhteisöömme!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Hanki apua, jaa vinkkejä ja pysy ajan tasalla.
+> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated.
--**Verkkosivusto**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Ongelmat**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [yhteisöryhmä](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Osallistuminen**: Katso [CONTRIBUTING.md](CONTRIBUTING.md), avaa PR tai valitse "hyvä ensimmäinen numero". -**Alkuperäinen projekti**: [9router by decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug?
+- **Website**: [omniroute.online](https://omniroute.online)
+- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute)
+- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue`
+- **Original Project**: [9router by decolua](https://github.com/decolua/9router)
-Kun avaat ongelman, suorita system-info-komento ja liitä luotu tiedosto:```bash
+### 🐛 Reporting a Bug?
+
+When opening an issue, please run the system-info command and attach the generated file:
+
+```bash
npm run system-info
-
```
-Tämä luo `system-info.txt-tiedoston, joka sisältää Node.js-versiosi, OmniRoute-versiosi, käyttöjärjestelmän tiedot, asennetut CLI-työkalut (qoder, gemini, claude, codex, antigravity, droidi jne.), Docker/PM2-tilan ja järjestelmäpaketit – kaikki mitä tarvitsemme ongelmasi nopeaan toistamiseen. Liitä tiedosto suoraan GitHub-ongelmaasi.---
+This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue.
+
+---
## 🔄 How It Works
```
-
┌─────────────┐
-│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
-│ Tool │
+│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
+│ Tool │
└──────┬──────┘
-│ http://localhost:20128/v1
-↓
+ │ http://localhost:20128/v1
+ ↓
┌─────────────────────────────────────────┐
-│ OmniRoute (Smart Router) │
-│ • Format translation (OpenAI ↔ Claude) │
-│ • Quota tracking + Embeddings + Images │
-│ • Auto token refresh │
+│ OmniRoute (Smart Router) │
+│ • Format translation (OpenAI ↔ Claude) │
+│ • Quota tracking + Embeddings + Images │
+│ • Auto token refresh │
└──────┬──────────────────────────────────┘
-│
-├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
-│ ↓ quota exhausted
-├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
-│ ↓ budget limit
-├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
-│ ↓ budget limit
-└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
+ │
+ ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
+ │ ↓ quota exhausted
+ ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
+ │ ↓ budget limit
+ ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
+ │ ↓ budget limit
+ └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
Result: Never stop coding, minimal cost
-
-````
+```
---
## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases
->**Jokainen tekoälytyökaluja käyttävä kehittäjä kohtaa nämä ongelmat päivittäin.**OmniRoute luotiin ratkaisemaan ne kaikki – kustannusten ylityksistä alueellisiin lohkoihin, rikkinäisistä OAuth-virroista protokollatoimintoihin ja yrityksen havainnointikykyyn.
+> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability.
-
-💸 1. "Maksan kalliista tilauksesta, mutta silti rajoitukset häiritsevät minua"
+
+💸 1. "I pay for an expensive subscription but still get interrupted by limits"
-Kehittäjät maksavat 20–200 dollaria kuukaudessa Claude Prosta, Codex Prosta tai GitHub Copilotista. Maksamallakin kiintiöllä on katto – 5 tuntia käyttöä, viikkorajat tai minuuttirajoitukset. Koodausistunnon puolivälissä palveluntarjoaja lakkaa vastaamasta ja kehittäjä menettää virtauksen ja tuottavuuden.
+Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity.
-**Kuinka OmniRoute ratkaisee sen:**
+**How OmniRoute solves it:**
--**Smart 4-Tier Fallback**— Jos tilauskiintiö loppuu, ohjataan automaattisesti kohtaan API-avain → Halpa → Ilmainen ilman manuaalista toimenpiteitä
--**Provider Limits Tracking**– Välimuistissa olevat kiintiön tilannevedokset päivittyvät palvelinpuolen aikataulun mukaan (oletus `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`), ja manuaalinen päivitys on saatavilla käyttöliittymässä
--**Useiden tilien tuki**— Useita tilejä palveluntarjoajaa kohden automaattisella kierrätyksellä — kun yksi loppuu, vaihtuu seuraavaan
--**Muokatut yhdistelmät**— Muokattavat varaketjut, joissa on 9 tasapainotusstrategiaa (prioriteetti, painotettu, täytä ensin, round-robin, P2C, satunnainen, vähiten käytetty, kustannusoptimoitu, tiukasti satunnainen)
--**Codex Business Quotat**— Yritysten/Tiimien työtilan kiintiöiden valvonta suoraan kojelaudassa
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention
+- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI
+- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next
+- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**)
+- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard
-
-🔌 2. "Minun täytyy käyttää useita palveluntarjoajia, mutta jokaisella on eri sovellusliittymä"
+
-OpenAI käyttää yhtä muotoa, Claude (Anthropic) käyttää toista, Gemini vielä toista. Jos kehittäjä haluaa testata eri palveluntarjoajien malleja tai vaihtoehtoja niiden välillä, hänen on määritettävä SDK:t uudelleen, muutettava päätepisteitä ja käsiteltävä yhteensopimattomia muotoja. Mukautetuilla palveluntarjoajilla (FriendLI, NIM) on mallista poikkeavat päätepisteet.
+
+🔌 2. "I need to use multiple providers but each has a different API"
-**Kuinka OmniRoute ratkaisee sen:**
+OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints.
--**Unified Endpoint**— Yksi "http://localhost:20128/v1" toimii välityspalvelimena kaikille yli 60 palveluntarjoajalle
--**Format Translation**- Automaattinen ja läpinäkyvä: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
--**Response Sanitization**– Poistaa standardista poikkeavat kentät (`x_groq`, `usage_breakdown`, `service_tier`), jotka rikkovat OpenAI SDK v1.83+:n
--**Roolin normalisointi**— Muuntaa "kehittäjä" → "järjestelmä" muille kuin OpenAI-palveluntarjoajille; "järjestelmä" → "käyttäjä" GLM:lle/ERNIE:lle
--**Think Tag Extraction**– Purkaa ""-lohkot malleista, kuten DeepSeek R1, standardoituun "reasoning_content" -sisältöön
--**Structured Output for Gemini**— `json_schema` → `responseMimeType`/`responseSchema` automaattinen muunnos
--**"stream" oletusarvo on "false"**- yhdenmukaistuu OpenAI-spesifikaatioiden kanssa välttäen odottamattoman SSE:n Python/Rust/Go SDK:issa
+**How OmniRoute solves it:**
-
-🌐 3. "Tekoälypalveluntarjoajani estää alueeni/maani"
+- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers
+- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
+- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+
+- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE
+- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content`
+- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion
+- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs
-Palveluntarjoajat, kuten OpenAI/Codex, estävät pääsyn tietyiltä maantieteellisiltä alueilta. Käyttäjät saavat virheitä, kuten "unsupported_country_region_territory" OAuth- ja API-yhteyksien aikana. Tämä on erityisen turhauttavaa kehitysmaiden kehittäjille.
+
-**Kuinka OmniRoute ratkaisee sen:**
+
+🌐 3. "My AI provider blocks my region/country"
--**3-tason välityspalvelimen määritys**– Muokattava välityspalvelin kolmella tasolla: yleinen (kaikki liikenne), palveluntarjoajakohtainen (vain yksi palveluntarjoaja) ja yhteys/avain
--**Värikoodatut välityspalvelinmerkit**— Visuaaliset ilmaisimet: 🟢 maailmanlaajuinen välityspalvelin, 🟡 tarjoajan välityspalvelin, 🔵 yhteysvälityspalvelin, joka näyttää aina IP-osoitteen
--**OAuth-tunnusten vaihto välityspalvelimen kautta**— OAuth-kulku kulkee myös välityspalvelimen kautta ja ratkaisee "unsupported_country_region_territory"
--**Yhteystestit välityspalvelimen kautta**- Yhteystestit käyttävät määritettyä välityspalvelinta (ei enää suoraa ohitusta)
--**SOCKS5-tuki**— Täysi SOCKS5-välityspalvelintuki lähtevään reititykseen
--**TLS-sormenjälkien huijaus**— Selaimen kaltainen TLS-sormenjälki wreq-js:n kautta botin tunnistuksen ohittamiseksi
--**🔏 CLI Fingerprint Matching**– Järjestää otsikot ja tekstikentät uudelleen vastaamaan alkuperäisiä CLI-binääriallekirjoituksia, mikä vähentää merkittävästi tilin ilmoittamisriskiä. Välityspalvelimen IP-osoite säilyy – saat sekä salaperäisen**- että**IP-peitetyksen samanaikaisesti
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries.
-
-🆓 4. "Haluan käyttää tekoälyä koodaukseen, mutta minulla ei ole rahaa"
+**How OmniRoute solves it:**
-Kaikki eivät voi maksaa 20–200 dollaria kuukaudessa tekoälytilauksista. Opiskelijat, kehittäjät nousevista maista, harrastajat ja freelancerit tarvitsevat pääsyn laadukkaisiin malleihin ilman kustannuksia.
+- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key
+- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP
+- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory`
+- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass)
+- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection
+- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously
-**Kuinka OmniRoute ratkaisee sen:**
+
--**Free Tier Providers -sisäänrakennettu**— Natiivituki 100 % ilmaisille palveluntarjoajille: Qoder (5 rajoittamatonta mallia OAuthin kautta: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited3 mallia) qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID ilmaiseksi), Gemini CLI (180 000 tokenia / kuukausi ilmaiseksi)
--**Ollama Cloud**— pilvissä isännöimät Ollama-mallit osoitteessa `api.ollama.com` ilmaisella "kevytkäyttö"-tasolla; käytä `ollamacloud/-etuliitettä
--**Vain ilmaiset yhdistelmät**— Ketju `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 $/kk ilman seisokkeja
--**NVIDIA NIM Free Access**- ~40 RPM:n kehittäjä - ikuisesti ilmainen pääsy yli 70 malliin osoitteessa build.nvidia.com (siirrytään hyvityksistä puhtaisiin hintarajoihin)
--**Kustannusoptimoitu strategia**— Reititysstrategia, joka valitsee automaattisesti halvimman saatavilla olevan palveluntarjoajan
+
+🆓 4. "I want to use AI for coding but I have no money"
-
-🔒 5. "Minun täytyy suojata tekoälyyhdyskäytävääni luvattomalta käytöltä"
+Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost.
-Kun paljastat tekoälyyhdyskäytävän verkkoon (LAN, VPS, Docker), kuka tahansa osoitteen tietävä voi kuluttaa kehittäjän tunnukset/kiintiöt. Ilman suojaa API:t ovat alttiita väärinkäytölle, nopealle injektiolle ja väärinkäytöksille.
+**How OmniRoute solves it:**
-**Kuinka OmniRoute ratkaisee sen:**
+- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free)
+- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix
+- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime
+- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits)
+- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider
--**API-avainten hallinta**– Luominen, kierto ja laajuus palveluntarjoajan mukaan erillisellä /dashboard/api-manager-sivulla
--**Mallitason käyttöoikeudet**– Rajoita API-avaimet tiettyihin malleihin ('openai/*', jokerimerkkimallit) Salli kaikki/Rajoita -kytkimellä
--**Sovellusliittymän päätepistesuojaus**– vaadi avainta /v1/modelsille ja estä tietyt palveluntarjoajat luettelosta
--**Auth Guard + CSRF-suojaus**— Kaikki kojelaudan reitit on suojattu "withAuth"-väliohjelmistolla + CSRF-tunnuksilla
--**Rate Limiter**— IP-nopeuden rajoitus konfiguroitavilla ikkunoilla
--**IP-suodatus**— Pääsynhallinnan sallittu-/estolista
--**Prompt Injection Guard**— Desinfiointi haitallisia kehotusmalleja vastaan
--**AES-256-GCM Encryption**— Tunnistetiedot on salattu lepotilassa
+
-
-🛑 6. "Palvelajani kaatui ja menetin koodauskulkuni"
+
+🔒 5. "I need to protect my AI gateway from unauthorized access"
-Tekoälypalveluntarjoajat voivat muuttua epävakaiksi, palauttaa 5xx-virheitä tai saavuttaa väliaikaiset nopeusrajoitukset. Jos kehittäjä on riippuvainen yhdestä palveluntarjoajasta, se keskeytyy. Ilman katkaisijoita toistuvat uudelleenyritykset voivat kaataa sovelluksen.
+When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse.
-**Kuinka OmniRoute ratkaisee sen:**
+**How OmniRoute solves it:**
--**Malleittainen katkaisija**- Automaattinen avautuminen/sulkeminen konfiguroitavilla kynnyksillä ja jäähdytys (suljettu/auki/puoliauki), mallikohtainen, jotta vältetään peräkkäiset lohkot
--**Eksponentiaalinen peruutus**— Progressiiviset uudelleenyritysviiveet
--**Anti-Thundering Herd**— Mutex + semaforisuoja samanaikaisia myrskyjä vastaan
--**Yhdistelmävaraketjut**– Jos ensisijainen toimittaja epäonnistuu, putoaa automaattisesti ketjun läpi ilman väliintuloa
--**Combo Circuit Breaker**— Poistaa automaattisesti käytöstä vialliset palveluntarjoajat yhdistelmäketjussa
--**Health Dashboard**— käytettävyyden valvonta, katkaisijoiden tilat, lukitukset, välimuistitilastot, p50/p95/p99-viive
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page
+- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle
+- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing
+- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens
+- **Rate Limiter** — Per-IP rate limiting with configurable windows
+- **IP Filtering** — Allowlist/blocklist for access control
+- **Prompt Injection Guard** — Sanitization against malicious prompt patterns
+- **AES-256-GCM Encryption** — Credentials encrypted at rest
-
-🔧 7. "Jokaisen tekoälytyökalun määrittäminen on työlästä ja toistuvaa"
+
-Kehittäjät käyttävät kursoria, Claude Codea, Codex CLI:tä, OpenClaw:ta, Gemini CLI:tä, Kilo Codea... Jokainen työkalu tarvitsee eri konfiguraation (API-päätepiste, avain, malli). Uudelleenmääritys toimittajaa tai mallia vaihdettaessa on ajanhukkaa.
+
+🛑 6. "My provider went down and I lost my coding flow"
-**Kuinka OmniRoute ratkaisee sen:**
+AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application.
--**CLI Tools Dashboard**- Erillinen sivu yhdellä napsautuksella Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
--**GitHub Copilot Config Generator**— Luo chatLanguageModels.json-tiedoston VS-koodille joukkomallin valinnalla
--**Ohjattu käyttöönottotoiminto**— Ohjattu 4-vaiheinen asennus ensikertalaisille
--**Yksi päätepiste, kaikki mallit**- Määritä "http://localhost:20128/v1" kerran, käytä yli 60 palveluntarjoajaa
+**How OmniRoute solves it:**
-
-🔑 8. "Useiden palveluntarjoajien OAuth-tunnusten hallinta on helvettiä"
+- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks
+- **Exponential Backoff** — Progressive retry delays
+- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms
+- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention
+- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain
+- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency
-Claude Code, Codex, Gemini CLI, Copilot – kaikki käyttävät OAuth 2.0:aa vanhentuvilla tunnuksilla. Kehittäjien täytyy todentaa jatkuvasti uudelleen, käsitellä "asiakassalaisuus puuttuu", "redirect_uri_mismatch" ja etäpalvelimien vikoja. OAuth LAN/VPS:ssä on erityisen ongelmallinen.
+
-**Kuinka OmniRoute ratkaisee sen:**
+
+🔧 7. "Configuring each AI tool is tedious and repetitive"
--**Automaattinen tunnuksen päivitys**- OAuth-tunnukset päivittyvät taustalla ennen vanhenemista
--**Sisäänrakennettu OAuth 2.0 (PKCE)**- Automaattinen kulku Claude Codelle, Codexille, Gemini CLI:lle, Copilotille, Kirolle, Qwenille, Qoderille
--**Multi-Account OAuth**- Useita tilejä palveluntarjoajaa kohden JWT/ID-tunnuksen purkamisen kautta
--**OAuth LAN/Remote Fix**- Yksityinen IP-tunnistus `redirect_uri':lle + manuaalinen URL-tila etäpalvelimille
--**OAuth Nginxin takana**- Käyttää "window.location.origin" käänteisen välityspalvelimen yhteensopivuutta varten
--**OAuth-etäopas**— Vaiheittainen opas Google Cloud -kirjautumistiedoille VPS/Dockerissa
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time.
-
-📊 9. "En tiedä kuinka paljon kulutan tai minne"
+**How OmniRoute solves it:**
-Kehittäjät käyttävät useita maksullisia palveluntarjoajia, mutta heillä ei ole yhtenäistä näkemystä kuluttamisesta. Jokaisella palveluntarjoajalla on oma laskutuksen hallintapaneeli, mutta yhdistettyä näkymää ei ole. Odottamattomat kustannukset voivat kasaantua.
+- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
+- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection
+- **Onboarding Wizard** — Guided 4-step setup for first-time users
+- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers
-**Kuinka OmniRoute ratkaisee sen:**
+
--**Cost Analytics Dashboard**– Token-kohtainen kustannusseuranta ja budjetin hallinta palveluntarjoajakohtaisesti
--**Tasokohtaiset budjettirajat**– Tasokohtainen kulutuskatto, joka laukaisee automaattisen varauksen
--**Malleittainen hinnoittelu**— Muokattavat hinnat mallikohtaisesti
--**Käyttötilastot API-avainta kohti**— Pyyntömäärä ja viimeksi käytetty aikaleima avainta kohti
--**Analytics Dashboard**- Tilastokortit, mallin käyttökaavio, toimittajataulukko onnistumisprosenteilla ja viiveellä
+
+🔑 8. "Managing OAuth tokens from multiple providers is hell"
-
-🐛 10. "En pysty diagnosoimaan tekoälypuhelujen virheitä ja ongelmia"
+Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic.
-Kun puhelu epäonnistuu, kehittäjä ei tiedä, oliko kyseessä nopeusrajoitus, vanhentunut tunnus, väärä muoto vai palveluntarjoajan virhe. Sirpaloituneet lokit eri terminaaleissa. Ilman havaittavuutta virheenkorjaus on yrityksen ja erehdysten menetelmää.
+**How OmniRoute solves it:**
-**Kuinka OmniRoute ratkaisee sen:**
+- **Auto Token Refresh** — OAuth tokens refresh in background before expiration
+- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
+- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction
+- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers
+- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility
+- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker
--**Yhdistettyjen lokien hallintapaneeli**- 4 välilehteä: pyyntölokit, välityspalvelimen lokit, tarkastuslokit, konsoli
--**Console Log Viewer**- Reaaliaikainen päätetyylinen katseluohjelma värikoodatuilla tasoilla, automaattinen vieritys, haku, suodatin
--**SQLite-välityspalvelimen lokit**— Pysyvät lokit, jotka kestävät palvelimen uudelleenkäynnistyksen
--**Kääntäjän leikkikenttä**— 4 virheenkorjaustilaa: Playground (muodon käännös), Chat Tester (meno-paluu), testipenkki (erä), Live Monitor (reaaliaikainen)
--**Pyyntötelemetria**— p50/p95/p99-latenssi + X-Request-Id-seuranta
--**Tiedostopohjainen kirjaaminen rotaatiolla**– Sovelluslokit pyörivät koon, säilytyspäivien ja arkiston määrän mukaan; puhelulokin artefaktit kiertävät säilytyspäivien ja tiedostojen määrän mukaan
--**Järjestelmätietoraportti**— `npm run system-info` luo `system-info.txt-tiedoston koko ympäristössäsi (solmuversio, OmniRoute-versio, käyttöjärjestelmä, CLI-työkalut, Docker/PM2-tila). Liitä se, kun ilmoitat ongelmista välittömässä triagessa.
+
-
-🏗️ 11. "Yhdyskäytävän käyttöönotto ja ylläpito on monimutkaista"
+
+📊 9. "I don't know how much I'm spending or where"
-AI-välityspalvelimen asentaminen, määrittäminen ja ylläpito eri ympäristöissä (paikallinen, VPS, Docker, pilvi) on työvoimavaltaista. Ongelmat, kuten kovakoodatut polut, 'EACCES' hakemistoissa, porttiristiriidat ja cross-platform buildit lisäävät kitkaa.
+Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up.
-**Kuinka OmniRoute ratkaisee sen:**
+**How OmniRoute solves it:**
--**npm globaali asennus**— `npm install -g omniroute && omniroute` — valmis
--**Docker Multi-Platform**- AMD64 + ARM64 natiivi (Apple Silicon, AWS Graviton, Raspberry Pi)
--**Docker Compose -profiilit**— "base" (ei CLI-työkaluja) ja "cli" (Claude Coden, Codexin, OpenClawn kanssa)
--**Electron Desktop App**- Natiivisovellus Windowsille/macOS:lle/Linuxille, jossa ilmaisinalue, automaattinen käynnistys, offline-tila
--**Split-Port Mode**— API ja Dashboard erillisissä porteissa edistyneille skenaarioille (käänteinen välityspalvelin, konttiverkko)
--**Cloud Sync**- Määritä synkronointi laitteiden välillä Cloudflare Workersin kautta
--**DB-varmuuskopiot**— Kaikkien asetusten automaattinen varmuuskopiointi, palautus, vienti ja tuonti DISABLE_SQLITE_AUTO_BACKUP-toiminnolla ulkoisesti hallituille varmuuskopioille
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider
+- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback
+- **Per-Model Pricing Configuration** — Configurable prices per model
+- **Usage Statistics Per API Key** — Request count and last-used timestamp per key
+- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency
-
-🌍 12. "Käyttöliittymä on vain englanninkielinen ja tiimini ei puhu englantia"
+
-Ryhmät muissa kuin englanninkielisissä maissa, erityisesti Latinalaisessa Amerikassa, Aasiassa ja Euroopassa, kamppailevat vain englanninkielisten käyttöliittymien kanssa. Kielimuurit vähentävät käyttöönottoa ja lisäävät konfigurointivirheitä.
+
+🐛 10. "I can't diagnose errors and problems in AI calls"
-**Kuinka OmniRoute ratkaisee sen:**
+When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error.
--**Dashboard i18n — 30 kieltä**— Kaikki yli 500 näppäintä käännetty mukaan lukien arabia, bulgaria, tanska, saksa, espanja, suomi, ranska, heprea, hindi, unkari, indonesia, italia, japani, korea, malaiji, hollanti, norja, puola, portugali (PT/BR), romania, thai, venäjä, ukraina, slovakki, ruotsi, englanti
--**RTL-tuki**— Tuki oikealta vasemmalle arabian ja heprean kielelle
--**Multi-Language READMEs**- 30 täydellistä dokumentaation käännöstä
--**Kielen valitsin**— Maapallokuvake otsikossa reaaliaikaista vaihtoa varten
+**How OmniRoute solves it:**
-
-🔄 13. "Tarvitsen muutakin kuin chatin – tarvitsen upotuksia, kuvia, ääntä"
+- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console
+- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter
+- **SQLite Proxy Logs** — Persistent logs that survive server restarts
+- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time)
+- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing
+- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count
+- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage.
-Tekoäly ei ole vain chatin loppuun saattamista. Kehittäjien on luotava kuvia, litteroitava ääni, luotava upotuksia RAG:lle, järjestettävä asiakirjat uudelleen ja valvottava sisältöä. Jokaisella API:lla on eri päätepiste ja muoto.
+
-**Kuinka OmniRoute ratkaisee sen:**
+
+🏗️ 11. "Deploying and maintaining the gateway is complex"
--**Upotukset**— `/v1/embeddings` kuudella palveluntarjoajalla ja 9+ mallilla
--**Image Generation**— `/v1/images/generations` 10 tarjoajalla ja 20+ mallilla (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
--**Tekstistä videoon**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) ja SD WebUI
--**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
--**Äänitranskriptio**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
--**Tekstistä puheeksi**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**ja olemassa olevat palveluntarjoajat
--**Moderations**— `/v1/moderations` — Sisällön turvallisuustarkastukset
--**Uudelleensijoitus**— `/v1/rerank` — Asiakirjan relevanssin uudelleensijoitus
--**Responses API**- Täysi `/v1/responses` -tuki Codexille
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction.
-
-🧪 14. "Minulla ei ole mahdollisuutta testata ja vertailla eri mallien laatua"
+**How OmniRoute solves it:**
-Kehittäjät haluavat tietää, mikä malli sopii parhaiten heidän käyttötapaukseensa – koodi, käännös, päättely – mutta manuaalinen vertailu on hidasta. Integroituja arviointityökaluja ei ole olemassa.
+- **npm global install** — `npm install -g omniroute && omniroute` — done
+- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi)
+- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw)
+- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode
+- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking)
+- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers
+- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups
-**Kuinka OmniRoute ratkaisee sen:**
+
--**LLM-arvioinnit**— Golden set -testaus 10 esiladatulla kotelolla, jotka kattavat tervehdyksen, matematiikan, maantieteen, koodin luomisen, JSON-yhteensopivuuden, käännöksen, merkinnän, turvallisuuden kieltämisen
--**4 vastaavuusstrategiaa**— "tarkka", "sisältää", "säännöllinen lauseke", "muokattu" (JS-funktio)
--**Translator Playground Test Bench**- Erätestaus useilla tuloilla ja odotetulla lähdöllä, tarjoajien välinen vertailu
--**Chat Tester**- Täysi edestakainen matka visuaalisen vasteen renderöinnillä
--**Live Monitor**— Reaaliaikainen tietovirta kaikista välityspalvelimen kautta kulkevista pyynnöistä
+
+🌍 12. "The interface is English-only and my team doesn't speak English"
-
-📈 15. "Minun täytyy skaalata suorituskykyä menettämättä"
+Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors.
-Pyynnön määrän kasvaessa samat kysymykset aiheuttavat päällekkäisiä kustannuksia välimuistiin tallentamatta. Ilman idempotenssia kaksoiskappaleet pyytävät jätteenkäsittelyä. Palveluntarjoajakohtaisia hintarajoja on noudatettava.
+**How OmniRoute solves it:**
-**Kuinka OmniRoute ratkaisee sen:**
+- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English
+- **RTL Support** — Right-to-left support for Arabic and Hebrew
+- **Multi-Language READMEs** — 30 complete documentation translations
+- **Language Selector** — Globe icon in header for real-time switching
--**Semanttinen välimuisti**– Kaksitasoinen välimuisti (allekirjoitus + semanttinen) vähentää kustannuksia ja viivettä
--**Request Idempotency**— 5 sekunnin deduplikaatioikkuna identtisille pyynnöille
--**nopeusrajoituksen tunnistus**– palveluntarjoajakohtainen RPM, pienin väli ja suurin samanaikainen seuranta
--**Muokattavat nopeusrajoitukset**- Määritettävissä olevat oletusasetukset kohdassa Asetukset → Resilience with persistence
--**API Key Validation Cache**– 3-tasoinen välimuisti tuotannon suorituskykyä varten
--**Health Dashboard telemetrialla**- p50/p95/p99 latenssi, välimuistitilastot, käyttöaika
+
-
-🤖 16. "Haluan hallita mallien käyttäytymistä maailmanlaajuisesti"
+
+🔄 13. "I need more than chat — I need embeddings, images, audio"
-Kehittäjät, jotka haluavat kaikki vastaukset tietyllä kielellä, tietyllä sävyllä tai haluavat rajoittaa perusteluita. Tämän määrittäminen jokaiseen työkaluun/pyyntöön on epäkäytännöllistä.
+AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format.
-**Kuinka OmniRoute ratkaisee sen:**
+**How OmniRoute solves it:**
--**Järjestelmäkehotteen lisäys**— Yleinen kehote koskee kaikkia pyyntöjä
--**Thinking Budget Validation**– perustelutunnisteen allokoinnin ohjaus pyyntöä kohti (läpivienti, automaattinen, mukautettu, mukautuva)
--**9 reititysstrategiaa**— Globaalit strategiat, jotka määrittävät pyyntöjen jakautumisen
--**Wildcard Router**- "palveluntarjoaja/*" -mallit reitittävät dynaamisesti mille tahansa palveluntarjoajalle
--**Yhdistelmä käyttöön/pois käytöstä**- Vaihda yhdistelmät suoraan kojelaudalta
--**Provider Toggle**— Ota käyttöön tai poista käytöstä kaikki palveluntarjoajan yhteydet yhdellä napsautuksella
--**Estetyt palveluntarjoajat**- Sulje pois tietyt palveluntarjoajat /v1/models-luettelosta
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models
+- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
+- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI
+- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
+- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
+- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers
+- **Moderations** — `/v1/moderations` — Content safety checks
+- **Reranking** — `/v1/rerank` — Document relevance reranking
+- **Responses API** — Full `/v1/responses` support for Codex
-
-🧰 17. "Tarvitsen MCP-työkaluja ensiluokkaisina tuoteominaisuuksina"
+
-Monet tekoälyyhdyskäytävät paljastavat MCP:n vain piilotettuna toteutustietona. Tiimit tarvitsevat näkyvän, hallittavan toimintakerroksen.
+
+🧪 14. "I have no way to test and compare quality across models"
-**Kuinka OmniRoute ratkaisee sen:**
+Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist.
-- MCP näkyy kojelaudan navigointi- ja päätepisteprotokolla-välilehdessä
-- Erillinen MCP-hallintasivu, jossa on prosessit, työkalut, laajuudet ja tarkastus
-- Sisäänrakennettu pikakäynnistys omniroute --mcp:lle ja asiakkaan käyttöönottoon
+**How OmniRoute solves it:**
-
-🧠 18. "Tarvitsen A2A-orkesterin synkronointi- ja stream-tehtäväpoluilla"
+- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal
+- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function)
+- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison
+- **Chat Tester** — Full round-trip with visual response rendering
+- **Live Monitor** — Real-time stream of all requests flowing through the proxy
-Agenttityönkulut tarvitsevat sekä suoria vastauksia että pitkäkestoista suoratoistoa elinkaariohjauksella.
+
-**Kuinka OmniRoute ratkaisee sen:**
+
+📈 15. "I need to scale without losing performance"
-- A2A JSON-RPC -päätepiste ("POST /a2a") ja "message/send" ja "message/stream"
-- SSE-suoratoisto päätetilan etenemisellä
-- Tehtävien elinkaaren sovellusliittymät tehtäville/get- ja tasks/cancel
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected.
-
-🛰️ 19. "Tarvitsen todellista MCP-prosessin kuntoa, en arvattua tilaa"
+**How OmniRoute solves it:**
-Operatiivisten tiimien on tiedettävä, onko MCP todella elossa, ei vain sitä, onko API tavoitettavissa.
+- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency
+- **Request Idempotency** — 5s deduplication window for identical requests
+- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking
+- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence
+- **API Key Validation Cache** — 3-tier cache for production performance
+- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime
-**Kuinka OmniRoute ratkaisee sen:**
+
-- Ajonaikainen syketiedosto, jossa on PID, aikaleimat, kuljetus, työkalujen määrä ja laajuustila
-- MCP-tilan API, joka yhdistää sykkeen + viimeaikaisen toiminnan
-- Käyttöliittymän tilakortit prosessin / käytettävyyden / sydämenlyöntien tuoreudelle
+
+🤖 16. "I want to control model behavior globally"
-
-📋 20. "Tarvitsen tarkastettavan MCP-työkalun suorittamisen"
+Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical.
-Kun työkalut muuttavat määrityksiä tai käynnistävät operaatioita, tiimit tarvitsevat rikosteknistä jäljitettävyyttä.
+**How OmniRoute solves it:**
-**Kuinka OmniRoute ratkaisee sen:**
+- **System Prompt Injection** — Global prompt applied to all requests
+- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive)
+- **9 Routing Strategies** — Global strategies that determine how requests are distributed
+- **Wildcard Router** — `provider/*` patterns route dynamically to any provider
+- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard
+- **Provider Toggle** — Enable/disable all connections for a provider with one click
+- **Blocked Providers** — Exclude specific providers from `/v1/models` listing
-- SQLite-tuettu tarkastusloki MCP-työkalukutsuille
-- Suodattimet työkalun, onnistumisen/epäonnistumisen, API-avaimen ja sivutuksen mukaan
-- Kojelaudan tarkastustaulukko + tilastopäätepisteet automatisointia varten
+
-
-🔐 21. "Tarvitsen laajennettuja MCP-oikeuksia integraatiota kohti"
+
+🧰 17. "I need MCP tools as first-class product capabilities"
-Eri asiakkailla tulisi olla vähiten käyttöoikeus työkaluluokkiin.
+Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer.
-**Kuinka OmniRoute ratkaisee sen:**
+**How OmniRoute solves it:**
-- 10 rakeista MCP-skooppia ohjattua työkalujen käyttöä varten
-- Laajuuden valvonta ja näkyvyys MCP-hallintaliittymässä
-- Turvallinen oletusasento käyttötyökaluille
+- MCP appears in the dashboard navigation and endpoint protocol tab
+- Dedicated MCP management page with process, tools, scopes, and audit
+- Built-in quick-start for `omniroute --mcp` and client onboarding
-
-⚙️ 22. "Tarvitsen toiminnanohjausta ilman uudelleensijoittamista"
+
-Tiimit tarvitsevat nopeita ajonaikaisia muutoksia tapausten tai kustannustapahtumien aikana.
+
+🧠 18. "I need A2A orchestration with sync + stream task paths"
-**Kuinka OmniRoute ratkaisee sen:**
+Agent workflows need both direct replies and long-running streamed execution with lifecycle control.
-- Vaihda yhdistelmäaktivointia suoraan MCP-kojelaudalta
-- Käytä joustavuusprofiileja ennalta määritetyistä käytäntöpaketeista
-- Nollaa katkaisijan tila samasta käyttöpaneelista
+**How OmniRoute solves it:**
-
-🔄 23. "Tarvitsen live-A2A-tehtävän elinkaaren näkyvyyden ja peruutuksen"
+- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream`
+- SSE streaming with terminal state propagation
+- Task lifecycle APIs for `tasks/get` and `tasks/cancel`
-Ilman elinkaaren näkyvyyttä tehtäväkohtauksista tulee vaikeasti luokiteltuja.
+
-**Kuinka OmniRoute ratkaisee sen:**
+
+🛰️ 19. "I need real MCP process health, not guessed status"
-- Tehtäväluettelo / suodatus tilan / taitojen mukaan ja sivutus
-- Tehtävän metatietojen, tapahtumien ja artefaktien yksityiskohdat
-- Tehtävän peruutuksen päätepiste ja käyttöliittymätoiminto vahvistuksen kanssa
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable.
-
-🌊 24. "Tarvitsen aktiivisia suoratoistotietoja A2A-kuormitukseen"
+**How OmniRoute solves it:**
-Streaming-työnkulut edellyttävät toiminnallista tietoa samanaikaisuudesta ja reaaliaikaisista yhteyksistä.
+- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode
+- MCP status API combining heartbeat + recent activity
+- UI status cards for process/uptime/heartbeat freshness
-**Kuinka OmniRoute ratkaisee sen:**
+
-- Aktiiviset virtalaskurit integroitu A2A-tilaan
-- Viimeisen tehtävän aikaleima ja tilakohtaiset määrät
-- A2A kojelautakortit reaaliaikaiseen toimintojen seurantaan
+
+📋 20. "I need auditable MCP tool execution"
-
-🪪 25. "Tarvitsen asiakkaille tavallisen agenttihaun"
+When tools mutate config or trigger ops actions, teams need forensic traceability.
-Ulkoiset asiakkaat ja orkesterit tarvitsevat koneellisesti luettavaa metadataa käyttöönottoa varten.
+**How OmniRoute solves it:**
-**Kuinka OmniRoute ratkaisee sen:**
+- SQLite-backed audit logging for MCP tool calls
+- Filters by tool, success/failure, API key, and pagination
+- Dashboard audit table + stats endpoints for automation
-- Agenttikortti esillä osoitteessa `/.well-known/agent.json'
-- Johdon käyttöliittymässä näkyvät valmiudet ja taidot
-- A2A status API sisältää etsintämetatiedot automatisointia varten
+
-
-🧭 26. "Tarvitsen protokollan löydettävyyden tuotteen käyttökokemuksessa"
+
+🔐 21. "I need scoped MCP permissions per integration"
-Jos käyttäjät eivät löydä protokollapintoja, käyttöönoton ja tuen laatu heikkenee.
+Different clients should have least-privilege access to tool categories.
-**Kuinka OmniRoute ratkaisee sen:**
+**How OmniRoute solves it:**
-- Yhdistetty**Päätepisteet**-sivu, jossa on välilehdet välityspalvelin-, MCP-, A2A- ja API-päätepisteille
-- Inline-palvelun tila vaihtuu (Online/Offline) MCP:lle ja A2A:lle
-- Linkit yleiskatsauksesta erityisiin hallintavälilehtiin
+- 10 granular MCP scopes for controlled tool access
+- Scope enforcement and visibility in MCP management UI
+- Safe default posture for operational tooling
-
-🧪 27. "Tarvitsen päästä päähän -protokollan validoinnin oikeiden asiakkaiden kanssa"
+
-Valetestit eivät riitä vahvistamaan protokollan yhteensopivuutta ennen julkaisua.
+
+⚙️ 22. "I need operational controls without redeploying"
-**Kuinka OmniRoute ratkaisee sen:**
+Teams need quick runtime changes during incidents or cost events.
-- E2E-paketti, joka käynnistää sovelluksen ja käyttää todellista MCP SDK -asiakassiirtoa
-- A2A-asiakas testaa virtojen löytämistä, lähettämistä, suoratoistoa, vastaanottamista ja peruuttamista
-- Tarkista väitteet MCP-tarkastuksen ja A2A-tehtävien sovellusliittymien kanssa
+**How OmniRoute solves it:**
-
-📡 28. "Tarvitsen yhtenäisen havaittavuuden kaikissa käyttöliittymissä"
+- Switch combo activation directly from MCP dashboard
+- Apply resilience profiles from pre-defined policy packs
+- Reset circuit breaker state from the same operations panel
-Havainnon jakaminen protokollan mukaan luo kuolleita kulmia ja pidemmän MTTR:n.
+
-**Kuinka OmniRoute ratkaisee sen:**
+
+🔄 23. "I need live A2A task lifecycle visibility and cancellation"
-- Yhdistetyt kojelaudat/lokit/analytiikka yhdessä tuotteessa
-- Terveys + auditointi + pyyntö telemetria OpenAI-, MCP- ja A2A-tasoilla
-- Toiminnalliset sovellusliittymät tilaa ja automaatiota varten
+Without lifecycle visibility, task incidents become hard to triage.
-
-💼 29. "Tarvitsen yhden suoritusajan välityspalvelimelle + työkaluille + agentin orkestraatiolle"
+**How OmniRoute solves it:**
-Useiden erillisten palvelujen suorittaminen lisää käyttökustannuksia ja vikatiloja.
+- Task listing/filtering by state/skill with pagination
+- Drill-down on task metadata, events, and artifacts
+- Task cancellation endpoint and UI action with confirmation
-**Kuinka OmniRoute ratkaisee sen:**
+
-- OpenAI-yhteensopiva välityspalvelin, MCP-palvelin ja A2A-palvelin yhdessä pinossa
-- Jaettu todennus, joustavuus, tietovarasto ja havaittavuus
-- Yhdenmukainen toimintamalli kaikilla vuorovaikutuspinnoilla
+
+🌊 24. "I need active stream metrics for A2A load"
-
-🚀 30. "Minun on lähetettävä agenttityönkulkuja ilman liimakoodin leviämistä"
+Streaming workflows require operational insight into concurrency and live connections.
-Tiimit menettävät nopeutta yhdistäessään useita ad-hoc-palveluita ja skriptejä.
+**How OmniRoute solves it:**
-**Kuinka OmniRoute ratkaisee sen:**
+- Active stream counters integrated into A2A status
+- Last task timestamp and per-state counts
+- A2A dashboard cards for real-time ops monitoring
-- Yhtenäinen päätepistestrategia asiakkaille ja edustajille
-- Sisäänrakennetut protokollien hallinnan käyttöliittymät ja savun vahvistuspolut
-- Tuotantovalmis perusta (turvallisuus, puunkorjuu, joustavuus, varmuuskopiointi)
+
+
+
+🪪 25. "I need standard agent discovery for clients"
+
+External clients and orchestrators need machine-readable metadata for onboarding.
+
+**How OmniRoute solves it:**
+
+- Agent Card exposed at `/.well-known/agent.json`
+- Capabilities and skills shown in management UI
+- A2A status API includes discovery metadata for automation
+
+
+
+
+🧭 26. "I need protocol discoverability in the product UX"
+
+If users cannot discover protocol surfaces, adoption and support quality drop.
+
+**How OmniRoute solves it:**
+
+- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints
+- Inline service status toggles (Online/Offline) for MCP and A2A
+- Links from overview to dedicated management tabs
+
+
+
+
+🧪 27. "I need end-to-end protocol validation with real clients"
+
+Mock tests are not enough to validate protocol compatibility before release.
+
+**How OmniRoute solves it:**
+
+- E2E suite that boots app and uses real MCP SDK client transport
+- A2A client tests for discovery, send, stream, get, and cancel flows
+- Cross-check assertions against MCP audit and A2A tasks APIs
+
+
+
+
+📡 28. "I need unified observability across all interfaces"
+
+Splitting observability by protocol creates blind spots and longer MTTR.
+
+**How OmniRoute solves it:**
+
+- Unified dashboards/logs/analytics in one product
+- Health + audit + request telemetry across OpenAI, MCP, and A2A layers
+- Operational APIs for status and automation
+
+
+
+
+💼 29. "I need one runtime for proxy + tools + agent orchestration"
+
+Running many separate services increases operational cost and failure modes.
+
+**How OmniRoute solves it:**
+
+- OpenAI-compatible proxy, MCP server, and A2A server in one stack
+- Shared auth, resilience, data store, and observability
+- Consistent policy model across all interaction surfaces
+
+
+
+
+🚀 30. "I need to ship agentic workflows without glue-code sprawl"
+
+Teams lose velocity when stitching multiple ad-hoc services and scripts.
+
+**How OmniRoute solves it:**
+
+- Unified endpoint strategy for clients and agents
+- Built-in protocol management UIs and smoke validation paths
+- Production-ready foundations (security, logging, resilience, backup)
+
+
### Example Playbooks (Integrated Use Cases)
-**Ohjekirja A: maksimoi maksullinen tilaus + halpa varmuuskopio**```txt
+**Playbook A: Maximize paid subscription + cheap backup**
+
+```txt
Combo: "maximize-claude"
1. cc/claude-opus-4-6
2. glm/glm-4.7
@@ -607,21 +690,23 @@ Combo: "maximize-claude"
Monthly cost: $20 + small backup spend
Outcome: higher quality, near-zero interruption
-````
+```
-**Ohjekirja B: Nollahintainen koodauspino**```txt
+**Playbook B: Zero-cost coding stack**
+
+```txt
Combo: "free-forever"
-
-1. gc/gemini-3-flash
-2. if/kimi-k2-thinking
-3. qw/qwen3-coder-plus
+ 1. gc/gemini-3-flash
+ 2. if/kimi-k2-thinking
+ 3. qw/qwen3-coder-plus
Monthly cost: $0
Outcome: stable free coding workflow
+```
-````
+**Playbook C: 24/7 always-on fallback chain**
-**Playbook C: 24/7 aina päällä oleva varaketju**```txt
+```txt
Combo: "always-on"
1. cc/claude-opus-4-6
2. cx/gpt-5.2-codex
@@ -630,122 +715,134 @@ Combo: "always-on"
5. if/kimi-k2-thinking
Outcome: deep fallback depth for deadline-critical workloads
-````
+```
-**Pelikirja D: Agentti toimii MCP:llä + A2A**```txt
+**Playbook D: Agent ops with MCP + A2A**
-1. Start MCP transport (`omniroute --mcp`) for tool-driven operations
-2. Run A2A tasks via `message/send` and `message/stream`
-3. Observe via /dashboard/endpoint (MCP and A2A tabs)
-4. Toggle services via inline status controls
-
-````
+```txt
+1) Start MCP transport (`omniroute --mcp`) for tool-driven operations
+2) Run A2A tasks via `message/send` and `message/stream`
+3) Observe via /dashboard/endpoint (MCP and A2A tabs)
+4) Toggle services via inline status controls
+```
---
## 🆓 Start Free — Zero Configuration Cost
-> Määritä AI-koodaus minuuteissa hintaan**0 $/kk**. Yhdistä nämä ilmaiset tilit ja käytä sisäänrakennettua**Free Stack**-yhdistelmää.
+> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo.
-| Vaihe | Toiminta | Palveluntarjoajat avattu |
-| ---- | --------------------------------------------------- | ------------------------------------------------------------------- |
-| 1 | Yhdistä**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**rajaton**|
-| 2 | Yhdistä**Qoder**(Google OAuth) | kimi-k2-ajattelu, qwen3-coder-plus, deepseek-r1... —**rajoittamaton**|
-| 3 | Yhdistä**Qwen**(laitekoodi) | qwen3-coder-plus, qwen3-coder-flash... —**rajoittamaton**|
-| 4 | Yhdistä**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180K/kk ilmaiseksi**|
-| 5 | `/dashboard/combos` →**Ilmainen pino ($0)**malli | Round-robin kaikki ilmaiset palveluntarjoajat automaattisesti |
+| Step | Action | Providers Unlocked |
+| ---- | -------------------------------------------------- | ------------------------------------------------------------------ |
+| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** |
+| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** |
+| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** |
+| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** |
+| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically |
-**Osoita mikä tahansa IDE/CLI osoitteeseen:**"http://localhost:20128/v1" · API-avain: "any-string" · Valmis.
+**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done.
->**Valinnainen lisäkattavuus (myös ilmainen):**Groq API-avain (30 RPM ilmaiseksi), NVIDIA NIM (40 RPM ilmaiseksi, 70+ mallit), Cerebras (1 milj. tok/päivä), LongCat API-avain (50 milj. tokenia/päivä!), Cloudflare Workers AI (10 000 neuronia/vrk, 50+ mallia).## Pikakäynnistys
+> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models).
+
+## Pikakäynnistys
### 1) Install and run
```bash
npm install -g omniroute
omniroute
-````
+```
-> **pnpm-käyttäjät:**Suorita `pnpm approve-builds -g` asennuksen jälkeen, jotta voit ottaa käyttöön `better-sqlite3` ja `@swc/core` vaatimat alkuperäiset koontiskriptit:
+> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`:
>
> ```bash
> pnpm install -g omniroute
-> pnpm approve-builds -g # Valitse kaikki paketit → hyväksy
-> kaikkialla
+> pnpm approve-builds -g # Select all packages → approve
+> omniroute
> ```
-Hallintapaneeli avautuu osoitteessa "http://localhost:20128" ja API-perus-URL-osoite on "http://localhost:20128/v1".
+Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`.
-| Komento | Kuvaus |
-| ----------------------- | -------------------------------------------------------------------- |
-| `omniroute` | Käynnistä palvelin (`PORT=20128`, API ja kojelauta samassa portissa) |
-| `omniroute --port 3000` | Aseta kanoninen/API-portiksi 3000 |
-| `omniroute --mcp` | Käynnistä MCP-palvelin (stdio-kuljetus) |
-| `omniroute --no-open` | Älä avaa selainta automaattisesti |
-| `omniroute --help` | Näytä ohje |
+| Command | Description |
+| ----------------------- | ----------------------------------------------------------- |
+| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) |
+| `omniroute --port 3000` | Set canonical/API port to 3000 |
+| `omniroute --mcp` | Start MCP server (stdio transport) |
+| `omniroute --no-open` | Don't auto-open browser |
+| `omniroute --help` | Show help |
-Valinnainen jaettu porttitila:```bash
+Optional split-port mode:
+
+```bash
PORT=20128 DASHBOARD_PORT=20129 omniroute
-
-# API: http://localhost:20128/v1
-
+# API: http://localhost:20128/v1
# Dashboard: http://localhost:20129
-
-````
+```
### Long-Running Streaming Timeouts
-Useimpiin käyttöönottoihin tarvitset vain:
+For most deployments, you only need:
-| Muuttuja | Oletus | Tarkoitus |
-| ------------------------- | ------------------------------ | -------------------------------------------------------------- --------------------------------------------------------------- |
-| `REQUEST_TIMEOUT_MS` | "600000" | Jaettu lähtötaso ylävirran noutoa, piilotettuja Undici-aikakatkaisuja, TLS-sormenjälkipyyntöjä ja API-siltapyyntöjen/välityspalvelinten aikakatkaisuja varten |
-| `STREAM_IDLE_TIMEOUT_MS` | perii REQUEST_TIMEOUT_MS | Suurin väli suoratoistopalojen välillä ennen kuin OmniRoute keskeyttää SSE-virran |
+| Variable | Default | Purpose |
+| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- |
+| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts |
+| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream |
-Taaksepäin yhteensopivuus säilyy: olemassa olevat "FETCH_TIMEOUT_MS", "API_BRIDGE_PROXY_TIMEOUT_MS" ja muut tasokohtaiset aikakatkaisumuuttujat toimivat edelleen ja ohittavat jaetun perustason.
+Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline.
-Edistyneet ohitukset ovat käytettävissä, jos tarvitset tarkempaa ohjausta:| Muuttuja | Oletus | Tarkoitus |
-| ----------------------------------------- | ------------------------------------------- | --------------------------------------------------------------------- |
-| `FETCH_TIMEOUT_MS` | perii REQUEST_TIMEOUT_MS | Päähaun keskeytyssignaalin käyttämä ylävirran pyynnön kokonaisaikakatkaisu |
-| `FETCH_HEADERS_TIMEOUT_MS` | perii `FETCH_TIMEOUT_MS` | Undici-aikaraja ylävirran vastausotsikoiden vastaanottamiselle |
-| `FETCH_BODY_TIMEOUT_MS` | perii `FETCH_TIMEOUT_MS` | Undici-aikaraja ylävirran runkokappaleiden välillä (`0` poistaa sen käytöstä) |
-| `FETCH_CONNECT_TIMEOUT_MS` | "30000" | Undici TCP-yhteyden aikakatkaisu |
-| `FETCH_KEEPALIVE_TIMEOUT_MS` | "4000" | Undici idle keep-alive socket timeout |
-| `TLS_CLIENT_TIMEOUT_MS` | perii `FETCH_TIMEOUT_MS` | Aikakatkaisu wreq-js:n kautta tehdyille TLS-sormenjälkipyynnöille |
-| `API_BRIDGE_PROXY_TIMEOUT_MS` | perii REQUEST_TIMEOUT_MS tai 30000 | Aikakatkaisu välityspalvelimen `/v1' edelleenlähetykselle API-portista kojelautaporttiin |
-| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | "max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)" | Saapuvan pyynnön aikakatkaisu API-siltapalvelimella |
-| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | "60000" | Saapuvan otsikon aikakatkaisu API-siltapalvelimella |
-| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | "5000" | Keep-alive aikakatkaisu API-siltapalvelimella |
-| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | "0" | Socketin passiivisuuden aikakatkaisu API-siltapalvelimessa (`0` poistaa sen käytöstä) |
+Advanced overrides are available if you need finer control:
-Jos käytät OmniRoutea Nginxin, Caddyn, Cloudflaren tai muun käänteisen välityspalvelimen takana, varmista, että välityspalvelin
-aikakatkaisut ovat myös korkeammat kuin OmniRoute-streamin/haun aikakatkaisut.### 2) Connect providers and create your API key
+| Variable | Default | Purpose |
+| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- |
+| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal |
+| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers |
+| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) |
+| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout |
+| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout |
+| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` |
+| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port |
+| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server |
+| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server |
+| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server |
+| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) |
-1. Avaa Dashboard → Providers ja yhdistä vähintään yksi palveluntarjoaja (OAuth- tai API-avain).
-2. Avaa Dashboard → `Endpoints` ja luo API-avain.
-3. (Valinnainen) Avaa Dashboard → "Yhdistelmät" ja aseta varaketju.### 3) Point your coding tool to OmniRoute
+If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy
+timeouts are also higher than your OmniRoute stream/fetch timeouts.
+
+### 2) Connect providers and create your API key
+
+1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key).
+2. Open Dashboard → `Endpoints` and create an API key.
+3. (Optional) Open Dashboard → `Combos` and set your fallback chain.
+
+### 3) Point your coding tool to OmniRoute
```txt
Base URL: http://localhost:20128/v1
API Key: [copy from Endpoint page]
Model: if/kimi-k2-thinking (or any provider/model prefix)
-````
+```
-Toimii Claude Coden, Codex CLI:n, Gemini CLI:n, Cursorin, Clinen, OpenClawn, OpenCoden ja OpenAI-yhteensopivien SDK:iden kanssa.### 4) Enable and validate protocols (v2.0)
+Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs.
-**MCP (työkaluohjattuihin toimintoihin):**```bash
+### 4) Enable and validate protocols (v2.0)
+
+**MCP (for tool-driven operations):**
+
+```bash
omniroute --mcp
+```
-````
+Then connect your MCP client over `stdio` and test tools like:
-Yhdistä sitten MCP-asiakkaasi "stdion" kautta ja testaa työkaluja, kuten:
+- `omniroute_get_health`
+- `omniroute_list_combos`
-- `omniroute_get_health'
-- "omniroute_list_combos".
+**A2A (for agent-to-agent workflows):**
-**A2A (agenttien välisille työnkuluille):**```bash
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
```bash
curl -X POST http://localhost:20128/a2a \
@@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \
npm run test:protocols:e2e
```
-Tämä sarja tarkistaa todelliset MCP- ja A2A-asiakasvirrat käynnissä olevaa sovellusta vastaan.### Alternative: run from source
+This suite validates real MCP and A2A client flows against a running app.
+
+### Alternative: run from source
```bash
cp .env.example .env
@@ -767,13 +866,13 @@ npm install
PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev
```
-
-Vid Linux (`xbps-src` malli)
+
+Void Linux (`xbps-src` template)
-Void Linux -käyttäjille voit rakentaa alkuperäisen paketin käyttämällä `xbps-src`. Tallenna tämä lohko nimellä "srcpkgs/omniroute/template":```bash
+For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`:
+```bash
# Template file for 'omniroute'
-
pkgname=omniroute
version=3.4.1
revision=1
@@ -785,7 +884,7 @@ license="MIT"
homepage="https://github.com/diegosouzapw/OmniRoute"
distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz"
checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b
-system_accounts="\_omniroute"
+system_accounts="_omniroute"
omniroute_homedir="/var/lib/omniroute"
export NODE_ENV=production
export npm_config_engine_strict=false
@@ -793,71 +892,70 @@ export npm_config_loglevel=error
export npm_config_fund=false
export npm_config_audit=false
-do_build() { # Determine target CPU arch for node-gyp
-local \_gyp_arch
-case "$XBPS_TARGET_MACHINE" in
-aarch64*) \_gyp_arch=arm64 ;;
-armv7*|armv6*) \_gyp_arch=arm ;;
-i686*) \_gyp_arch=ia32 ;;
-\*) \_gyp_arch=x64 ;;
-esac
+do_build() {
+ # Determine target CPU arch for node-gyp
+ local _gyp_arch
+ case "$XBPS_TARGET_MACHINE" in
+ aarch64*) _gyp_arch=arm64 ;;
+ armv7*|armv6*) _gyp_arch=arm ;;
+ i686*) _gyp_arch=ia32 ;;
+ *) _gyp_arch=x64 ;;
+ esac
- # 1) Install all deps – skip scripts (no network in do_build, native modules
- # compiled separately below; better-sqlite3 is serverExternalPackage so
- # Next.js does not execute it during next build)
- NODE_ENV=development npm ci --ignore-scripts
+ # 1) Install all deps – skip scripts (no network in do_build, native modules
+ # compiled separately below; better-sqlite3 is serverExternalPackage so
+ # Next.js does not execute it during next build)
+ NODE_ENV=development npm ci --ignore-scripts
- # 2) Build the Next.js standalone bundle
- npm run build
+ # 2) Build the Next.js standalone bundle
+ npm run build
- # 3) Copy static assets into standalone
- cp -r .next/static .next/standalone/.next/static
- [ -d public ] && cp -r public .next/standalone/public || true
+ # 3) Copy static assets into standalone
+ cp -r .next/static .next/standalone/.next/static
+ [ -d public ] && cp -r public .next/standalone/public || true
- # 4) Compile better-sqlite3 native binding for the target architecture.
- # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
- # without npm altering them.
- local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
- (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
+ # 4) Compile better-sqlite3 native binding for the target architecture.
+ # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
+ # without npm altering them.
+ local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
+ (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
- # 5) Place the compiled binding into the standalone bundle
- local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
- mkdir -p "$_bs3_release"
- cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
+ # 5) Place the compiled binding into the standalone bundle
+ local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
+ mkdir -p "$_bs3_release"
+ cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
- # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
- # so sharp is not used at runtime; x64 .so files would break aarch64 strip
- rm -rf .next/standalone/node_modules/@img
-
- # 7) Copy pino runtime deps omitted by Next.js static analysis:
- # pino-abstract-transport – required by pino's worker thread
- # split2 – dep of pino-abstract-transport
- # process-warning – dep of pino itself
- for _mod in pino-abstract-transport split2 process-warning; do
- cp -r "node_modules/$_mod" .next/standalone/node_modules/
- done
+ # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
+ # so sharp is not used at runtime; x64 .so files would break aarch64 strip
+ rm -rf .next/standalone/node_modules/@img
+ # 7) Copy pino runtime deps omitted by Next.js static analysis:
+ # pino-abstract-transport – required by pino's worker thread
+ # split2 – dep of pino-abstract-transport
+ # process-warning – dep of pino itself
+ for _mod in pino-abstract-transport split2 process-warning; do
+ cp -r "node_modules/$_mod" .next/standalone/node_modules/
+ done
}
do_check() {
-npm run test:unit
+ npm run test:unit
}
do_install() {
-vmkdir usr/lib/omniroute/.next
+ vmkdir usr/lib/omniroute/.next
- vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
+ vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
- # Prevent removal of empty Next.js app router dirs by the post-install hook
- for _d in \
- .next/standalone/.next/server/app/dashboard \
- .next/standalone/.next/server/app/dashboard/settings \
- .next/standalone/.next/server/app/dashboard/providers; do
- touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
- done
-
- cat > "${WRKDIR}/omniroute" <<'EOF'
+ # Prevent removal of empty Next.js app router dirs by the post-install hook
+ for _d in \
+ .next/standalone/.next/server/app/dashboard \
+ .next/standalone/.next/server/app/dashboard/settings \
+ .next/standalone/.next/server/app/dashboard/providers; do
+ touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
+ done
+ cat > "${WRKDIR}/omniroute" <<'EOF'
#!/bin/sh
export PORT="${PORT:-20128}"
export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}"
@@ -869,10 +967,9 @@ EOF
}
post_install() {
-vlicense LICENSE
+ vlicense LICENSE
}
-
-````
+```
@@ -880,9 +977,11 @@ vlicense LICENSE
## 🐳 Docker
-OmniRoute on saatavilla julkisena Docker-kuvana [Docker Hubista](https://hub.docker.com/r/diegosouzapw/omniroute).
+OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
-**Pikaajo:**```bash
+**Quick run:**
+
+```bash
docker run -d \
--name omniroute \
--restart unless-stopped \
@@ -890,85 +989,96 @@ docker run -d \
-p 20128:20128 \
-v omniroute-data:/app/data \
diegosouzapw/omniroute:latest
-````
+```
-**Ympäristötiedostolla:**```bash
+**With environment file:**
+```bash
# Copy and edit .env first
-
cp .env.example .env
docker run -d \
- --name omniroute \
- --restart unless-stopped \
- --stop-timeout 40 \
- --env-file .env \
- -p 20128:20128 \
- -v omniroute-data:/app/data \
- diegosouzapw/omniroute:latest
+ --name omniroute \
+ --restart unless-stopped \
+ --stop-timeout 40 \
+ --env-file .env \
+ -p 20128:20128 \
+ -v omniroute-data:/app/data \
+ diegosouzapw/omniroute:latest
+```
-````
+**Using Docker Compose:**
-**Docker Composen käyttäminen:**```bash
+```bash
# Base profile (no CLI tools)
docker compose --profile base up -d
# CLI profile (Claude Code, Codex, OpenClaw built-in)
docker compose --profile cli up -d
-````
+```
-Dashboard-tuki Dockerin käyttöönottoille sisältää nyt yhden napsautuksen**Cloudflare Quick Tunnel**kohdassa "Dashboard → Endpoints". Ensimmäinen sallii lataukset "cloudflared" vain tarvittaessa, käynnistää väliaikaisen tunnelin nykyiseen "/v1"-päätepisteeseen ja näyttää luodun "https://\*.trycloudflare.com/v1" URL-osoitteen suoraan tavallisen julkisen URL-osoitteesi alapuolelle.
+Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL.
-Huomautuksia:
+Notes:
-- Quick Tunnelin URL-osoitteet ovat väliaikaisia ja muuttuvat jokaisen uudelleenkäynnistyksen jälkeen.
-- Pikatunneleita ei palauteta automaattisesti OmniRouten tai kontin uudelleenkäynnistyksen jälkeen. Ota ne uudelleen käyttöön kojelaudasta tarvittaessa.
-- Hallittu asennus tukee tällä hetkellä Linuxia, macOS:ää ja Windowsia x64-/arm64-käyttöjärjestelmässä.
-- Hallitut pikatunnelit käyttävät oletusarvoisesti HTTP/2-siirtoa, jotta vältetään meluisat QUIC UDP -puskurivaroitukset rajoitetuissa säiliöympäristöissä. Aseta CLOUDFLARED_PROTOCOL=quic tai auto, jos haluat toisenlaisen kuljetuksen.
-- Docker-kuvat niputtavat järjestelmän CA-juuret ja välittävät ne hallitulle "cloudflaredille", mikä välttää TLS-luottamushäiriöt, kun tunneli käynnistyy säilön sisällä.
-- SQLite toimii WAL-tilassa. `docker stop`:n tulee antaa päättyä, jotta OmniRoute voi tarkistaa viimeisimmät muutokset takaisin `storage.sqlite`-tiedostoon.
-- Mukana olevissa Compose-tiedostoissa on jo asetettu 40 s stop lisäaika. Jos suoritat kuvan suoraan, pidä `--stop-timeout 40` (tai vastaava), jotta manuaaliset pysäytykset eivät katkaise sammutusta.
-- Aseta `CLOUDFLARED_BIN=/absolute/path/to/cloudflared', jos haluat OmniRouten käyttävän olemassa olevaa binaaria sen lataamisen sijaan.
+- Quick Tunnel URLs are temporary and change after every restart.
+- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed.
+- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`.
+- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport.
+- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container.
+- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`.
+- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup.
+- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one.
-**Docker Compose with Caddy (HTTPS Auto-TLS) käyttäminen:**
+**Using Docker Compose with Caddy (HTTPS Auto-TLS):**
-OmniRoute voidaan paljastaa turvallisesti käyttämällä Caddyn automaattista SSL-tilausta. Varmista, että verkkotunnuksesi DNS-tietue A osoittaa palvelimesi IP-osoitteeseen.```yaml
+OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP.
+
+```yaml
services:
-omniroute:
-image: diegosouzapw/omniroute:latest
-container_name: omniroute
-restart: unless-stopped
-volumes: - omniroute-data:/app/data
-environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com
+ omniroute:
+ image: diegosouzapw/omniroute:latest
+ container_name: omniroute
+ restart: unless-stopped
+ volumes:
+ - omniroute-data:/app/data
+ environment:
+ - PORT=20128
+ - NEXT_PUBLIC_BASE_URL=https://your-domain.com
-caddy:
-image: caddy:latest
-container_name: caddy
-restart: unless-stopped
-ports: - "80:80" - "443:443"
-command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
+ caddy:
+ image: caddy:latest
+ container_name: caddy
+ restart: unless-stopped
+ ports:
+ - "80:80"
+ - "443:443"
+ command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
volumes:
-omniroute-data:
+ omniroute-data:
+```
-````
+| Image | Tag | Size | Description |
+| ------------------------ | -------- | ------ | --------------------- |
+| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release |
+| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version |
-| Kuva | Tag | Koko | Kuvaus |
-| ------------------------- | -------- | ------ | ---------------------- |
-| "diegosouzapw/omniroute" | "viimeisin" | ~250 Mt | Uusin vakaa julkaisu |
-| "diegosouzapw/omniroute" | "1.0.3" | ~250 Mt | Nykyinen versio |---
+---
## 🖥️ Desktop App — Offline & Always-On
-> 🆕**UUSI!**OmniRoute on nyt saatavilla**alkuperäisenä työpöytäsovelluksena**Windowsille, macOS:lle ja Linuxille.
+> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux.
-Suorita OmniRoute itsenäisenä työpöytäsovelluksena – ei päätelaitetta, ei selainta, ei vaadi Internetiä paikallisiin malleihin. Elektronipohjainen sovellus sisältää:
+Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes:
-- 🖥️**Native Window**- Oma sovellusikkuna, jossa on integraatio järjestelmälokeroon
-- 🔄**Automaattinen käynnistys**— Käynnistä OmniRoute järjestelmään kirjautuessasi
-- 🔔**Alkuperäiset ilmoitukset**- Saat ilmoituksia kiintiön loppumisesta tai palveluntarjoajan ongelmista
-- ⚡**Asennus yhdellä napsautuksella**— NSIS (Windows), DMG (macOS), AppImage (Linux)
-- 🌐**Offline-tila**— Toimii täysin offline-tilassa mukana toimitetun palvelimen kanssa### Pikakäynnistys
+- 🖥️ **Native Window** — Dedicated app window with system tray integration
+- 🔄 **Auto-Start** — Launch OmniRoute on system login
+- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues
+- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux)
+- 🌐 **Offline Mode** — Works fully offline with bundled server
+
+### Pikakäynnistys
```bash
# Development mode
@@ -979,308 +1089,370 @@ npm run electron:build # Current platform
npm run electron:build:win # Windows (.exe)
npm run electron:build:mac # macOS (.dmg) — x64 & arm64
npm run electron:build:linux # Linux (.AppImage)
-````
+```
### System Tray
-Kun OmniRoute on minimoitu, se elää ilmaisinalueellasi nopeilla toimilla:
+When minimized, OmniRoute lives in your system tray with quick actions:
-- Avaa kojelauta
-- Vaihda palvelimen portti
-- Lopeta sovellus
+- Open dashboard
+- Change server port
+- Quit application
-📖 Täydellinen dokumentaatio: [`electron/README.md`](electron/README.md)---
+📖 Full documentation: [`electron/README.md`](electron/README.md)
+
+---
## 💰 Pricing at a Glance
-| Taso | Palveluntarjoaja | Kustannukset | Kiintiön nollaus | Paras |
-| ---------------- | --------------------------- | ---------------------------------- | ---------------------- | ---------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
-| **💳 TILAUS** | Claude Code (Pro) | 20 dollaria/kk | 5h + viikoittain | jo tilattu |
-| | Codex (Plus/Pro) | 20-200 $/kk | 5h + viikoittain | OpenAI-käyttäjät |
-| | Gemini CLI | **ILMAINEN** | 180 tk/kk + 1 tk/päivä | Kaikki! |
-| | GitHub Copilot | 10-19 $/kk | Kuukausittain | GitHub-käyttäjät |
-| **🔑 API-AVAIN** | NVIDIA NIM | **ILMAINEN**(kehittäjä ikuisesti) | ~40 rpm | 70+ avointa mallia |
-| | Aivot | **ILMAINEN**(1M tok/päivä) | 60K TPM / 30 RPM | Maailman nopein |
-| | Groq | **ILMAINEN**(30 RPM) | 14.4K RPD | Erittäin nopea Llama/Gemma |
-| | DeepSeek V3.2 | 0,27 $/1,10 $ per 1 milj | Ei yhtään | Paras hinta/laatu perustelu |
-| | xAI Grok-4 Fast | **0,20 $/0,50 $ per 1 milj**🆕 | Ei yhtään | Nopein + työkalukutsu, ultralow |
-| | xAI Grok-4 (vakio) | 0,20 $/1,50 $ per 1 milj 🆕 | Ei yhtään | Päättelyn lippulaiva xAI:lta |
-| | Mistral | Ilmainen kokeilu + maksullinen | Hinta rajoitettu | Eurooppalainen tekoäly |
-| | OpenRouter | Maksu per käyttö | Ei yhtään | 100+ mallia agr. |
-| **💰 EDULLISET** | GLM-5 (Z.AI:n kautta) 🆕 | 0,5 $/1 milj. | Päivittäin klo 10 | 128K lähtö, uusin lippulaiva |
-| | GLM-4.7 | 0,6 $/1 milj. | Päivittäin klo 10 | Budjetin varmuuskopio |
-| | MiniMax M2.5 🆕 | 0,3 $/1 milj. tulo | 5 tunnin rullaus | Päättely + agenttitehtävät |
-| | MiniMax M2.1 | 0,2 $/1 milj. | 5 tunnin rullaus | Halvin vaihtoehto |
-| | Kimi K2.5 (Moonshot API) 🆕 | Maksu per käyttö | Ei yhtään | Suora Moonshot API -käyttö |
-| | Kimi K2 | 9 dollaria/kk asunto | 10 milj. rahakkeita/kk | Ennustettavat kustannukset |
-| **🆓 ILMAINEN** | Qoder | **0 $** | Rajoittamaton | 5 mallia rajoittamaton |
-| | Qwen | **0 $** | Rajoittamaton | 4 mallia rajoittamaton |
-| | Kiro | **0 $** | Rajoittamaton | Claude Sonnet/Haiku (AWS Builder) |
-| | LongCat Flash-Lite 🆕 | **0 $**(50 milj. tok/päivä 🔥) | 1 RPS | Suurin ilmainen kiintiö maailmassa |
-| | Pölytys AI 🆕 | **0 $**(avainta ei tarvita) | 1 kpl/15 s | GPT-5, Claude, DeepSeek, Llama 4 |
-| | Cloudflare Workers AI 🆕 | **0 $**(10 000 neuronia/päivä) | ~150 tk/päivä | Yli 50 mallia, globaali reuna |
-| | Scaleway AI 🆕 | **0 $**(yhteensä 1 milj. tokeneja) | Hinta rajoitettu | EU/GDPR, Qwen3 235B, Llama 70B | > 🆕**Uusia malleja lisätty (maaliskuu 2026):**Grok-4 Fast -perhe hintaan 0,20 $/0,50 $/M (vertailuarvo 1143 ms – 30 % nopeampi kuin Gemini 2.5 Flash), GLM-5 Z.AI:n kautta 128K:n lähdöllä, MiniMax M2.5 Vc3 -perustelu, KiepSeed2-päivitys. Moonshot Direct API. |
+| Tier | Provider | Cost | Quota Reset | Best For |
+| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- |
+| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed |
+| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users |
+| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! |
+| | GitHub Copilot | $10-19/mo | Monthly | GitHub users |
+| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models |
+| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest |
+| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma |
+| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning |
+| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow |
+| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI |
+| | Mistral | Free trial + paid | Rate limited | European AI |
+| | OpenRouter | Pay-per-use | None | 100+ models aggr. |
+| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship |
+| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup |
+| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks |
+| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option |
+| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access |
+| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost |
+| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited |
+| | Qwen | **$0** | Unlimited | 4 models unlimited |
+| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) |
+| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth |
+| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 |
+| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge |
+| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B |
-**💡 0 dollarin yhdistelmäpino – täydellinen ilmainen asennus:**```
+> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API.
+**💡 $0 Combo Stack — The Complete Free Setup:**
+
+```
# 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever
+Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
+Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
+Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
+Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
+NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+```
-Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
-Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
-LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
-Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
-Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
-Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
-Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
-Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
-Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
-NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
-Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever.
-````
-
-**Nolla kustannuksia. Ei koskaan lopeta koodausta.**Määritä tämä yhdeksi OmniRoute-yhdistelmäksi, ja kaikki palautukset tapahtuvat automaattisesti – ei manuaalista vaihtoa koskaan.---
+---
---
## 🆓 Free Models — What You Actually Get
-> Kaikki alla olevat mallit ovat**100 % ilmaisia ilman luottokorttia**. OmniRoute reitittää automaattisesti niiden välillä, kun yksi kiintiö loppuu – yhdistä ne kaikki rikkomattomaksi 0 dollarin yhdistelmäksi.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
+> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo.
-| Malli | Etuliite | Raja | Hintarajoitus |
-| -------------------- | ------ | ------------- | ---------------------- |
-| `claude-sonnet-4,5` | `kr/` |**Rajoittamaton**| Ei raportoitu päivittäistä ylärajaa |
-| `claude-haiku-4,5` | `kr/` |**Rajoittamaton**| Ei raportoitu päivittäistä ylärajaa |
-| `claude-opus-4.6` | `kr/` |**Rajoittamaton**| Uusin Opus kautta Kiro |### 🟢 QODER MODELS (Free PAT via qodercli)
+### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
-| Malli | Etuliite | Raja | Hintarajoitus |
-| ------------------- | ------ | ------------- | ---------------- |
-| `kimi-k2-ajattelu` | "jos/" |**Rajoittamaton**| Ei ilmoitettu yläraja |
-| "qwen3-coder-plus" | "jos/" |**Rajoittamaton**| Ei ilmoitettu yläraja |
-| `deepseek-r1` | "jos/" |**Rajoittamaton**| Ei ilmoitettu yläraja |
-| "minimi-m2,1" | "jos/" |**Rajoittamaton**| Ei ilmoitettu yläraja |
-| "kimi-k2" | "jos/" |**Rajoittamaton**| Ei ilmoitettu yläraja |
+| Model | Prefix | Limit | Rate Limit |
+| ------------------- | ------ | ------------- | --------------------- |
+| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro |
-> Suositeltu yhteystapa:**Personal Access Token + `qodercli`**. Selain OAuth on
-> kokeellinen ja oletuksena poistettu käytöstä, ellei QODER_OAUTH_*-ympäristömuuttujia ole määritetty.### 🟡 QWEN MODELS (Device Code Auth)
+### 🟢 QODER MODELS (Free PAT via qodercli)
-| Malli | Etuliite | Raja | Hintarajoitus |
-| -------------------- | ------ | ------------- | -------------------- |
-| "qwen3-coder-plus" | `qw/` |**Rajoittamaton**| Ei ilmoitettu yläraja |
-| "qwen3-coder-flash" | `qw/` |**Rajoittamaton**| Ei ilmoitettu yläraja |
-| "qwen3-coder-next" | `qw/` |**Rajoittamaton**| Ei ilmoitettu yläraja |
-| "näön malli" | `qw/` |**Rajoittamaton**| Multimodaalinen (kuvat) |### 🟣 GEMINI CLI (Google OAuth)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------ | ------ | ------------- | --------------- |
+| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap |
+| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap |
+| `deepseek-r1` | `if/` | **Unlimited** | No reported cap |
+| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap |
+| `kimi-k2` | `if/` | **Unlimited** | No reported cap |
-| Malli | Etuliite | Raja | Hintarajoitus |
-| ------------------------- | ------ | ---------------------------- | ------------- |
-| `gemini-3-flash-preview` | `gc/` |**180 tk/kk**+ 1 tk/päivä | Kuukausittainen nollaus |
-| "gemini-2.5-pro" | `gc/` | 180 000/kk (jaettu pool) | Korkea laatu |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
+> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is
+> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured.
-| Taso | Päiväraja | Hintarajoitus | Huomautuksia |
-| ----------- | ------------ | ----------- | ------------------------------------------------------- |
-| Ilmainen (kehittäjä) | Ei tunnuskorkkia |**~40 RPM**| 70+ mallia; siirtyminen puhtaisiin korkorajoihin vuoden 2025 puolivälissä |
+### 🟡 QWEN MODELS (Device Code Auth)
-Suositut ilmaiset mallit: moonshotai/kimi-k2.5 (Kimi K2.5), z-ai/glm4.7 (GLM 4.7), deepseek-ai/deepseek-v3.2 (DeepSeek V3.2), nvidia/llama-3.3-70b-deepseekr,/deepseekr### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------- | ------ | ------------- | ------------------- |
+| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap |
+| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) |
-| Taso | Päiväraja | Hintarajoitus | Huomautuksia |
-| ---- | ------------------ | ----------------- | -------------------------------------------- |
-| Ilmainen |**1 milj. rahakkeita/päivä**| 60K TPM / 30 RPM | Maailman nopein LLM-päätelmä; nollautuu päivittäin |
+### 🟣 GEMINI CLI (Google OAuth)
-Saatavilla ilmaiseksi: "llama-3.3-70b", "llama-3.1-8b", "deepseek-r1-distill-llama-70b"### 🔴 GROQ (Free API Key — console.groq.com)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------------ | ------ | --------------------------- | ------------- |
+| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset |
+| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality |
-| Taso | Päiväraja | Hintarajoitus | Huomautuksia |
-| ---- | ------------- | ----------------- | ------------------------------------------ |
-| Ilmainen |**14.4K RPD**| 30 rpm mallia kohden | Ei luottokorttia; 429 rajalla, ei veloiteta |
+### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
-Saatavilla ilmaiseksi: "llama-3.3-70b-versatile", "gemma2-9b-it", "mixtral-8x7b", "whisper-large-v3"### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---------- | ------------ | ----------- | ------------------------------------------------------ |
+| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 |
-| Malli | Etuliite | Päivittäinen ilmainen kiintiö | Huomautuksia |
-| ------------------------------ | ------ | ------------------ | ------------------------ |
-| "LongCat-Flash-Lite" | `lc/` |**50 milj. rahakkeita**💥 | Suurin ilmainen kiintiö koskaan |
-| "LongCat-Flash-Chat" | `lc/` | 500 000 kuponkia | Monipuolinen chat |
-| "LongCat-Flash-Thinking" | `lc/` | 500 000 kuponkia | Päättely / CoT |
-| "LongCat-Flash-Thinking-2601" | `lc/` | 500 000 kuponkia | Tammikuun 2026 versio |
-| "LongCat-Flash-Omni-2603" | `lc/` | 500 000 kuponkia | Multimodaalinen |
+Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`
-> 100 % ilmainen julkisessa beta-tilassa. Rekisteröidy osoitteessa [longcat.chat](https://longcat.chat) sähköpostitse tai puhelimitse. Nollautuu päivittäin 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
-| Malli | Etuliite | Hintarajoitus | Toimittaja Takana |
-| ----------- | ------ | ----------- | ------------------- |
-| "openai" | `pol/` | 1 tarve/15 s | GPT-5 |
-| `claude` | `pol/` | 1 tarve/15 s | Antrooppinen Claude |
-| "kaksoset" | `pol/` | 1 tarve/15 s | Google Gemini |
-| `deepseek` | `pol/` | 1 kpl/15 s | DeepSeek V3 |
-| `laama` | `pol/` | 1 kpl/15 s | Meta Llama 4 Scout |
-| "mistral" | `pol/` | 1 kpl/15 s | Mistral AI |
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---- | ----------------- | ---------------- | ------------------------------------------- |
+| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily |
-> ✨**Nolla kitkaa:**Ei rekisteröitymistä, ei API-avainta. Lisää Pollinations-palveluntarjoaja tyhjällä avainkentällä ja se toimii välittömästi.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`
-| Taso | Päivittäiset neuronit | Vastaava käyttö | Huomautuksia |
-| ---- | ------------- | ---------------------------------------- | ------------------------ |
-| Ilmainen |**10 000**| ~150 LLM vastinetta / 500 s ääni / 15 000 upotusta | Maailmanlaajuinen etu, yli 50 mallia |
+### 🔴 GROQ (Free API Key — console.groq.com)
-Suositut ilmaiset mallit: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (ilmainen ääni!), `@cf/qwen/qwen2.5-coder-`15b-coder-
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---- | ------------- | ---------------- | ----------------------------------------- |
+| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged |
-> Vaatii API-tunnuksen + tilitunnuksen osoitteesta [dash.cloudflare.com](https://dash.cloudflare.com). Tallenna tilitunnus palveluntarjoajan asetuksiin.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`
-| Taso | Ilmainen kiintiö | Sijainti | Huomautuksia |
-| ---- | ------------- | ------------ | ------------------------------------ |
-| Ilmainen |**1 milj. rahakkeita**| 🇫🇷 Pariisi, EU | Luottokorttia ei tarvita rajoissa |
+### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
-Saatavilla ilmaiseksi: "qwen3-235b-a22b-instruct-2507" (Qwen3 235B!), "llama-3.1-70b-instruct", "mistral-small-3.2-24b-instruct-2506", "deepseek-v3-032"
+| Model | Prefix | Daily Free Quota | Notes |
+| ----------------------------- | ------ | ----------------- | ----------------------- |
+| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever |
+| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat |
+| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT |
+| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version |
+| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal |
-> EU/GDPR-yhteensopiva. Hanki API-avain osoitteesta [console.scaleway.com](https://console.scaleway.com).
+> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC.
->**💡 The Ultimate Free Stack (11 tarjoajaa, 0 dollaria ikuisesti):**
+### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+
+| Model | Prefix | Rate Limit | Provider Behind |
+| ---------- | ------ | ---------- | ------------------ |
+| `openai` | `pol/` | 1 req/15s | GPT-5 |
+| `claude` | `pol/` | 1 req/15s | Anthropic Claude |
+| `gemini` | `pol/` | 1 req/15s | Google Gemini |
+| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 |
+| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout |
+| `mistral` | `pol/` | 1 req/15s | Mistral AI |
+
+> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately.
+
+### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+
+| Tier | Daily Neurons | Equivalent Usage | Notes |
+| ---- | ------------- | --------------------------------------- | ----------------------- |
+| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models |
+
+Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct`
+
+> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings.
+
+### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+
+| Tier | Free Quota | Location | Notes |
+| ---- | ------------- | ------------ | ----------------------------------- |
+| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits |
+
+Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
+
+> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com).
+
+> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):**
>
> ```
-> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
-> Qoder (if/) → kimi-k2-ajattelu, qwen3-coder-plus, deepseek-r1 UNLIMITED
-> LongCat Lite (lc/) → LongCat-Flash-Lite – 50 miljoonaa rahaketta/päivä 🔥
-> Pölytys (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — avainta ei tarvita
-> Qwen (qw/) → qwen3-kooderimallit RAJOITTAMATTA
-> Gemini (gemini/) → Gemini 2.5 Flash – 1 500 rekv/päivä ilmaiseksi
-> Cloudflare AI (vrt./) → 50+ mallia – 10 000 neuronia/päivä
-> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 miljoona ilmaista rahaketta (EU)
-> Groq (groq/) → Llama/Gemma – 14,4 000 req/päivä erittäin nopea
-> NVIDIA NIM (nvidia/) → 70+ avointa mallia – 40 RPM ikuisesti
-> Aivot (cerebras/) → Llama/Qwen maailman nopein – 1 milj. tok/päivä
-> ```## 🎙️ Free Transcription Combo
+> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+> Qwen (qw/) → qwen3-coder models UNLIMITED
+> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free
+> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day
+> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast
+> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+> ```
-> Literoi mikä tahansa ääni/video hintaan**$0**— Deepgram-johdot 200 dollarilla ilmaiseksi, AssemblyAI 50 dollarin varavara, Groq Whisper rajoittamattomana hätävarmuuskopiona.
+## 🎙️ Free Transcription Combo
-| Palveluntarjoaja | Ilmaisia luottoja | Paras malli | Hintarajoitus |
-| ------------------ | ----------------------- | --------------------------------------------- | ----------------------------- |
-| 🟢**Deepgram**|**200 dollaria ilmaiseksi**(kirjautuminen) | `nova-3` — paras tarkkuus, yli 30 kieltä | Ei RPM-rajoitusta ilmaisille luottoille |
-| 🔵**AssemblyAI**|**50 dollaria ilmaiseksi**(kirjautuminen) | "universal-3-pro" — luvut, tunnelma, henkilötiedot | Ei RPM-rajoitusta ilmaisille luottoille |
-| 🔴**Groq**|**Ilmainen ikuisesti**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (nopeus rajoitettu) |
+> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup.
-**Ehdotettu yhdistelmä kohdassa `/dashboard/combos':**```
+| Provider | Free Credits | Best Model | Rate Limit |
+| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- |
+| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits |
+| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits |
+| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) |
+
+**Suggested combo in `/dashboard/combos`:**
+
+```
Name: free-transcription
Strategy: Priority
Nodes:
[1] deepgram/nova-3 → uses $200 free first
[2] assemblyai/universal-3-pro → fallback when Deepgram credits run out
[3] groq/whisper-large-v3 → free forever, emergency fallback
-````
+```
-Sitten `/dashboard/media` →**Transkriptio**-välilehti: lataa mikä tahansa ääni- tai videotiedosto → valitse yhdistelmäpäätepiste → hanki transkriptio tuetuissa muodoissa.## 💡 Key Features
+Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats.
-OmniRoute v2.0 on rakennettu toiminnalliseksi alustaksi, ei vain välityspalvelimeksi.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
+## 💡 Key Features
-| Ominaisuus | Mitä se tekee |
-| ------------------------------------------------ | ----------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ |
-| ⚡**Grok-4 Fast Family** | xAI-mallit hintaan 0,20 $/0,50 $/M – vertailuarvo 1143 ms (30 % nopeampi kuin Gemini 2.5 Flash) |
-| 🧠**GLM-5 Z.AI:n kautta** | 128 000 tuloskonteksti, 0,5 $/1 milj. GLM-perheen uusin lippulaiva |
-| 🔮**MiniMax M2.5** | Päättely + agenttitehtävät hintaan 0,30 $/1M – merkittävä parannus M2.1:stä |
-| 🎯**ToolCalling Flag mallikohtainen** | Mallikohtainen työkalukutsu: tosi/epätosi rekisterissä — AutoCombo ohittaa mallit, joissa ei ole työkaluja |
-| 🌍**Monikielinen tarkoituksentunnistus** | PT/ZH/ES/AR avainsanat AutoCombo-pisteytyksissä — parempi mallivalinta ei-englanninkieliselle sisällölle |
-| 📊**Vertailuarvoihin perustuvat varaehdotukset** | Todellinen p95-viive live-pyyntöjen syötteiden yhdistelmäpisteistä — AutoCombo oppii todellisista tiedoista |
-| 🔁**Pyydä päällekkäisyyden poistamista** | Sisältö-hash-pohjainen dedup-ikkuna — turvallinen usealle agentille, estää päällekkäiset veloitukset |
-| 🔌**Pluggable RouterStrategy** | Laajentuva RouterStrategy-käyttöliittymä — lisää mukautettu reitityslogiikka laajennuksiksi | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP |
+OmniRoute v3.5 is built as an operational platform, not just a relay proxy.
-| Ominaisuus | Mitä se tekee |
-| ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |
-| 🎮**Model Playground** | Dashboard-sivu minkä tahansa mallin testaamiseen suoraan – palveluntarjoajan/mallin/päätepisteen valitsimet, Monaco Editor, suoratoisto, keskeytys, ajoitus |
-| 🔏**CLI-sormenjälkien vastaavuus** | Palveluntarjoajakohtainen otsikko/runkojärjestys vastaamaan alkuperäisiä CLI-allekirjoituksia – vaihda palveluntarjoajan mukaan kohdassa Asetukset > Suojaus.**Välipalvelimesi IP-osoite säilyy** |
-| 🤝**ACP-tuki (Agent Client Protocol)** | CLI-agentin etsintä (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 muuta), prosessin synnyttäjä, /api/acp/agents-päätepiste |
-| 🤖**ACP Agents Dashboard** | Vianetsintä › Agentit -sivu — 14 agentin ruudukko, jossa on asennustila, versio ja mukautettu agenttilomake mille tahansa CLI-työkalulle.**OpenCode**-käyttäjät saavat "Download opencode.json" -painikkeen, joka luo automaattisesti käyttövalmiin kokoonpanon kaikille saatavilla oleville malleille. |
-| 🔧**Muokatun mallin `apiFormat` -reititys** | Mukautetut mallit, joissa on `apiFormat: "responses"`, ohjaavat nyt oikein Responses API -kääntäjään |
-| 🏢**Codex Workspace Isolation** | Useita Codex-työtiloja sähköpostissa — OAuth erottaa yhteydet oikein työtilan tunnuksen |
-| 🔄**Automaattinen elektroninen päivitys** | Työpöytäsovellus tarkistaa päivitykset + automaattinen asennus uudelleenkäynnistyksen yhteydessä | ### 🤖 Agent & Protocol Operations (v2.0) |
+### 🆕 New — v3.5.5 Highlights (Apr 2026)
-| Ominaisuus | Mitä se tekee |
-| ------------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- |
-| 🔧**MCP-palvelin (25 työkalua)** | IDE/agenttityökalut kolmen kuljetuksen kautta: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 ydintä + 3 muistia + 4 taitotyökalua |
-| 🤝**A2A-palvelin (JSON-RPC + SSE)** | Agenttien välinen tehtävien suorittaminen synkronointi- ja suoratoistovirroilla |
-| 🧭**Konsolidoitu päätepistesivu** | Välilehdillä varustettu hallintasivu Endpoint Proxy-, MCP-, A2A- ja API Endpoints -välilehdillä |
-| 🎚️**Palvelun käyttöönoton/poistamisen valinnat** | ON/OFF kytkimet MCP:lle ja A2A:lle ja asetusten pysyvyys (oletus: OFF) |
-| 🛰️**MCP Runtime Heartbeat** | Todellinen prosessin tila (pid, käytettävyys, sykeikä, kuljetus, mittaustila) |
-| 📋**MCP Audit Trail** | Suodatettavat tarkastuslokit onnistumis/epäonnistuminen ja avainattribuutio |
-| 🔐**MCP Scope Enforcement** | 10 yksityiskohtaista käyttöoikeutta työkalujen hallitukselle |
-| 📡**A2A-tehtävän elinkaaren hallinta** | Listaa/suodata tehtäviä, tarkasta tapahtumat/artefaktit, peruuta käynnissä olevat tehtävät |
-| 📋**Agenttikortin löytäminen** | `/.well-known/agent.json` asiakkaan automaattiseen löytämiseen |
-| 🧪**Protokollan E2E-testivaljaat** | Todellinen MCP SDK + A2A-asiakas kulkee muodossa "test:protocols:e2e" |
-| ⚙️**Toimintaohjaimet** | Vaihda yhdistelmä, käytä kimmoisuusprofiileja, nollaa katkaisijat yhdeltä ohjauspinnalta | ### 🧠 Routing & Intelligence |
+| Feature | What It Does |
+| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation |
+| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs |
+| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions |
+| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection |
+| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts |
-| Ominaisuus | Mitä se tekee |
-| ------------------------------------ | --------------------------------------------------------------------------------------- | ----------------------- |
-| 🎯**Smart 4-Tier Fallback** | Automaattinen reitti: Tilaus → API-avain → Halpa → Ilmainen |
-| 📊**Reaaliaikainen kiintiöseuranta** | Live-tunnusten määrä + nollaa lähtölaskenta palveluntarjoajaa kohti |
-| 🔄**Käännösmuoto** | OpenAI ↔ Claude ↔ Gemini ↔ Vastaukset skeematurvallisilla muunnoksilla |
-| 👥**Useiden tilien tuki** | Useita tilejä per palveluntarjoaja älykkäällä valinnalla |
-| 🔄**Automaattinen Token Refresh** | OAuth-tunnukset päivittyvät automaattisesti yrittämällä uudelleen |
-| 🎨**Muokatut yhdistelmät** | 9 tasapainotusstrategiaa + varaketjun ohjaus |
-| 🌐**Wildcard-reititin** | `provider/*` dynaaminen reititys |
-| 🧠**Ajatteleva budjettihallinta** | Läpivienti-, automaatti-, mukautetut ja mukautuvat päättelyrajat |
-| 🔀**Mallialiakset** | Sisäänrakennettu + mukautetun mallin alias ja siirtoturva |
-| ⚡**Taustan heikkeneminen** | Ohjaa matalan prioriteetin taustatehtävät halvempiin malleihin |
-| 🧪**Task-Aware Smart Routing** | Automaattinen mallin valinta sisältötyypin mukaan (koodaus/näkemys/analyysi/yhteenveto) |
-| 🔄**A2A-agenttityönkulut** | Deterministinen FSM-organisaattori tilallisiin monivaiheisiin agenttien suorituksiin |
-| 🔀**Adaptiivinen reititys** | Dynaaminen strategian ohitus tunnuksen määrän ja nopean monimutkaisuuden perusteella |
-| 🎲**Tarjoajien monimuotoisuus** | Shannonin entropiapisteytys tasapainottava automaattinen yhdistelmäliikenteen jakelu |
-| 💬**Järjestelmän pikaruiskutus** | Globaalia käyttäytymisen valvontaa sovelletaan johdonmukaisesti |
-| 📄**Responses API -yhteensopivuus** | Täysi "/v1/responses" tuki Codexille ja edistyneille agenttityönkuluille | ### 🎵 Multi-Modal APIs |
+### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
-| Ominaisuus | Mitä se tekee |
-| ------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- |
-| 🖼️**Kuvan luominen** | `/v1/images/generations' pilvipalveluilla ja paikallisilla taustaohjelmilla |
-| 📐**Upotukset** | `/v1/embeddings' haku- ja RAG-putkistoja varten |
-| 🎤**Äänitranskriptio** | `/v1/audio/transcriptions' — 7 palveluntarjoajaa (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), automaattinen kielentunnistus, MP4/MP3/WAV-tuki |
-| 🔊**Tekstistä puheeksi** | `/v1/audio/speech` – 10 palveluntarjoajaa (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) oikeilla virheilmoituksilla |
-| 🎬**Videon sukupolvi** | `/v1/videos/generations' (ComfyUI + SD WebUI -työnkulut) |
-| 🎵**Musiikin sukupolvi** | `/v1/music/generations' (ComfyUI-työnkulut) |
-| 🛡️**Moderaatiot** | "/v1/moderations" turvallisuustarkastukset |
-| 🔀**Uudelleenjärjestys** | `/v1/uudelleensijoitus' osuvuuden arvioimiseksi |
-| 🔍**Verkkohaku**🆕 | "/v1/search" – 5 palveluntarjoajaa (Serper, Brave, Perplexity, Exa, Tavily), 6 500+ ilmaista kuukaudessa, automaattinen vikasieto, välimuisti | ### 🛡️ Resilience, Security & Governance |
+| Feature | What It Does |
+| ------------------------------------ | ------------------------------------------------------------------------------------------- |
+| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) |
+| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family |
+| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 |
+| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models |
+| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content |
+| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data |
+| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges |
+| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins |
-| Ominaisuus | Mitä se tekee |
-| ----------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------- | -------------------------------- |
-| 🔌**Katkaisijat** | Mallikohtainen laukaisu/palautus kynnysohjaimilla |
-| 🎯**Endpoint-Aware mallit** | Mukautetut mallit ilmoittavat tuetut päätepisteet + API-muoto |
-| 🛡️**Ukkosen vastainen lauma** | Mutex + semaforisuojat uudelleenyritys-/nopeustapahtumissa |
-| 🧠**Semanttinen + allekirjoitusvälimuisti** | Kustannusten/viiveen vähentäminen kahdella välimuistikerroksella |
-| ⚡**Pyydä idempotenssia** | Kaksoissuojaikkuna |
-| 🔒**TLS-sormenjälkien huijaus** | Selaimen kaltainen TLS-sormenjälki —**vähentää botin havaitsemista ja tilimerkintöjä** |
-| 🔏**CLI-sormenjälkien vastaavuus** | Vastaa alkuperäisten CLI-pyyntöjen allekirjoituksia —**vähentää eston riskiä säilyttäen samalla välityspalvelimen IP-osoitteen** |
-| 🌐**IP-suodatus** | Salli-/estoluettelon hallinta paljaille käyttöönotuksille |
-| 📊**Muokattavat hintarajat** | Muokattavat globaalit/toimittajatason rajoitukset pysyvillä |
-| 📉**Graceful Degradation** | Monikerroksiset valmiudet, jotka suojaavat ydinyhdyskäytävätoimintoja |
-| 📜**Config Audit Trail** | Diff-pohjainen muutosseuranta, joka estää toiminnan ajautumisen yksinkertaisilla palautuksilla |
-| ⏳**Provider Health Sync** | Ennakoiva tunnuksen vanhenemisen valvonta laukaisee hälytyksiä ennen valtuutusvirheitä |
-| 🚪**Poista kielletyt tilit automaattisesti käytöstä** | Toiminnassa oleva katkaisija sinetöi pysyvästi estettyjen tokentilien automaattisesti |
-| 🔑**API-avainten hallinta + rajaus** | Suojattu avainten myöntäminen/kierto ja mallin/toimittajan hallintalaitteet |
-| 👁️**Scoped API Key Reveal**🆕 | Ota käyttöön API-avainten palautus `ALLOW_API_KEY_REVEAL` |
-| 🛡️**Suojattu `/mallit`** | Valinnainen todennus ja palveluntarjoajan piilottaminen malliluetteloon | ### 📊 Observability & Analytics |
+### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP
-| Ominaisuus | Mitä se tekee |
-| ------------------------------------------ | ----------------------------------------------------------------------- | ---------------------------- |
-| 📝**Pyyntö + välityspalvelimen kirjaus** | Täysi pyyntö/vastaus ja välityspalvelimen kirjaus |
-| 📉**Striimatut yksityiskohtaiset lokit**🆕 | Rekonstruoi SSE-hyötykuormavirrat puhtaasti käyttöliittymään |
-| 📋**Unified Logs Dashboard** | Pyyntö-, välityspalvelin-, tarkastus- ja konsolinäkymät yhdellä sivulla |
-| 🔍**Pyydä telemetriaa** | p50/p95/p99 latenssi ja pyynnön jäljitys |
-| 🏥**Terveyden hallintapaneeli** | Käyttöaika, katkaisutilat, lukitukset, välimuistitilastot |
-| 💰**Kustannusseuranta** | Budjetin hallinta ja mallikohtainen hinnoittelun näkyvyys |
-| 📈**Analytiikan visualisoinnit** | Mallin/palveluntarjoajan käyttötiedot ja trendinäkymät |
-| 🧪**Arviointikehys** | Golden set -testaus konfiguroitavilla ottelustrategioilla |
-| 📡**Live Diagnostics**🆕 | Semanttisen välimuistin ohitus tarkkaan yhdistelmätestaukseen livenä | ### ☁️ Deployment & Platform |
+| Feature | What It Does |
+| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing |
+| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** |
+| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint |
+| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. |
+| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator |
+| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID |
+| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart |
-| Ominaisuus | Mitä se tekee |
-| ---------------------------------- | ----------------------------------------------------------------------------------------------- | --------------------- |
-| 🌐**Ota käyttöön missä tahansa** | Localhost, VPS, Docker, pilviympäristöt |
-| 🚇**Cloudflare-tunneli**🆕 | Yhden napsautuksen Quick Tunnel -integrointi kojelaudalta |
-| 🔑**API-avainmallin suodatus** | Alkuperäinen /v1/models-vastaus suodatettu määritettyjen verkkopalvelukontekstin roolien kautta |
-| ⚡**Smart Cache Bypass** | Konfiguroitava TTL-heuristiikka ja pakotetut palautusohjaimet |
-| 🔄**Varmuuskopioi/Palauta** | Vienti/tuonti ja katastrofien palautusvirrat |
-| 🧙**Ohjattu käyttöönottotoiminto** | Ensimmäisen kerran ohjattu asennus |
-| 🔧**CLI Tools Dashboard** | Asennus yhdellä napsautuksella suosittuja koodaustyökaluja varten |
-| 🎮**Model Playground** | Testaa mitä tahansa palveluntarjoajaa/mallia/päätepistettä hallintapaneelista |
-| 🔏**CLI-sormenjälkivalitsin** | Palveluntarjoajakohtainen sormenjälkien vastaavuus kohdassa Asetukset > Suojaus |
-| 🌐**i18n (30 kieltä)** | Täysi kojelauta + asiakirjojen kielen tuki RTL-kattauksella |
-| 🧹**Tyhjennä kaikki mallit** | Yhden napsautuksen malliluettelon tyhjennys toimittajan tiedoissa |
-| 👁️**Sivupalkin säätimet**🆕 | Piilota komponentit ja integraatiot ulkoasuasetuksista |
-| 📋**Ongelman mallit** | Standardoidut GitHub-mallit bugeille ja ominaisuuksille |
-| 📂**Muokattu tietohakemisto** | Tallennuspaikan DATA_DIR-ohitus | ### Feature Deep Dive |
+### 🤖 Agent & Protocol Operations (v2.0)
+
+| Feature | What It Does |
+| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
+| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools |
+| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows |
+| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs |
+| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) |
+| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) |
+| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution |
+| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access |
+| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks |
+| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery |
+| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` |
+| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface |
+
+### 🧠 Routing & Intelligence
+
+| Feature | What It Does |
+| ---------------------------------- | ------------------------------------------------------------------------ |
+| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free |
+| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider |
+| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions |
+| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection |
+| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry |
+| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control |
+| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session |
+| 🌐 **Wildcard Router** | `provider/*` dynamic routing |
+| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits |
+| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety |
+| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models |
+| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) |
+| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions |
+| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity |
+| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution |
+| 💬 **System Prompt Injection** | Global behavior controls applied consistently |
+| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows |
+
+### 🎵 Multi-Modal APIs
+
+| Feature | What It Does |
+| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends |
+| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines |
+| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support |
+| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages |
+| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) |
+| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) |
+| 🛡️ **Moderations** | `/v1/moderations` safety checks |
+| 🔀 **Reranking** | `/v1/rerank` for relevance scoring |
+| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache |
+
+### 🛡️ Resilience, Security & Governance
+
+| Feature | What It Does |
+| ----------------------------------- | -------------------------------------------------------------------------------------- |
+| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls |
+| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format |
+| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events |
+| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers |
+| ⚡ **Request Idempotency** | Duplicate protection window |
+| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** |
+| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** |
+| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments |
+| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence |
+| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations |
+| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks |
+| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures |
+| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically |
+| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls |
+| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` |
+| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog |
+
+### 📊 Observability & Analytics
+
+| Feature | What It Does |
+| -------------------------------- | ----------------------------------------------------- |
+| 📝 **Request + Proxy Logging** | Full request/response and proxy logging |
+| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI |
+| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page |
+| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing |
+| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats |
+| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility |
+| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views |
+| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies |
+| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing |
+
+### ☁️ Deployment & Platform
+
+| Feature | What It Does |
+| ------------------------------ | --------------------------------------------------------------------- |
+| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments |
+| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard |
+| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles |
+| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls |
+| 🔄 **Backup/Restore** | Export/import and disaster recovery flows |
+| 🧙 **Onboarding Wizard** | First-run guided setup |
+| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools |
+| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard |
+| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security |
+| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage |
+| 🧹 **Clear All Models** | One-click model list clearing in provider details |
+| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings |
+| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features |
+| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location |
+
+### Feature Deep Dive
#### Smart fallback with practical cost control
@@ -1292,103 +1464,132 @@ Combo: "my-coding-stack"
4. if/kimi-k2-thinking
```
-Kun kiintiö, korko tai kunto epäonnistuu, OmniRoute siirtyy automaattisesti seuraavaan ehdokkaaseen ilman manuaalista vaihtoa.#### Protocol management that is visible and operable
+When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching.
-- MCP + A2A ovat löydettävissä käyttöliittymässä ja asiakirjoissa (ei piilotettu)
-- Protokollan tilasovellusliittymät paljastavat reaaliaikaiset toimintatiedot (`/api/mcp/*`, `/api/a2a/*`)
-- Hallintapaneelit sisältävät toimintoja 2. päivän toimintoihin (kombinvaihto, katkaisijan nollaukset, tehtävien peruutus)#### Translator + validation workflow
+#### Protocol management that is visible and operable
-Kääntäjä-alue sisältää:
+- MCP + A2A are discoverable in UI and docs (not hidden)
+- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`)
+- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation)
--**Playground**: pyydä muunnostarkistuksia -**Chat Tester**: täydellinen pyyntö/vastaus edestakaisin -**Testipenkki**: useita tapauksia yhdellä kertaa -**Live Monitor**: reaaliaikainen liikennenäkymä
+#### Translator + validation workflow
-Plus protokollan validointi oikeiden asiakkaiden kanssa komennolla "npm run test:protocols:e2e".
+The Translator area includes:
-> 📖**[MCP-palvelimen README](open-sse/mcp-server/README.md)**— työkaluviittaus, IDE-määritykset ja asiakasesimerkit
+- **Playground**: request transformation checks
+- **Chat Tester**: full request/response round-trip
+- **Test Bench**: multiple cases in one run
+- **Live Monitor**: real-time traffic view
+
+Plus protocol validation with real clients via `npm run test:protocols:e2e`.
+
+> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples
>
-> 📖**[A2A-palvelimen README](src/lib/a2a/README.md)**— Taidot, JSON-RPC-menetelmät, suoratoisto ja tehtävien elinkaari## 🧪 Evaluations (Evals)
+> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle
-OmniRoute sisältää sisäänrakennetun arviointikehyksen, jolla testataan LLM-vastauksen laatua kultaiseen joukkoon verrattuna. Käytä sitä kojelaudan kohdassa**Analytics → Evals**.### Built-in Golden Set
+## 🧪 Evaluations (Evals)
-Esiladattu "OmniRoute Golden Set" sisältää testitapauksia:
+OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard.
-- Tervehdys, matematiikka, maantiede, koodin luominen
-- JSON-muodon yhteensopivuus, käännös, hinnanalennusten luominen
-- Turvallisuuskielto (haitallinen sisältö), laskenta, boolen logiikka### Evaluation Strategies
+### Built-in Golden Set
-| Strategia | Kuvaus | Esimerkki |
-| ---------- | ------------------------------------------------------------------------ | ------------------------------- | --- |
-| "tarkka" | Tulosten on vastattava tarkasti | "4" |
-| "sisältää" | Tulosteen tulee sisältää alimerkkijono (kirjainkoolla ei ole merkitystä) | `"Pariisi"` |
-| "regulex" | Tulostuksen on vastattava regex-mallia | "1.*2.*3" |
-| "muokattu" | Mukautettu JS-funktio palauttaa true/false | `(lähtö) => output.length > 10` | --- |
+The pre-loaded "OmniRoute Golden Set" contains test cases for:
+
+- Greetings, math, geography, code generation
+- JSON format compliance, translation, markdown generation
+- Safety refusal (harmful content), counting, boolean logic
+
+### Evaluation Strategies
+
+| Strategy | Description | Example |
+| ---------- | ------------------------------------------------ | -------------------------------- |
+| `exact` | Output must match exactly | `"4"` |
+| `contains` | Output must contain substring (case-insensitive) | `"Paris"` |
+| `regex` | Output must match regex pattern | `"1.*2.*3"` |
+| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` |
+
+---
## 📖 Setup Guide
### Protocol Setup (MCP + A2A)
-
-🧩 MCP-asetukset (mallikontekstiprotokolla)
+
+🧩 MCP Setup (Model Context Protocol)
-Aloita MCP-siirto stdio-tilassa:```bash
+Start MCP transport in stdio mode:
+
+```bash
omniroute --mcp
+```
-````
+Recommended validation flow:
-Suositeltu vahvistuskulku:
+1. Connect your MCP client over stdio.
+2. Run `omniroute_get_health`.
+3. Run `omniroute_list_combos`.
+4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit.
-1. Yhdistä MCP-asiakas stdion kautta.
-2. Suorita "omniroute_get_health".
-3. Suorita "omniroute_list_combos".
-4. Avaa `/dashboard/mcp' vahvistaaksesi syke, toiminta ja tarkastus.
+Useful APIs for automation:
-Hyödyllisiä sovellusliittymiä automatisointiin:
+- `GET /api/mcp/status`
+- `GET /api/mcp/tools`
+- `GET /api/mcp/audit`
+- `GET /api/mcp/audit/stats`
-- "GET /api/mcp/status".
-- "GET /api/mcp/tools".
-- "GET /api/mcp/audit".
-- "GET /api/mcp/audit/stats".
+
-
-🤝 A2A-asetukset (Agent2Agent)
+
+🤝 A2A Setup (Agent2Agent)
-Tutustu agenttiin:```bash
+Discover the agent:
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
-Lähetä tehtävä:```bash
+Send a task:
+
+```bash
curl -X POST http://localhost:20128/a2a \
- -H 'content-type: application/json' \
- -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+ -H 'content-type: application/json' \
+ -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+```
-````
+Manage lifecycle:
-Hallitse elinkaarta:
+- `GET /api/a2a/status`
+- `GET /api/a2a/tasks`
+- `GET /api/a2a/tasks/:id`
+- `POST /api/a2a/tasks/:id/cancel`
-- "GET /api/a2a/status".
-- "GET /api/a2a/tasks".
-- "GET /api/a2a/tasks/:id".
-- POST /api/a2a/tasks/:id/cancel
+Operational UI:
-Käyttöliittymä:
+- `/dashboard/a2a` for task/state/stream observability and smoke actions
-- `/dashboard/a2a` tehtävän/tilan/virran havainnointia ja savutoimintoja varten
+
-
-🧪 Päästä päähän -protokollan validointi
+
+🧪 End-to-end protocol validation
-Vahvista molemmat protokollat oikeilla asiakkailla:```bash
+Validate both protocols with real clients:
+
+```bash
npm run test:protocols:e2e
-````
+```
-Tämä varmistaa:
+This verifies:
-- MCP SDK -asiakas yhdistä/luettelo/soita
-- A2A Discovery/Send/stream/get/cancel
-- Tarkista tiedot MCP-tarkastuksessa ja A2A-tehtävienhallinnan sovellusliittymissä
+- MCP SDK client connect/list/call
+- A2A discovery/send/stream/get/cancel
+- Cross-check data in MCP audit and A2A task management APIs
-
-💳 Tilauspalveluntarjoajat### Claude Code (Pro/Max)
+
+
+
+💳 Subscription Providers
+
+### Claude Code (Pro/Max)
```bash
Dashboard → Providers → Connect Claude Code
@@ -1401,7 +1602,9 @@ Models:
cc/claude-haiku-4-5-20251001
```
-**Provinkki:**Käytä Opusta monimutkaisiin tehtäviin ja Sonnetia nopeutta varten. OmniRoute jäljityskiintiö mallia kohti!### OpenAI Codex (Plus/Pro)
+**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model!
+
+### OpenAI Codex (Plus/Pro)
```bash
Dashboard → Providers → Connect Codex
@@ -1415,20 +1618,22 @@ Models:
#### Codex Account Limit Management (5h + Weekly)
-Jokaisella Codex-tilillä on nyt käytäntökytkimet kohdassa "Dashboard -> Providers":
+Each Codex account now has policy toggles in `Dashboard -> Providers`:
-- "5h" (ON/OFF): pakottaa 5 tunnin ikkunan kynnyskäytäntö.
-- "Viikoittain" (ON/OFF): pakottaa viikoittaisen ikkunan kynnyskäytäntöä.
-- Kynnyskäyttäytyminen: kun käytössä oleva ikkuna saavuttaa >=90 % käytön, tili ohitetaan.
-- Kiertokäyttäytyminen: OmniRoute reitittää automaattisesti seuraavalle kelvolliselle Codex-tilille.
-- Nollauskäyttäytyminen: kun palveluntarjoajan "resetAt" aika kuluu, tili tulee uudelleen kelpoiseksi automaattisesti.
+- `5h` (ON/OFF): enforce the 5-hour window threshold policy.
+- `Weekly` (ON/OFF): enforce the weekly window threshold policy.
+- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped.
+- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically.
+- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically.
-Skenaariot:
+Scenarios:
-- "5h PÄÄLLÄ" + "Viikoittain PÄÄLLÄ": tili ohitetaan, kun jompikumpi ikkuna saavuttaa kynnyksen.
-- "5h OFF" + "Weekly ON": vain viikoittainen käyttö voi estää tilin.
-- "5h ON" + "Weekly OFF": vain 5 tunnin käyttö voi estää tilin.
-- `resetAt` hyväksytty: tili siirtyy uudelleen kiertoon automaattisesti (ei manuaalista uudelleenkäyttöä).### Gemini CLI (FREE 180K/month!)
+- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold.
+- `5h OFF` + `Weekly ON`: only weekly usage can block the account.
+- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account.
+- `resetAt` passed: account re-enters rotation automatically (no manual re-enable).
+
+### Gemini CLI (FREE 180K/month!)
```bash
Dashboard → Providers → Connect Gemini CLI
@@ -1440,7 +1645,9 @@ Models:
gc/gemini-2.5-pro
```
-**Paras hinta-laatusuhde:**Valtava ilmainen taso! Käytä tätä ennen maksettuja tasoja.### GitHub Copilot
+**Best Value:** Huge free tier! Use this before paid tiers.
+
+### GitHub Copilot
```bash
Dashboard → Providers → Connect GitHub
@@ -1455,71 +1662,91 @@ Models:
-
-🔑 API-avaintoimittajat### NVIDIA NIM (FREE developer access — 70+ models)
+
+🔑 API Key Providers
-1. Rekisteröidy: [build.nvidia.com](https://build.nvidia.com)
-2. Hanki ilmainen API-avain (sisältää 1000 johtopäätöskrediittiä)
-3. Kojelauta → Lisää toimittaja → NVIDIA NIM:
- - API-avain: "nvapi-your-key".
+### NVIDIA NIM (FREE developer access — 70+ models)
-**Mallit:**"nvidia/llama-3.3-70b-instruct", "nvidia/mistral-7b-instruct" ja yli 50 muuta
+1. Sign up: [build.nvidia.com](https://build.nvidia.com)
+2. Get free API key (1000 inference credits included)
+3. Dashboard → Add Provider → NVIDIA NIM:
+ - API Key: `nvapi-your-key`
-**Provinkki:**OpenAI-yhteensopiva API – toimii saumattomasti OmniRouten muotokäännöksen kanssa!### DeepSeek
+**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more
-1. Rekisteröidy: [platform.deepseek.com](https://platform.deepseek.com)
-2. Hanki API-avain
+**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation!
+
+### DeepSeek
+
+1. Sign up: [platform.deepseek.com](https://platform.deepseek.com)
+2. Get API key
3. Dashboard → Add Provider → DeepSeek
-**Mallit:**"deepseek/deepseek-chat", "deepseek/deepseek-coder"### Groq (Free Tier Available!)
+**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder`
-1. Rekisteröidy: [console.groq.com](https://console.groq.com)
-2. Hanki API-avain (ilmainen taso mukana)
+### Groq (Free Tier Available!)
+
+1. Sign up: [console.groq.com](https://console.groq.com)
+2. Get API key (free tier included)
3. Dashboard → Add Provider → Groq
-**Mallit:**"groq/llama-3.3-70b", "groq/mixtral-8x7b"
+**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b`
-**Provinkki:**Äärimmäisen nopea johtopäätös – paras reaaliaikaiseen koodaukseen!### OpenRouter (100+ Models)
+**Pro Tip:** Ultra-fast inference — best for real-time coding!
-1. Rekisteröidy: [openrouter.ai](https://openrouter.ai)
-2. Hanki API-avain
+### OpenRouter (100+ Models)
+
+1. Sign up: [openrouter.ai](https://openrouter.ai)
+2. Get API key
3. Dashboard → Add Provider → OpenRouter
-**Mallit:**Käytä yli 100 mallia kaikilta tärkeimmiltä palveluntarjoajilta yhdellä API-avaimella.
+**Models:** Access 100+ models from all major providers through a single API key.
-**Kojelaudan toiminta:**OpenRouter-malleja hallitaan**Saatavilla olevista malleista**. Manuaalinen lisääminen, tuonti ja automaattinen synkronointi päivittävät kaikki saman luettelon.
+**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list.
-
-💰 Halvat palveluntarjoajat (Varmuuskopio)### GLM-4.7 (Daily reset, $0.6/1M)
+
-1. Rekisteröidy: [Zhipu AI](https://open.bigmodel.cn/)
-2. Hanki API-avain Coding Planista
-3. Hallintapaneeli → Lisää API-avain:
- - Palveluntarjoaja: "glm".
- - API-avain: "oma-avain".
+
+💰 Cheap Providers (Backup)
-**Käytä:**`glm/glm-4.7`
+### GLM-4.7 (Daily reset, $0.6/1M)
-**Provinkki:**Koodaussuunnitelma tarjoaa 3-kertaisen kiintiön 1/7 hinnalla! Nollaa päivittäin klo 10.00.### MiniMax M2.1 (5h reset, $0.20/1M)
+1. Sign up: [Zhipu AI](https://open.bigmodel.cn/)
+2. Get API key from Coding Plan
+3. Dashboard → Add API Key:
+ - Provider: `glm`
+ - API Key: `your-key`
-1. Rekisteröidy: [MiniMax](https://www.minimax.io/)
-2. Hanki API-avain
-3. Kojelauta → Lisää API-avain
+**Use:** `glm/glm-4.7`
-**Käytä:**`minimax/MiniMax-M2.1`
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM.
-**Ammattilaisen vinkki:**Halvin vaihtoehto pitkälle kontekstille (1 milj. merkkiä)!### Kimi K2 ($9/month flat)
+### MiniMax M2.1 (5h reset, $0.20/1M)
-1. Tilaa: [Moonshot AI](https://platform.moonshot.ai/)
-2. Hanki API-avain
-3. Kojelauta → Lisää API-avain
+1. Sign up: [MiniMax](https://www.minimax.io/)
+2. Get API key
+3. Dashboard → Add API Key
-**Käytä:**`kimi/kimi-latest`
+**Use:** `minimax/MiniMax-M2.1`
-**Ammattilaisen vinkki:**Kiinteä 9 dollaria kuukaudessa 10 miljoonalle tokenille = 0,90 dollaria / 1 miljoona todellista hintaa!
+**Pro Tip:** Cheapest option for long context (1M tokens)!
-
-🆓 ILMAISIA palveluntarjoajia (hätävarmuuskopiointi)### Qoder (5 FREE models via OAuth)
+### Kimi K2 ($9/month flat)
+
+1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `kimi/kimi-latest`
+
+**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost!
+
+
+
+
+🆓 FREE Providers (Emergency Backup)
+
+### Qoder (5 FREE models via OAuth)
```bash
Dashboard → Connect Qoder
@@ -1560,8 +1787,10 @@ Models:
-
-🎨 Luo komboja### Example 1: Maximize Subscription → Cheap Backup
+
+🎨 Create Combos
+
+### Example 1: Maximize Subscription → Cheap Backup
```
Dashboard → Combos → Create New
@@ -1589,8 +1818,10 @@ Cost: $0 forever!
-
-🔧 CLI-integrointi### Cursor IDE
+
+🔧 CLI Integration
+
+### Cursor IDE
```
Settings → Models → Advanced:
@@ -1601,7 +1832,9 @@ Settings → Models → Advanced:
### Claude Code
-Käytä kojelaudan**CLI Tools**-sivua määritysten tekemiseen yhdellä napsautuksella tai muokkaa `~/.claude/settings.json` manuaalisesti.### Codex CLI
+Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually.
+
+### Codex CLI
```bash
export OPENAI_BASE_URL="http://localhost:20128"
@@ -1612,12 +1845,15 @@ codex "your prompt"
### OpenClaw
-**Vaihtoehto 1 – hallintapaneeli (suositus):**```
+**Option 1 — Dashboard (recommended):**
+
+```
Dashboard → CLI Tools → OpenClaw → Select Model → Apply
+```
-````
+**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`:
-**Vaihtoehto 2 – Manuaalinen:**Muokkaa `~/.openclaw/openclaw.json`:```json
+```json
{
"models": {
"providers": {
@@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply
}
}
}
-````
+```
-> **Huomaa:**OpenClaw toimii vain paikallisen OmniRouten kanssa. Käytä "127.0.0.1" "localhost" sijaan IPv6-resoluutioongelmien välttämiseksi.### Cline / Continue / RooCode
+> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues.
+
+### Cline / Continue / RooCode
```
Settings → API Configuration:
@@ -1643,15 +1881,17 @@ Settings → API Configuration:
### OpenCode
-**Vaihe 1:**Lisää OmniRoute mukautetuksi palveluntarjoajaksi:```bash
+**Step 1:** Add OmniRoute as a custom provider:
+
+```bash
opencode
/connect
-
# Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key
+```
-````
+**Step 2:** Create/edit `opencode.json` in your project root:
-**Vaihe 2:**Luo/muokkaa `opencode.json` projektisi juuressa:```json
+```json
{
"$schema": "https://opencode.ai/config.json",
"provider": {
@@ -1669,117 +1909,130 @@ opencode
}
}
}
-````
+```
-**Vaihe 3:**Valitse malli OpenCodessa:```bash
+**Step 3:** Select the model in OpenCode:
+
+```bash
/models
-
# Select any OmniRoute model from the list
+```
-````
+> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard.
->**Vinkki:**Lisää mikä tahansa malli, joka on saatavilla OmniRoute `/v1/models' -päätepisteessäsi "mallit"-osioon. Käytä muotoa "provider/model-id" OmniRoute-hallintapaneelista.
+
---
## Vianmääritys
-
-Laajenna vianetsintäopas napsauttamalla
+
+Click to expand troubleshooting guide
-**"Kielimalli ei antanut viestejä"**
+**"Language model did not provide messages"**
-- Palveluntarjoajan kiintiö käytetty loppuun → Tarkista kojelaudan kiintiön seuranta
-- Ratkaisu: Käytä yhdistelmävaraa tai vaihda halvempaan tasoon
+- Provider quota exhausted → Check dashboard quota tracker
+- Solution: Use combo fallback or switch to cheaper tier
-**hintarajoitus**
+**Rate limiting**
-- Tilauskiintiö loppu → Varaa GLM/MiniMaxiin
-- Lisää yhdistelmä: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Subscription quota out → Fallback to GLM/MiniMax
+- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-**OAuth-tunnus vanhentunut**
+**OAuth token expired**
-- OmniRoute päivittää automaattisesti
-- Jos ongelmat jatkuvat: Kojelauta → Palveluntarjoaja → Yhdistä uudelleen
+- Auto-refreshed by OmniRoute
+- If issues persist: Dashboard → Provider → Reconnect
-**Korkeat kustannukset**
+**High costs**
-- Tarkista käyttötilastot kohdassa Dashboard → Costs
-- Vaihda ensisijaiseksi malliksi GLM/MiniMax
-- Käytä ilmaista tasoa (Gemini CLI, Qoder) ei-kriittisiin tehtäviin
+- Check usage stats in Dashboard → Costs
+- Switch primary model to GLM/MiniMax
+- Use free tier (Gemini CLI, Qoder) for non-critical tasks
-**Kojelauta/API-portit ovat väärin**
+**Dashboard/API ports are wrong**
-- "PORT" on kanoninen perusportti (ja oletuksena API-portti)
-- "API_PORT" ohittaa vain OpenAI-yhteensopivan API-kuuntelijan
-- `DASHBOARD_PORT' ohittaa vain kojelaudan/Next.js-kuuntelijan
-- Aseta "NEXT_PUBLIC_BASE_URL" kojelaudaksi/julkiseksi URL-osoitteeksi (OAuth-takaisinsoittoja varten)
+- `PORT` is the canonical base port (and API port by default)
+- `API_PORT` overrides only OpenAI-compatible API listener
+- `DASHBOARD_PORT` overrides only dashboard/Next.js listener
+- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks)
-**Pilvisynkronointivirheet**
+**Cloud sync errors**
-- Varmista, että BASE_URL osoittaa käynnissä olevaan esiintymääsi
-- Varmista, että CLOUD_URL osoittaa odotettuun pilvipäätepisteeseen
-- Pidä NEXT_PUBLIC_*-arvot kohdakkain palvelinpuolen arvojen kanssa
+- Verify `BASE_URL` points to your running instance
+- Verify `CLOUD_URL` points to your expected cloud endpoint
+- Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**Ensimmäinen kirjautuminen ei toimi**
+**First login not working**
-- Tarkista .env:stä ALKUPERÄINEN_SALASANA
-- Jos ei ole asetettu, varasalasana on "123456".
+- Check `INITIAL_PASSWORD` in `.env`
+- If unset, fallback password is `123456`
-**Ei pyyntölokeja**
+**No request logs**
-- Pyynnön artefaktit kirjoitetaan hakemistoon DATA_DIR/call_logs/ yhtenä JSON-tiedostona pyyntöä kohden.
-- Ota käyttöön liukuhihnan sieppaus kohdasta Dashboard → Logs → Request Logs, jos tarvitset yksityiskohtaisia vaihekohtaisia hyötykuormia
-- Aseta APP_LOG_TO_FILE=true, jos haluat myös sovelluskonsolin lokit hakemistoon `logs/application/app.log'
-- Säädä APP_LOG_MAX_FILE_SIZE, APP_LOG_RETENTION_DAYS, APP_LOG_MAX_FILES ja CALL_LOG_MAX_ENTRIES tarpeen mukaan
+- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request
+- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads
+- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log`
+- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed
-**Yhteystesti näyttää "Virheellinen" OpenAI-yhteensopiville palveluntarjoajille**
+**Connection test shows "Invalid" for OpenAI-compatible providers**
-- Monet palveluntarjoajat eivät paljasta /mallit-päätepistettä
-- OmniRoute v1.0.6+ sisältää varatarkistuksen chatin loppuunsaattamisen kautta
-- Varmista, että perus-URL sisältää /v1-liitteen### 🔐 OAuth on a Remote Server
+- Many providers don't expose a `/models` endpoint
+- OmniRoute v1.0.6+ includes fallback validation via chat completions
+- Ensure base URL includes `/v1` suffix
+
+### 🔐 OAuth on a Remote Server
->**⚠️ Tärkeää käyttäjille, jotka käyttävät OmniRoutea VPS:ssä, Dockerissa tai millä tahansa etäpalvelimella**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
+> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server**
-**Antigravity**- ja**Gemini CLI**-palveluntarjoajat käyttävät**Google OAuth 2.0**-versiota. Google edellyttää, että OAuth-kulun "redirect_uri" vastaa täsmälleen yhtä sovelluksen Google Cloud Consolessa esirekisteröityistä URI:ista.
+#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
-OmniRouteen niputetut OAuth-tunnistetiedot on rekisteröity**vain `localhost'ille**. Kun käytät OmniRoutea etäpalvelimella (esim. `https://omniroute.myserver.com`), Google hylkää todennuksen seuraavilla tavoilla:```
+The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console.
+
+The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solution: Configure your own OAuth credentials
-Sinun on luotava**OAuth 2.0 -asiakastunnus**Google Cloud Consolessa palvelimesi URI:lla.#### Step-by-step
+You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI.
-**1. Avaa Google Cloud Console**
+#### Step-by-step
-Siirry osoitteeseen: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
+**1. Open Google Cloud Console**
-**2. Luo uusi OAuth 2.0 -asiakastunnus**
+Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-- Napsauta**"+ Luo kirjautumistiedot"**→**"OAuth-asiakastunnus"**
-- Sovellustyyppi:**"Web-sovellus"**
-- Nimi: kaikki mistä pidät (esim. "OmniRoute Remote")
+**2. Create a new OAuth 2.0 Client ID**
+
+- Click **"+ Create Credentials"** → **"OAuth client ID"**
+- Application type: **"Web application"**
+- Name: anything you like (e.g. `OmniRoute Remote`)
**3. Add Authorized Redirect URIs**
-Lisää**"Authorized redirect URIs"**-kenttään:```
+In the **"Authorized redirect URIs"** field, add:
+
+```
https://your-server.com/callback
+```
-````
+> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`).
-> Korvaa "your-server.com" palvelimesi verkkotunnuksella tai IP-osoitteella (lisää tarvittaessa portti, esim. "http://45.33.32.156:20128/callback").
+**4. Save and copy the credentials**
-**4. Tallenna ja kopioi tunnistetiedot**
+After creating, Google will show the **Client ID** and **Client Secret**.
-Luomisen jälkeen Google näyttää**Client ID**ja**Client Secret**.
+**5. Set environment variables**
-**5. Aseta ympäristömuuttujat**
+In your `.env` (or Docker environment variables):
-.env-tiedostossa (tai Docker-ympäristömuuttujat):```bash
+```bash
# For Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
@@ -1788,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
-````
+```
-**6. Käynnistä OmniRoute uudelleen**```bash
+**6. Restart OmniRoute**
+```bash
# npm:
-
npm run dev
# Docker:
-
docker restart omniroute
+```
-````
+**7. Try connecting again**
-**7. Yritä muodostaa yhteys uudelleen**
+Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth
-Dashboard → Providers → Antigravity (tai Gemini CLI) → OAuth
+Google will now redirect correctly to `https://your-server.com/callback`.
-Google ohjaa nyt oikein osoitteeseen https://your-server.com/callback.---
+---
#### Temporary workaround (without custom credentials)
-Jos et halua määrittää omia tunnistetietojasi juuri nyt, voit silti käyttää**manuaalista URL-kulkua**:
+If you don't want to set up your own credentials right now, you can still use the **manual URL flow**:
-1. OmniRoute avaa Googlen valtuutus-URL-osoitteen
-2. Valtuutuksen jälkeen Google yrittää uudelleenohjata palvelimeen "localhost" (joka epäonnistuu etäpalvelimella)
-3.**Kopioi koko URL-osoite**selaimesi osoitepalkista (vaikka sivu ei latautuisi)
-4. Liitä URL-osoite OmniRoute-yhteysmodaalissa näkyvään kenttään
-5. Napsauta**"Yhdistä"**
+1. OmniRoute opens the Google authorization URL
+2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server)
+3. **Copy the full URL** from your browser's address bar (even if the page doesn't load)
+4. Paste that URL into the field shown in the OmniRoute connection modal
+5. Click **"Connect"**
-> Tämä toimii, koska URL-osoitteessa oleva valtuutuskoodi on kelvollinen riippumatta siitä, ladattiinko uudelleenohjaussivu.---
+> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded.
-
-🇧🇷 Versão em Português
#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+---
-Os provedores**Antigravity**ja**Gemini CLI**usam**Google OAuth 2.0**para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja**exatamente**uma das URIs pre-cadastradas no Google Cloud Console do aplicativo.
+
+🇧🇷 Versão em Português
-As credenciais OAuth embutidas no OmniRoute estão cadastradas**apenas para `localhost`**. Quando você acessa tai OmniRoute em um servidor Remoto (esim. `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:```
+#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+
+Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo.
+
+As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solução: Configure suas próprias credenciais OAuth
-Você precisa criar um**OAuth 2.0 Client ID**ei Google Cloud Console com URI do seu servidor.#### Passo a passo
+Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor.
-**1. Acesse tai Google Cloud Console**
+#### Passo a passo
+
+**1. Acesse o Google Cloud Console**
Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-**2. Crie um novo OAuth 2.0 -asiakastunnus**
+**2. Crie um novo OAuth 2.0 Client ID**
-- Klikkaa em**"+ Luo kirjautumistiedot"**→**"OAuth-asiakastunnus"**
-- Tipo de aplicativo:**"Web-sovellus"**
-- Nimi: escolha qualquer nome (esim. "OmniRoute Remote")
+- Clique em **"+ Create Credentials"** → **"OAuth client ID"**
+- Tipo de aplicativo: **"Web application"**
+- Nome: escolha qualquer nome (ex: `OmniRoute Remote`)
-**3. Adicione valtuutettuina uudelleenohjaus-URI:ina**
+**3. Adicione as Authorized Redirect URIs**
-No campo**"Authorized redirect URIs"**, lisäys:```
+No campo **"Authorized redirect URIs"**, adicione:
+
+```
https://seu-servidor.com/callback
+```
-````
+> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`).
-> Korvaa `seu-servidor.com` pelo domínio tai IP do seu servidor (mukaan lukien porta se necessário, esim. `http://45.33.32.156:20128/callback`).
+**4. Salve e copie as credenciais**
-**4. Tallenna kopio valtuutuksena**
+Após criar, o Google mostrará o **Client ID** e o **Client Secret**.
-Após criar, o Google mostrará o**Client ID**e o**Client Secret**.
+**5. Configure as variáveis de ambiente**
-**5. Määritä variáveis de ambiente**
+No seu `.env` (ou nas variáveis de ambiente do Docker):
-Ei seu `.env` (ou nas variáveis de ambiente do Docker):```bash
+```bash
# Para Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
@@ -1867,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
-````
+```
-**6. Reinicie o OmniRoute**```bash
+**6. Reinicie o OmniRoute**
+```bash
# Se usando npm:
-
npm run dev
# Se usando Docker:
-
docker restart omniroute
-
-````
+```
**7. Tente conectar novamente**
Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth
-Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará.---
+Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará.
+
+---
#### Workaround temporário (sem configurar credenciais próprias)
-Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo**manual de URL**:
+Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**:
-1. OmniRoute lähettää Googlen lupa-osoitteen
-2. Após você autorizar, o Google tentará redirecionar para "localhost" (que falha no servidor Remoto)
-3.**Kopioi URL-osoite täydellinen**da barra de endereço do seu selaimessa (mesmo que a página não carregue)
+1. O OmniRoute abrirá a URL de autorização do Google
+2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto)
+3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue)
4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute
-5. Klikkaa em**"Yhdistä"**
+5. Clique em **"Connect"**
-> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+
+
---
@@ -1905,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux
## 🛠️ Tech Stack
-
+
Click to expand tech stack details
--**Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is**not supported**— `better-sqlite3` native binaries are incompatible)
--**Language**: TypeScript 5.9 —**100% TypeScript**across `src/` and `open-sse/` (zero `any` in core modules since v2.0)
--**Framework**: Next.js 16 + React 19 + Tailwind CSS 4
--**Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions)
--**Schemas**: Zod (MCP tool I/O validation, API contracts)
--**Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
--**Striimaus**: Palvelimen lähettämät tapahtumat (SSE)
--**Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization
--**Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E)
--**CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release)
--**Verkkosivusto**: [omniroute.online](https://omniroute.online)
--**Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
--**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
--**Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible)
+- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0)
+- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4
+- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions)
+- **Schemas**: Zod (MCP tool I/O validation, API contracts)
+- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
+- **Streaming**: Server-Sent Events (SSE)
+- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization
+- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E)
+- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release)
+- **Website**: [omniroute.online](https://omniroute.online)
+- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
+- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
+- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing
+
+
---
## Dokumentaatio
-| Asiakirja | Kuvaus |
-| ----------------------------------------------- | ---------------------------------------------------- |
-| [Käyttöopas](docs/USER_GUIDE.md) | Palveluntarjoajat, yhdistelmät, CLI-integrointi, käyttöönotto |
-| [API-viite](docs/API_REFERENCE.md) | Kaikki päätepisteet esimerkeineen |
-| [MCP-palvelin](open-sse/mcp-server/README.md) | 16 MCP-työkalua, IDE-konfiguraatioita, Python/TS/Go-asiakkaita |
-| [A2A-palvelin](src/lib/a2a/README.md) | JSON-RPC 2.0 -protokolla, taidot, suoratoisto, tehtävänhallinta |
-| [Auto-Combo Engine](docs/auto-combo.md) | 6-faktorinen pisteytys, tilapaketit, itsestään paraneva |
-| [Vianetsintä](docs/TROUBLESHOOTING.md) | Yleisiä ongelmia ja ratkaisuja |
-| [Arkkitehtuuri](docs/ARCHITECTURE.md) | Järjestelmäarkkitehtuuri ja sisäosat |
-| [Osallistuva](CONTRIBUTING.md) | Kehittämisjärjestelyt ja -ohjeet |
-| [OpenAPI-määritys](docs/openapi.yaml) | OpenAPI 3.0 -spesifikaatio |
-| [Turvallisuuspolitiikka](SECURITY.md) | Haavoittuvuusraportointi ja tietoturvakäytännöt |
-| [VM-käyttöönotto](docs/VM_DEPLOYMENT_GUIDE.md) | Täydellinen opas: VM + nginx + Cloudflare-asennus |
-| [Ominaisuudet Galleria](docs/FEATURES.md) | Visuaalinen kojelautakierros kuvakaappauksilla |
-| [Julkaisun tarkistuslista](docs/RELEASE_CHECKLIST.md) | Julkaisua edeltävän vahvistuksen vaiheet |---
+| Document | Description |
+| ---------------------------------------------- | --------------------------------------------------- |
+| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment |
+| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples |
+| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients |
+| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt |
+| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing |
+| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation |
+| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions |
+| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals |
+| [Contributing](CONTRIBUTING.md) | Development setup and guidelines |
+| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification |
+| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices |
+| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup |
+| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots |
+| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps |
+
+---
## 🗺️ Roadmap
-OmniRoutella on**210+ suunniteltua ominaisuutta**useissa kehitysvaiheissa. Tässä ovat tärkeimmät alueet:
+OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas:
-| Category | Suunnitellut ominaisuudet | Kohokohdat |
-| ------------------------------ | ----------------- | --------------------------------------------------------------------------------------- |
-| 🧠**Routing & Intelligence**| 25+ | Pienimmän viiveen reititys, tunnistepohjainen reititys, kiintiön esitarkastus, P2C-tilin valinta |
-| 🔒**Turvallisuus ja vaatimustenmukaisuus**| 20+ | SSRF-karkaisu, valtuustietojen peittäminen, päätepistekohtainen nopeusraja, hallintaavaimen laajuus |
-| 📊**Havaittavuus**| 15+ | OpenTelemetry-integraatio, reaaliaikainen kiintiöiden seuranta, kustannusseuranta mallikohtaisesti |
-| 🔄**Tarjoajien integraatiot**| 20+ | Dynaaminen mallirekisteri, palveluntarjoajan jäähtyminen, usean tilin Codex, Copilot-kiintiön jäsentäminen |
-| ⚡**Suorituskyky**| 15+ | Kaksoisvälimuistikerros, kehotevälimuisti, vastausvälimuisti, suoratoiston ylläpitäminen, erä-API |
-| 🌐**Ekosysteemi**| 10+ | WebSocket API, konfiguroinnin hot-reload, hajautettu konfiguraatiosäilö, kaupallinen tila |### 🔜 Coming Soon
+| Category | Planned Features | Highlights |
+| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- |
+| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection |
+| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping |
+| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model |
+| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing |
+| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API |
+| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode |
-- 🔗**OpenCode Integration**- Natiivitoimittajan tuki OpenCode AI -koodaus-IDE:lle
-- 🔗**TRAE-integraatio**— Täysi tuki TRAE AI -kehityskehykselle
-- 📦**Eräsovellusliittymä**— Asynkroninen eräkäsittely joukkopyyntöille
-- 🎯**Tagipohjainen reititys**- Reittipyynnöt mukautettujen tunnisteiden ja metatietojen perusteella
-- 💰**Alhaisimman kustannustason strategia**- Valitse automaattisesti halvin saatavilla oleva palveluntarjoaja
+### 🔜 Coming Soon
-> 📝 Täydelliset ominaisuudet saatavilla osoitteesta [`docs/new-features/`](docs/new-features/) (217 yksityiskohtaista spesifikaatiota)---
+- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE
+- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework
+- 📦 **Batch API** — Asynchronous batch processing for bulk requests
+- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata
+- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider
+
+> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs)
+
+---
## 👥 Contributors
@@ -1970,18 +2245,20 @@ OmniRoutella on**210+ suunniteltua ominaisuutta**useissa kehitysvaiheissa. Täss
### How to Contribute
-1. Haarukka arkisto
-2. Luo ominaisuushaara (`git checkout -b feature/amazing-feature`)
-3. Vahvista muutokset (`git commit -m 'Lisää upea ominaisuus')
-4. Työnnä haaraan (`git push origin ominaisuus/amazing-feature`)
-5. Avaa vetopyyntö
+1. Fork the repository
+2. Create your feature branch (`git checkout -b feature/amazing-feature`)
+3. Commit your changes (`git commit -m 'Add amazing feature'`)
+4. Push to the branch (`git push origin feature/amazing-feature`)
+5. Open a Pull Request
-Katso tarkemmat ohjeet kohdasta [CONTRIBUTING.md](CONTRIBUTING.md).### Releasing a New Version
+See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines.
+
+### Releasing a New Version
```bash
# Create a release — npm publish happens automatically
gh release create v2.0.0 --title "v2.0.0" --generate-notes
-````
+```
---
@@ -1993,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes
## 🙏 Acknowledgments
-Erityinen kiitos**[decolua](https://github.com/decolua)\*\***[9router](https://github.com/decolua/9router)\*\*– alkuperäiselle projektille, joka inspiroi tätä haarukkaa. OmniRoute rakentaa tälle uskomattomalle perustalle lisäominaisuuksia, multimodaalisia sovellusliittymiä ja täydellistä TypeScript-uudelleenkirjoitusta.
+Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite.
-Erityiset kiitokset**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**-sovellukselle, joka on alkuperäinen Go-toteutus, joka inspiroi tätä JavaScript-porttia.---
+Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port.
+
+---
## Lisenssi
-MIT-lisenssi – katso [LICENSE](LICENSE) saadaksesi lisätietoja.---
+MIT License - see [LICENSE](LICENSE) for details.
+
+---
Built with ❤️ for developers who code 24/7
diff --git a/docs/i18n/fi/docs/ARCHITECTURE.md b/docs/i18n/fi/docs/ARCHITECTURE.md
index c0d7433ec1..2b20d116ee 100644
--- a/docs/i18n/fi/docs/ARCHITECTURE.md
+++ b/docs/i18n/fi/docs/ARCHITECTURE.md
@@ -4,80 +4,93 @@
---
-_Viimeksi päivitetty: 2026-03-28_## Executive Summary
-OmniRoute on paikallinen AI-reititysyhdyskäytävä ja kojelauta, joka on rakennettu Next.js:lle.
-Se tarjoaa yhden OpenAI-yhteensopivan päätepisteen (`/v1/*`) ja reitittää liikenteen useiden alkupään palveluntarjoajien kesken kääntämisen, varaajan, tunnuksen päivityksen ja käytön seurannan avulla.
-Ydinominaisuudet:
+_Last updated: 2026-03-28_
-- OpenAI-yhteensopiva API-pinta CLI:lle/työkaluille (28 toimittajaa)
-- Pyydä/vastaa käännös palveluntarjoajan eri formaattien välillä
-- Mallin yhdistelmävara (usean mallin sarja)
-- Tilitason varatoiminto (usea tili palveluntarjoajaa kohti)
-- OAuth + API-avain tarjoajan yhteyden hallinta
-- Upottamisen luominen /v1/embeddings-tiedoston kautta (6 palveluntarjoajaa, 9 mallia)
-- Kuvien luominen /v1/images/generations-tiedoston kautta (4 toimittajaa, 9 mallia)
-- Ajattele tagien jäsentämistä (`
...`) päättelymalleille
-- Vastauksen desinfiointi tiukan OpenAI SDK -yhteensopivuuden takaamiseksi
-- Roolien normalisointi (kehittäjä→järjestelmä, järjestelmä→käyttäjä) palveluntarjoajien välistä yhteensopivuutta varten
-- Strukturoitu lähdön muunnos (json_schema → Gemini responseSchema)
-- Paikallinen pysyvyys tarjoajille, avaimille, aliaksille, yhdistelmille, asetuksille, hinnoittelulle
-- Käytön/kustannusten seuranta ja pyyntöjen kirjaaminen
-- Valinnainen pilvisynkronointi usean laitteen/tilan synkronointiin
-- IP-sallitut / estolistat API-käyttöoikeuksien hallinnassa
-- Ajatteleva budjetin hallinta (läpivienti/automaattinen/mukautettu/mukautuva)
-- Globaali järjestelmän nopea ruiskutus
-- Istunnon seuranta ja sormenjäljet
-- Tilikohtainen tehostettu hintarajoitus tarjoajakohtaisilla profiileilla
-- Katkaisijakuvio palveluntarjoajan joustavuuden parantamiseksi
-- Ukkosta estävä laumasuoja mutex-lukolla
-- Allekirjoituspohjainen pyyntöjen duplikoinnin välimuisti
-- Verkkotunnustaso: mallin saatavuus, hintasäännöt, varakäytäntö, lukituskäytäntö
-- Verkkotunnuksen tilan pysyvyys (SQLite-kirjoitusvälimuisti varauksille, budjeteille, lukituksille, katkaisimille)
-- Käytäntömoottori keskitettyä pyyntöjen arviointia varten (sulku → budjetti → vara)
-- Pyydä telemetriaa p50/p95/p99-latenssiaggregaatiolla
-- Korrelaatiotunnus (X-Request-Id) päästä päähän -jäljitykseen
-- Vaatimustenmukaisuuden tarkastuksen kirjaaminen ja opt-out API-avaimella
-- Eval-kehys LLM-laadunvarmistukseen
-- Joustavan käyttöliittymän kojelauta, jossa on reaaliaikainen katkaisijatila
-- Modulaariset OAuth-palveluntarjoajat (12 yksittäistä moduulia kohdassa "src/lib/oauth/providers/")
+## Executive Summary
-Ensisijainen suoritusaikamalli:
+OmniRoute is a local AI routing gateway and dashboard built on Next.js.
+It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking.
-- Next.js-sovellusreitit kohdassa `src/app/api/*` toteuttavat sekä hallintapaneelin sovellusliittymiä että yhteensopivuussovellusliittymiä
-- Jaettu SSE/reititysydin kohdassa `src/sse/*` + `open-sse/*` hoitaa palveluntarjoajan suorittamisen, käännöksen, suoratoiston, varatoiminnon ja käytön## Scope and Boundaries
+Core capabilities:
+
+- OpenAI-compatible API surface for CLI/tools (28 providers)
+- Request/response translation across provider formats
+- Model combo fallback (multi-model sequence)
+- Account-level fallback (multi-account per provider)
+- OAuth + API-key provider connection management
+- Embedding generation via `/v1/embeddings` (6 providers, 9 models)
+- Image generation via `/v1/images/generations` (4 providers, 9 models)
+- Think tag parsing (`
...`) for reasoning models
+- Response sanitization for strict OpenAI SDK compatibility
+- Role normalization (developer→system, system→user) for cross-provider compatibility
+- Structured output conversion (json_schema → Gemini responseSchema)
+- Local persistence for providers, keys, aliases, combos, settings, pricing
+- Usage/cost tracking and request logging
+- Optional cloud sync for multi-device/state sync
+- IP allowlist/blocklist for API access control
+- Thinking budget management (passthrough/auto/custom/adaptive)
+- Global system prompt injection
+- Session tracking and fingerprinting
+- Per-account enhanced rate limiting with provider-specific profiles
+- Circuit breaker pattern for provider resilience
+- Anti-thundering herd protection with mutex locking
+- Signature-based request deduplication cache
+- Domain layer: model availability, cost rules, fallback policy, lockout policy
+- Context Relay: session handoff summaries for account rotation continuity
+- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers)
+- Policy engine for centralized request evaluation (lockout → budget → fallback)
+- Request telemetry with p50/p95/p99 latency aggregation
+- Correlation ID (X-Request-Id) for end-to-end tracing
+- Compliance audit logging with opt-out per API key
+- Eval framework for LLM quality assurance
+- Resilience UI dashboard with real-time circuit breaker status
+- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`)
+
+Primary runtime model:
+
+- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs
+- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage
+
+## Scope and Boundaries
### In Scope
-- Paikallisen yhdyskäytävän suoritusaika
-- Kojelaudan hallintasovellusliittymät
-- Palveluntarjoajan todennus ja tunnuksen päivitys
-- Pyydä käännöstä ja SSE-suoratoistoa
-- Paikallinen tila + käytön pysyvyys
-- Valinnainen pilvisynkronointiorkesteri### Out of Scope
+- Local gateway runtime
+- Dashboard management APIs
+- Provider authentication and token refresh
+- Request translation and SSE streaming
+- Local state + usage persistence
+- Optional cloud sync orchestration
-- Pilvipalvelun toteutus osoitteen "NEXT_PUBLIC_CLOUD_URL" takana
-- Palveluntarjoajan SLA/ohjaustaso paikallisen prosessin ulkopuolella
-- Itse ulkoiset CLI-binaarit (Claude CLI, Codex CLI jne.)## Dashboard Surface (Current)
+### Out of Scope
-Pääsivut kohdassa `src/app/(dashboard)/dashboard/`:
+- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL`
+- Provider SLA/control plane outside local process
+- External CLI binaries themselves (Claude CLI, Codex CLI, etc.)
-- `/dashboard` — pika-aloitus + palveluntarjoajan yleiskatsaus
-- "/dashboard/endpoint" - päätepisteen välityspalvelin + MCP + A2A + API-päätepisteen välilehdet
-- "/dashboard/providers" — palveluntarjoajan yhteydet ja tunnistetiedot
-- "/dashboard/combos" - yhdistelmästrategiat, mallit, mallin reitityssäännöt
-- "/dashboard/costs" — kustannusten yhteenlaskettu ja hinnoittelun näkyvyys
-- `/dashboard/analytics' — käyttöanalytiikka ja arvioinnit
-- "/dashboard/limits" - kiintiön/hinnan säätimet
-- "/dashboard/cli-tools" - CLI:n käyttöönotto, suorituksenaikainen tunnistus, asetusten luominen
-- "/dashboard/agents" — havaitut ACP-agentit + mukautetun agentin rekisteröinti
-- `/dashboard/media` — kuvan/videon/musiikin leikkipaikka
-- `/dashboard/search-tools' — hakupalveluntarjoajan testaus ja historia
-- `/dashboard/health' — käytettävyysaika, katkaisijat, nopeusrajoitukset
-- "/dashboard/logs" — pyyntö/välityspalvelin/tarkastus/konsolilokit
-- "/dashboard/settings" — järjestelmäasetusten välilehdet (yleiset, reititys, yhdistelmäoletusasetukset jne.)
-- `/dashboard/api-manager` — API-avaimen elinkaaren ja mallin käyttöoikeudet## High-Level System Context
+## Dashboard Surface (Current)
+
+Main pages under `src/app/(dashboard)/dashboard/`:
+
+- `/dashboard` — quick start + provider overview
+- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs
+- `/dashboard/providers` — provider connections and credentials
+- `/dashboard/combos` — combo strategies, templates, model routing rules
+- `/dashboard/costs` — cost aggregation and pricing visibility
+- `/dashboard/analytics` — usage analytics and evaluations
+- `/dashboard/limits` — quota/rate controls
+- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation
+- `/dashboard/agents` — detected ACP agents + custom agent registration
+- `/dashboard/media` — image/video/music playground
+- `/dashboard/search-tools` — search provider testing and history
+- `/dashboard/health` — uptime, circuit breakers, rate limits
+- `/dashboard/logs` — request/proxy/audit/console logs
+- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.)
+- `/dashboard/api-manager` — API key lifecycle and model permissions
+
+## High-Level System Context
```mermaid
flowchart LR
@@ -129,139 +142,151 @@ flowchart LR
## 1) API and Routing Layer (Next.js App Routes)
-Päähakemistot:
+Main directories:
-- `src/app/api/v1/*` ja `src/app/api/v1beta/*` yhteensopiville sovellusliittymille
-- `src/app/api/*` hallinta-/määrityssovellusliittymille
-- Seuraavaksi kirjoitetaan uudelleen `next.config.mjs`-kartassa `/v1/*` muotoon `/api/v1/*`
+- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs
+- `src/app/api/*` for management/configuration APIs
+- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*`
-Tärkeitä yhteensopivuusreittejä:
+Important compatibility routes:
-- `src/app/api/v1/chat/completions/route.ts'
-- "src/app/api/v1/messages/route.ts".
-- "src/app/api/v1/responses/route.ts".
-- "src/app/api/v1/models/route.ts" - sisältää mukautettuja malleja "custom: true"
-- "src/app/api/v1/embeddings/route.ts" - upotuksen sukupolvi (6 tarjoajaa)
-- "src/app/api/v1/images/generations/route.ts" - kuvien luominen (4+ tarjoajaa, mukaan lukien Antigravity/Nebius)
+- `src/app/api/v1/chat/completions/route.ts`
+- `src/app/api/v1/messages/route.ts`
+- `src/app/api/v1/responses/route.ts`
+- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true`
+- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers)
+- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius)
- `src/app/api/v1/messages/count_tokens/route.ts`
-- "src/app/api/v1/providers/[provider]/chat/completions/route.ts" - palveluntarjoajakohtainen keskustelu
-- `src/app/api/v1/providers/[provider]/embeddings/route.ts' – omat palveluntarjoajakohtaiset upotukset
-- "src/app/api/v1/providers/[provider]/images/generations/route.ts" - palveluntarjoajakohtaiset kuvat
-- "src/app/api/v1beta/models/route.ts".
-- `src/app/api/v1beta/models/[...polku]/route.ts`
+- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat
+- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings
+- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images
+- `src/app/api/v1beta/models/route.ts`
+- `src/app/api/v1beta/models/[...path]/route.ts`
-Hallintoverkkotunnukset:
+Management domains:
-- Todennus/asetukset: `src/app/api/auth/*`, `src/app/api/settings/*`
-- Palveluntarjoajat/yhteydet: `src/app/api/providers*`
-- Palveluntarjoajan solmut: `src/app/api/provider-nodes\*'
-- Mukautetut mallit: `src/app/api/provider-models' (GET/POST/DELETE)
-- Malliluettelo: `src/app/api/models/route.ts' (GET)
-- Välityspalvelimen konfiguraatio: "src/app/api/settings/proxy" (GET/PUT/DELETE) + "src/app/api/settings/proxy/test" (POST)
+- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*`
+- Providers/connections: `src/app/api/providers*`
+- Provider nodes: `src/app/api/provider-nodes*`
+- Custom models: `src/app/api/provider-models` (GET/POST/DELETE)
+- Model catalog: `src/app/api/models/route.ts` (GET)
+- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
- OAuth: `src/app/api/oauth/*`
-- Keys/aliases/combos/pricing: "src/app/api/keys*", "src/app/api/models/alias", "src/app/api/combos*", "src/app/api/pricing"
-- Käyttö: `src/app/api/usage/*`
-- Synkronointi/pilvi: `src/app/api/sync/*`, `src/app/api/cloud/*`
-- CLI-työkalujen apuohjelmat: `src/app/api/cli-tools/*`
-- IP-suodatin: `src/app/api/settings/ip-filter' (GET/PUT)
-- Thinking-budjetti: `src/app/api/settings/thinking-budget' (GET/PUT)
-- Järjestelmäkehote: `src/app/api/settings/system-prompt' (GET/PUT)
-- Istunnot: `src/app/api/sessions' (GET)
-- Nopeusrajoitukset: `src/app/api/rate-limits' (GET)
-- Kestävyys: "src/app/api/resilience" (GET/PATCH) – palveluntarjoajan profiilit, katkaisija, nopeusrajoitustila
-- Kestävyyden nollaus: "src/app/api/resilience/reset" (POST) - nollaa katkaisijat + jäähdytys
-- Välimuistitilastot: `src/app/api/cache/stats' (GET/DELETE)
-- Mallin saatavuus: `src/app/api/models/availability' (GET/POST)
-- Telemetria: "src/app/api/telemetry/summary" (GET)
-- Budjetti: `src/app/api/usage/budget' (GET/POST)
-- Varaketjut: `src/app/api/fallback/chains' (GET/POST/DELETE)
-- Vaatimustenmukaisuuden tarkastus: `src/app/api/compliance/audit-log' (GET)
-- Evals: "src/app/api/evals" (GET/POST), "src/app/api/evals/[suiteId]" (GET)
-- Käytännöt: `src/app/api/policies' (GET/POST)## 2) SSE + Translation Core
+- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing`
+- Usage: `src/app/api/usage/*`
+- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*`
+- CLI tooling helpers: `src/app/api/cli-tools/*`
+- IP filter: `src/app/api/settings/ip-filter` (GET/PUT)
+- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT)
+- System prompt: `src/app/api/settings/system-prompt` (GET/PUT)
+- Sessions: `src/app/api/sessions` (GET)
+- Rate limits: `src/app/api/rate-limits` (GET)
+- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state
+- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns
+- Cache stats: `src/app/api/cache/stats` (GET/DELETE)
+- Model availability: `src/app/api/models/availability` (GET/POST)
+- Telemetry: `src/app/api/telemetry/summary` (GET)
+- Budget: `src/app/api/usage/budget` (GET/POST)
+- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE)
+- Compliance audit: `src/app/api/compliance/audit-log` (GET)
+- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET)
+- Policies: `src/app/api/policies` (GET/POST)
-Päävirtausmoduulit:
+## 2) SSE + Translation Core
-- Merkintä: "src/sse/handlers/chat.ts".
-- Ydinorkesteri: `open-sse/handlers/chatCore.ts`
-- Tarjoajan suoritussovittimet: `open-sse/executors/*`
-- Muototunnistuksen/palveluntarjoajan kokoonpano: `open-sse/services/provider.ts`
-- Mallin jäsennys/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts`
-- Tilin varalogiikka: "open-sse/services/accountFallback.ts".
-- Käännösrekisteri: "open-sse/translator/index.ts".
-- Stream-muunnokset: "open-sse/utils/stream.ts", "open-sse/utils/streamHandler.ts"
-- Käytön purkaminen/normalisointi: `open-sse/utils/usageTracking.ts`
-- Think tag -parser: `open-sse/utils/thinkTagParser.ts`
-- Upotuskäsittelijä: "open-sse/handlers/embeddings.ts".
-- Upotuspalveluntarjoajan rekisteri: `open-sse/config/embeddingRegistry.ts`
-- Kuvanluontikäsittelijä: "open-sse/handlers/imageGeneration.ts".
-- Kuvantarjoajan rekisteri: "open-sse/config/imageRegistry.ts".
-- Vastauksen desinfiointi: "open-sse/handlers/responseSanitizer.ts"
-- Roolin normalisointi: "open-sse/services/roleNormalizer.ts".
+Main flow modules:
-Palvelut (liiketoimintalogiikka):
+- Entry: `src/sse/handlers/chat.ts`
+- Core orchestration: `open-sse/handlers/chatCore.ts`
+- Provider execution adapters: `open-sse/executors/*`
+- Format detection/provider config: `open-sse/services/provider.ts`
+- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts`
+- Account fallback logic: `open-sse/services/accountFallback.ts`
+- Translation registry: `open-sse/translator/index.ts`
+- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
+- Usage extraction/normalization: `open-sse/utils/usageTracking.ts`
+- Think tag parser: `open-sse/utils/thinkTagParser.ts`
+- Embedding handler: `open-sse/handlers/embeddings.ts`
+- Embedding provider registry: `open-sse/config/embeddingRegistry.ts`
+- Image generation handler: `open-sse/handlers/imageGeneration.ts`
+- Image provider registry: `open-sse/config/imageRegistry.ts`
+- Response sanitization: `open-sse/handlers/responseSanitizer.ts`
+- Role normalization: `open-sse/services/roleNormalizer.ts`
-- Tilin valinta/pisteytys: `open-sse/services/accountSelector.ts`
-- Kontekstin elinkaarihallinta: `open-sse/services/contextManager.ts`
-- IP-suodattimen valvonta: "open-sse/services/ipFilter.ts".
-- Istunnon seuranta: `open-sse/services/sessionManager.ts`
-- Pyydä päällekkäisyyden poistoa: `open-sse/services/signatureCache.ts`
-- Järjestelmäkehotteen lisäys: "open-sse/services/systemPrompt.ts".
-- Ajatteleva budjetin hallinta: "open-sse/services/thinkingBudget.ts"
-- Jokerimerkkimallin reititys: `open-sse/services/wildcardRouter.ts`
-- Hintarajan hallinta: `open-sse/services/rateLimitManager.ts`
-- Katkaisija: "open-sse/services/circuitBreaker.ts"
+Services (business logic):
-Domain-kerroksen moduulit:
+- Account selection/scoring: `open-sse/services/accountSelector.ts`
+- Context lifecycle management: `open-sse/services/contextManager.ts`
+- IP filter enforcement: `open-sse/services/ipFilter.ts`
+- Session tracking: `open-sse/services/sessionManager.ts`
+- Request deduplication: `open-sse/services/signatureCache.ts`
+- System prompt injection: `open-sse/services/systemPrompt.ts`
+- Thinking budget management: `open-sse/services/thinkingBudget.ts`
+- Wildcard model routing: `open-sse/services/wildcardRouter.ts`
+- Rate limit management: `open-sse/services/rateLimitManager.ts`
+- Circuit breaker: `open-sse/services/circuitBreaker.ts`
+- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy
+- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions
-- Mallin saatavuus: "src/lib/domain/modelAvailability.ts".
-- Kustannussäännöt/budjetit: `src/lib/domain/costRules.ts`
-- Varakäytäntö: "src/lib/domain/fallbackPolicy.ts".
-- Yhdistelmäratkaisu: `src/lib/domain/comboResolver.ts'
-- Lukituskäytäntö: "src/lib/domain/lockoutPolicy.ts".
-- Käytäntömoottori: "src/domain/policyEngine.ts" — keskitetty lukitus → budjetti → varaarviointi
-- Virhekoodiluettelo: "src/lib/domain/errorCodes.ts".
-- Pyyntötunnus: `src/lib/domain/requestId.ts'
-- Haun aikakatkaisu: "src/lib/domain/fetchTimeout.ts".
-- Pyydä telemetriaa: `src/lib/domain/requestTelemetry.ts`
-- Vaatimustenmukaisuus/tarkastus: `src/lib/domain/compliance/index.ts'
+Domain layer modules:
+
+- Model availability: `src/lib/domain/modelAvailability.ts`
+- Cost rules/budgets: `src/lib/domain/costRules.ts`
+- Fallback policy: `src/lib/domain/fallbackPolicy.ts`
+- Combo resolver: `src/lib/domain/comboResolver.ts`
+- Lockout policy: `src/lib/domain/lockoutPolicy.ts`
+- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation
+- Error codes catalog: `src/lib/domain/errorCodes.ts`
+- Request ID: `src/lib/domain/requestId.ts`
+- Fetch timeout: `src/lib/domain/fetchTimeout.ts`
+- Request telemetry: `src/lib/domain/requestTelemetry.ts`
+- Compliance/audit: `src/lib/domain/compliance/index.ts`
- Eval runner: `src/lib/domain/evalRunner.ts`
-- Verkkotunnuksen tilan pysyvyys: `src/lib/db/domainState.ts' — SQLite CRUD varaketjuille, budjeteille, kustannushistorialle, lukitustilalle, katkaisimille
+- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers
-OAuth-palveluntarjoajan moduulit (12 yksittäistä tiedostoa kohdassa "src/lib/oauth/providers/"):
+OAuth provider modules (12 individual files under `src/lib/oauth/providers/`):
-- Rekisterihakemisto: "src/lib/oauth/providers/index.ts".
-- Yksittäiset palveluntarjoajat: "claude.ts", "codex.ts", "gemini.ts", "antigravity.ts", "qoder.ts", "qwen.ts", "kimi-coding.ts", "github.ts", "kiro.cursorts", `cline.ts`
-- Ohut kääre: "src/lib/oauth/providers.ts" - uudelleenvienti yksittäisistä moduuleista## 3) Persistence Layer
+- Registry index: `src/lib/oauth/providers/index.ts`
+- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
+- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules
-Ensisijainen tila DB (SQLite):
+## 3) Persistence Layer
-- Ydininfrastruktuuri: "src/lib/db/core.ts" (better-sqlite3, migraatiot, WAL)
-- Vie julkisivu uudelleen: "src/lib/localDb.ts" (ohut yhteensopivuuskerros soittajille)
-- tiedosto: `${DATA_DIR}/storage.sqlite` (tai `$XDG_CONFIG_HOME/omniroute/storage.sqlite`, kun se on asetettu, muuten `~/.omniroute/storage.sqlite`)
-- entiteetit (taulukot + KV-nimitilat): providerConnections, providerNodes, mallialiakset, yhdistelmät, apiKeys, asetukset, hinnoittelu,**customModels**,**proxyConfig**,**ipFilter**,**thhinkingBudget**,**systemPrompt**
+Primary state DB (SQLite):
-Käytön pysyvyys:
+- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL)
+- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers)
+- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`)
+- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt**
-- julkisivu: "src/lib/usageDb.ts" (hajotetut moduulit tiedostossa "src/lib/usage/\*")
-- SQLite-taulukot tiedostossa "storage.sqlite": "usage_history", "call_logs", "proxy_logs"
-- valinnaiset tiedostoartefaktit jäävät yhteensopivuutta/virheenkorjausta varten (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `
/logs/...`)
-- Vanhat JSON-tiedostot siirretään SQLiteen käynnistyssiirroilla, kun ne ovat olemassa
+Usage persistence:
+
+- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`)
+- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs`
+- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`)
+- legacy JSON files are migrated to SQLite by startup migrations when present
Domain State DB (SQLite):
-- "src/lib/db/domainState.ts" - CRUD-toiminnot toimialueen tilalle
-- Taulukot (luodut tiedostossa "src/lib/db/core.ts"): "domain_fallback_chains", "domain_budgets", "domain_cost_history", "domain_lockout_state", "domain_circuit_breakers"
-- Kirjoitusvälimuistin malli: muistissa olevat kartat ovat arvovaltaisia ajon aikana; mutaatiot kirjoitetaan synkronisesti SQLiten kanssa; tila palautetaan DB:stä kylmäkäynnistyksen yhteydessä## 4) Auth + Security Surfaces
+- `src/lib/db/domainState.ts` — CRUD operations for domain state
+- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
+- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start
-- Hallintapaneelin evästeiden todennus: "src/proxy.ts", "src/app/api/auth/login/route.ts"
-- API-avaimen luonti/vahvistus: `src/shared/utils/apiKey.ts`
-- Palveluntarjoajan salaisuudet säilyivät "providerConnections"-merkinnöissä
-- Lähtevän välityspalvelimen tuki "open-sse/utils/proxyFetch.ts" (env vars) ja "open-sse/utils/networkProxy.ts" kautta (määritettävä palveluntarjoajakohtaisesti tai globaali)## 5) Cloud Sync
+## 4) Auth + Security Surfaces
-- Aikataulun aloitus: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
-- Säännöllinen tehtävä: `src/shared/services/cloudSyncScheduler.ts`
-- Säännöllinen tehtävä: `src/shared/services/modelSyncScheduler.ts`
-- Hallitse reittiä: `src/app/api/sync/cloud/route.ts'## Request Lifecycle (`/v1/chat/completions`)
+- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts`
+- API key generation/verification: `src/shared/utils/apiKey.ts`
+- Provider secrets persisted in `providerConnections` entries
+- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global)
+
+## 5) Cloud Sync
+
+- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
+- Periodic task: `src/shared/services/cloudSyncScheduler.ts`
+- Periodic task: `src/shared/services/modelSyncScheduler.ts`
+- Control route: `src/app/api/sync/cloud/route.ts`
+
+## Request Lifecycle (`/v1/chat/completions`)
```mermaid
sequenceDiagram
@@ -338,7 +363,9 @@ flowchart TD
Q -- No --> R[Return all unavailable]
```
-Varapäätökset tehdään "open-sse/services/accountFallback.ts":n avulla tilakoodeja ja virheviestiheuristiikkaa käyttäen. Yhdistelmäreititys lisää yhden ylimääräisen suojan: palveluntarjoajan kattamat 400:t, kuten ylävirran sisällön lohko- ja roolivahvistuksen epäonnistumiset, käsitellään mallin paikallisina virheinä, jotta myöhempiä yhdistelmäkohteita voidaan edelleen suorittaa.## OAuth Onboarding and Token Refresh Lifecycle
+Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run.
+
+## OAuth Onboarding and Token Refresh Lifecycle
```mermaid
sequenceDiagram
@@ -368,7 +395,9 @@ sequenceDiagram
Test-->>UI: validation result
```
-Päivitys reaaliaikaisen liikenteen aikana suoritetaan "open-sse/handlers/chatCore.ts" -tiedostossa suorittimen "refreshCredentials()" kautta.## Cloud Sync Lifecycle (Enable / Sync / Disable)
+Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`.
+
+## Cloud Sync Lifecycle (Enable / Sync / Disable)
```mermaid
sequenceDiagram
@@ -400,7 +429,9 @@ sequenceDiagram
Sync-->>UI: disabled
```
-"CloudSyncScheduler" käynnistää säännöllisen synkronoinnin, kun pilvi on käytössä.## Data Model and Storage Map
+Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled.
+
+## Data Model and Storage Map
```mermaid
erDiagram
@@ -501,12 +532,14 @@ erDiagram
}
```
-Fyysiset tallennustiedostot:
+Physical storage files:
-- ensisijainen ajonaikainen tietokanta: `${DATA_DIR}/storage.sqlite`
-- pyyntölokin rivit: `${DATA_DIR}/log.txt` (compat/debug artefact)
-- jäsennellyt puhelun hyötykuorma-arkistot: `${DATA_DIR}/call_logs/`
-- valinnainen kääntäjä/pyydä virheenkorjausistuntoja: `/logs/...`## Deployment Topology
+- primary runtime DB: `${DATA_DIR}/storage.sqlite`
+- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact)
+- structured call payload archives: `${DATA_DIR}/call_logs/`
+- optional translator/request debug sessions: `/logs/...`
+
+## Deployment Topology
```mermaid
flowchart LR
@@ -541,205 +574,249 @@ flowchart LR
### Route and API Modules
-- `src/app/api/v1/*`, `src/app/api/v1beta/*`: yhteensopivuussovellusliittymät
-- `src/app/api/v1/providers/[provider]/*`: omat palveluntarjoajakohtaiset reitit (chat, upotukset, kuvat)
-- `src/app/api/providers\*: palveluntarjoajan CRUD, validointi, testaus
-- `src/app/api/provider-nodes\*: mukautettu yhteensopiva solmuhallinta
-- "src/app/api/provider-models": mukautetun mallin hallinta (CRUD)
-- "src/app/api/models/route.ts": malliluettelon sovellusliittymä (aliakset + mukautetut mallit)
-- `src/app/api/oauth/*`: OAuth/laitekoodikulku
-- `src/app/api/keys\*: paikallisen API-avaimen elinkaari
-- "src/app/api/models/alias": aliaksen hallinta
-- `src/app/api/combos*`: varayhdistelmähallinta
-- "src/app/api/pricing": hinnoittelu ohittaa kustannuslaskennan
-- "src/app/api/settings/proxy": välityspalvelimen määritykset (GET/PUT/DELETE)
-- "src/app/api/settings/proxy/test": lähtevän välityspalvelimen yhteystesti (POST)
-- `src/app/api/usage/*`: käyttö- ja lokisovellusliittymät
-- `src/app/api/sync/*` + `src/app/api/cloud/*`: pilvisynkronointi ja pilveen suuntautuvat apuohjelmat
-- `src/app/api/cli-tools/*`: paikalliset CLI-asetusten kirjoittajat/tarkistajat
-- `src/app/api/settings/ip-filter': IP-sallittujen luettelo/estolista (GET/PUT)
-- `src/app/api/settings/thhinking-budget': ajattelutunnuksen budjetin konfiguraatio (GET/PUT)
-- "src/app/api/settings/system-prompt": yleinen järjestelmäkehote (GET/PUT)
-- `src/app/api/sessions': aktiivisten istuntojen luettelo (GET)
-- "src/app/api/rate-limits": tilikohtainen korkorajoitustila (GET)### Routing and Execution Core
+- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs
+- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images)
+- `src/app/api/providers*`: provider CRUD, validation, testing
+- `src/app/api/provider-nodes*`: custom compatible node management
+- `src/app/api/provider-models`: custom model management (CRUD)
+- `src/app/api/models/route.ts`: model catalog API (aliases + custom models)
+- `src/app/api/oauth/*`: OAuth/device-code flows
+- `src/app/api/keys*`: local API key lifecycle
+- `src/app/api/models/alias`: alias management
+- `src/app/api/combos*`: fallback combo management
+- `src/app/api/pricing`: pricing overrides for cost calculation
+- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE)
+- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST)
+- `src/app/api/usage/*`: usage and logs APIs
+- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers
+- `src/app/api/cli-tools/*`: local CLI config writers/checkers
+- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT)
+- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT)
+- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT)
+- `src/app/api/sessions`: active session listing (GET)
+- `src/app/api/rate-limits`: per-account rate limit status (GET)
-- `src/sse/handlers/chat.ts: pyynnön jäsennys, yhdistelmäkäsittely, tilin valintasilmukka
-- `open-sse/handlers/chatCore.ts`: käännös, suorittimen lähettäminen, uudelleenyritys/päivityskäsittely, streamin määritys
-- `open-sse/executors/*`: palveluntarjoajakohtainen verkko- ja muotokäyttäytyminen### Translation Registry and Format Converters
+### Routing and Execution Core
-- "open-sse/translator/index.ts": kääntäjien rekisteri ja orkestrointi
-- Pyydä kääntäjiä: `open-sse/translator/request/*`
-- Vastauskääntäjät: `open-sse/translator/response/*`
-- Muotovakiot: "open-sse/translator/formats.ts".### Persistence
+- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop
+- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup
+- `open-sse/executors/*`: provider-specific network and format behavior
-- `src/lib/db/*`: pysyvä konfiguraatio/tila ja verkkotunnuksen pysyvyys SQLitessa
-- `src/lib/localDb.ts`: DB-moduulien yhteensopivuuden uudelleenvienti
-- `src/lib/usageDb.ts`: käyttöhistorian/puhelulokien julkisivu SQLite-taulukoiden päällä## Provider Executor Coverage (Strategy Pattern)
+### Translation Registry and Format Converters
-Jokaisella palveluntarjoajalla on erikoistunut suorittaja, joka laajentaa "BaseExecutoria" (hakemistossa "open-sse/executors/base.ts"), joka tarjoaa URL-osoitteen rakentamisen, otsikon rakentamisen, uudelleenyrityksen eksponentiaalisella perääntymisellä, valtuustietojen päivityskoukut ja execute()-orkesterimenetelmän.
+- `open-sse/translator/index.ts`: translator registry and orchestration
+- Request translators: `open-sse/translator/request/*`
+- Response translators: `open-sse/translator/response/*`
+- Format constants: `open-sse/translator/formats.ts`
-| Toteuttaja | Palveluntarjoaja(t) | Erikoiskäsittely |
-| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------- |
-| "DefaultExecutor" | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, ilotulitus, Cerebras, Cohere, NVIDIA | Dynaaminen URL-/otsikkomääritykset tarjoajakohtaisesti |
-| "AntigravityExecutor" | Google Antigravity | Mukautetut projekti-/istuntotunnukset, Yritä uudelleen jäsentämisen jälkeen |
-| "CodexExecutor" | OpenAI Codex | Syöttää järjestelmäohjeita, pakottaa päättelyponnistuksen |
-| "CursorExecutor" | Kohdistin IDE | ConnectRPC-protokolla, Protobuf-koodaus, pyynnön allekirjoitus tarkistussumman kautta |
-| "GithubExecutor" | GitHub Copilot | Copilot-tunnuksen päivitys, VSC-koodia jäljittelevät otsikot |
-| "KiroExecutor" | AWS CodeWhisperer/Kiro | AWS EventStream binaarimuoto → SSE-muunnos |
-| "GeminiCLIExecutor" | Gemini CLI | Google OAuth -tunnuksen päivitysjakso |
+### Persistence
-Kaikki muut palveluntarjoajat (mukaan lukien mukautetut yhteensopivat solmut) käyttävät DefaultExecutoria.## Provider Compatibility Matrix
+- `src/lib/db/*`: persistent config/state and domain persistence on SQLite
+- `src/lib/localDb.ts`: compatibility re-export for DB modules
+- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables
-| Palveluntarjoaja | Muoto | Auth | Striimaa | Ei-stream | Token Refresh | Käyttösovellusliittymä |
-| ---------------- | ----------------- | ------------------------- | -------------------- | --------- | ------------- | --------------------------- | ------------------------------ |
-| Claude | claude | API-avain / OAuth | ✅ | ✅ | ✅ | ⚠️ Vain järjestelmänvalvoja |
-| Kaksoset | kaksoset | API-avain / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
-| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
-| Antigravitaatio | antigravitaatio | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API |
-| OpenAI | openai | API-avain | ✅ | ✅ | ❌ | ❌ |
-| Codex | openai-vastaukset | OAuth | ✅ pakotettu | ❌ | ✅ | ✅ Hintarajat |
-| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Kiintiön tilannekuvat |
-| Kursori | kohdistin | Mukautettu tarkistussumma | ✅ | ✅ | ❌ | ❌ |
-| Kiro | kiro | AWS SSO OIDC | ✅ (TapahtumaStream) | ❌ | ✅ | ✅ Käyttörajoitukset |
-| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Pyynnöstä |
-| Qoder | openai | OAuth (Perus) | ✅ | ✅ | ✅ | ⚠️ Pyynnöstä |
-| OpenRouter | openai | API-avain | ✅ | ✅ | ❌ | ❌ |
-| GLM/Kimi/MiniMax | claude | API-avain | ✅ | ✅ | ❌ | ❌ |
-| DeepSeek | openai | API-avain | ✅ | ✅ | ❌ | ❌ |
-| Groq | openai | API-avain | ✅ | ✅ | ❌ | ❌ |
-| xAI (Grok) | openai | API-avain | ✅ | ✅ | ❌ | ❌ |
-| Mistral | openai | API-avain | ✅ | ✅ | ❌ | ❌ |
-| Hämmennys | openai | API-avain | ✅ | ✅ | ❌ | ❌ |
-| Yhdessä AI | openai | API-avain | ✅ | ✅ | ❌ | ❌ |
-| Ilotulitus AI | openai | API-avain | ✅ | ✅ | ❌ | ❌ |
-| Aivot | openai | API-avain | ✅ | ✅ | ❌ | ❌ |
-| Cohere | openai | API-avain | ✅ | ✅ | ❌ | ❌ |
-| NVIDIA NIM | openai | API-avain | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage |
+## Provider Executor Coverage (Strategy Pattern)
-Havaittuja lähdemuotoja ovat:
+Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method.
-- "openai".
-- "openai-vastaukset".
-- "claude".
-- "kaksoset".
+| Executor | Provider(s) | Special Handling |
+| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- |
+| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider |
+| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing |
+| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort |
+| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum |
+| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers |
+| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion |
+| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle |
-Kohdemuotoja ovat:
+All other providers (including custom compatible nodes) use the `DefaultExecutor`.
-- OpenAI chat / vastaukset
+## Provider Compatibility Matrix
+
+| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API |
+| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ |
+| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only |
+| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API |
+| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits |
+| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots |
+| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ |
+| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits |
+| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request |
+| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request |
+| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ |
+| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+
+## Format Translation Coverage
+
+Detected source formats include:
+
+- `openai`
+- `openai-responses`
+- `claude`
+- `gemini`
+
+Target formats include:
+
+- OpenAI chat/Responses
- Claude
-- Gemini/Gemini-CLI/Antigravity-kuori
+- Gemini/Gemini-CLI/Antigravity envelope
- Kiro
-- Kursori
+- Cursor
-Käännöksissä käytetään keskitinmuotona**OpenAI-muotoa**— kaikki konversiot menevät OpenAI:n kautta välimuotona:```
+Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate:
+
+```
Source Format → OpenAI (hub) → Target Format
+```
-````
+Translations are selected dynamically based on source payload shape and provider target format.
-Käännökset valitaan dynaamisesti lähteen hyötykuorman muodon ja toimittajan kohdemuodon perusteella.
+Additional processing layers in the translation pipeline:
-Muut käsittelytasot käännösputkessa:
+- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance
+- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE)
+- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field
+- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema`
--**Vastausten desinfiointi**– Poistaa standardista poikkeavat kentät OpenAI-muotoisista vastauksista (sekä suoratoistosta että ei-suoratoistosta) varmistaakseen tiukan SDK-yhteensopivuuden
--**Roolin normalisointi**— Muuntaa "kehittäjä" → "järjestelmä" muille kuin OpenAI-kohteille; yhdistää `system` → `user` malleille, jotka hylkäävät järjestelmäroolin (GLM, ERNIE)
--**Think-tunnisteen purkaminen**— Jäsentää "..." -lohkot sisällöstä "reasoning_content"-kenttään
--**Strukturoitu tulos**— Muuntaa OpenAI `response_format.json_schema` Geminin `responseMimeType` + `responseSchema`.## Supported API Endpoints
+## Supported API Endpoints
-| Päätepiste | Muoto | Käsittelijä |
-| --------------------------------------------------- | ------------------- | -------------------------------------------------------------------- |
-| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` |
-| `POST /v1/messages` | Claude Viestit | Sama käsittelijä (tunnistettu automaattisesti) |
-| `POST /v1/responses` | OpenAI-vastaukset | `open-sse/handlers/responsesHandler.ts` |
-| `POST /v1/embeddings` | OpenAI Embeddings | "open-sse/handlers/embeddings.ts" |
-| `HAE /v1/embeddings` | Malliluettelo | API reitti |
-| `POST /v1/images/generations` | OpenAI-kuvat | `open-sse/handlers/imageGeneration.ts` |
-| `GET /v1/images/generations` | Malliluettelo | API reitti |
-| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Palveluntarjoajakohtainen mallin validointi |
-| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Palveluntarjoajakohtainen mallin validointi |
-| `POST /v1/providers/{provider}/images/generations` | OpenAI-kuvat | Palveluntarjoajakohtainen mallin validointi |
-| `POST /v1/messages/count_tokens` | Claude Token Count | API reitti |
-| `HAE /v1/mallit` | OpenAI-mallien luettelo | API-reitti (chat + upotus + kuva + mukautetut mallit) |
-| "GET /api/models/catalog" | Luettelo | Kaikki mallit ryhmitelty tarjoajan + tyypin mukaan |
-| `POST /v1beta/models/*:streamGenerateContent` | Gemini syntyperäinen | API reitti |
-| `GET/PUT/DELETE /api/settings/proxy` | Välityspalvelimen kokoonpano | Verkon välityspalvelimen määritykset |
-| "POST /api/settings/proxy/test" | Välityspalvelinyhteydet | Välityspalvelimen kunto/yhteystestin päätepiste |
-| `GET/POST/DELETE /api/provider-models` | Palveluntarjoajan mallit | Palveluntarjoajan mallin metatietojen tausta mukautettuja ja hallittuja saatavilla olevia malleja |## Bypass Handler
+| Endpoint | Format | Handler |
+| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- |
+| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` |
+| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) |
+| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` |
+| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` |
+| `GET /v1/embeddings` | Model listing | API route |
+| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` |
+| `GET /v1/images/generations` | Model listing | API route |
+| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation |
+| `POST /v1/messages/count_tokens` | Claude Token Count | API route |
+| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) |
+| `GET /api/models/catalog` | Catalog | All models grouped by provider + type |
+| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route |
+| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration |
+| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint |
+| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models |
-Ohituskäsittelijä (`open-sse/utils/bypassHandler.ts`) sieppaa Claude CLI:n tunnetut "heittopyynnöt" – lämmittelypingit, otsikon poiminnot ja tunnukset - ja palauttaa**väärennetyn vastauksen**kuluttamatta alkupään toimittajatunnuksia. Tämä käynnistyy vain, kun "User-Agent" sisältää "claude-cli".## Request Logger Pipeline
+## Bypass Handler
-Pyyntöloggeri (`open-sse/utils/requestLogger.ts`) tarjoaa 7-vaiheisen virheenkorjauslokiputken, joka on oletusarvoisesti pois käytöstä ja joka on käytössä kohdassa ENABLE_REQUEST_LOGS=true:```
+The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`.
+
+## Request Logger Pipeline
+
+The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`:
+
+```
1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json
→ 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt
-````
+```
-Tiedostot kirjoitetaan hakemistoon `/logs//` jokaista pyyntöistuntoa varten.## Failure Modes and Resilience
+Files are written to `/logs//` for each request session.
+
+## Failure Modes and Resilience
## 1) Account/Provider Availability
-- Palveluntarjoajan tilin jäähtyminen ohimenevien / nopeus / todennusvirheiden vuoksi
-- tilin varaosa ennen epäonnistunutta pyyntöä
-- Yhdistelmämallin palautus, kun nykyisen mallin/palveluntarjoajan polku on käytetty loppuun## 2) Token Expiry
+- provider account cooldown on transient/rate/auth errors
+- account fallback before failing request
+- combo model fallback when current model/provider path is exhausted
-- esitarkista ja päivitä yrittämällä uudelleen päivitettävien palveluntarjoajien kohdalla
-- 401/403 yritä uudelleen päivitysyrityksen jälkeen ydinpolulla## 3) Stream Safety
+## 2) Token Expiry
-- irrotettava stream-ohjain
-- käännösvirta streamin lopun huuhtelulla ja [VALMIS]-käsittelyllä
-- käyttöarvion varavaihtoehto, kun palveluntarjoajan käytön metatiedot puuttuvat## 4) Cloud Sync Degradation
+- pre-check and refresh with retry for refreshable providers
+- 401/403 retry after refresh attempt in core path
-- Synkronointivirheet tulevat esiin, mutta paikallinen suoritusaika jatkuu
-- ajastimessa on uudelleenyrityslogiikka, mutta säännöllinen suoritus tällä hetkellä kutsuu oletusarvoisesti yhden yrityksen synkronointia## 5) Data Integrity
+## 3) Stream Safety
-- SQLite-skeeman siirrot ja automaattisen päivityksen koukut käynnistyksen yhteydessä
-- vanha JSON → SQLite-siirtoyhteensopivuuspolku## Observability and Operational Signals
+- disconnect-aware stream controller
+- translation stream with end-of-stream flush and `[DONE]` handling
+- usage estimation fallback when provider usage metadata is missing
-Ajonaikaisen näkyvyyden lähteet:
+## 4) Cloud Sync Degradation
-- konsolin lokit osoitteesta "src/sse/utils/logger.ts".
-- SQLiten pyyntökohtaiset käyttöaggregaatit ("usage_history", "call_logs", "proxy_logs")
-- nelivaiheiset yksityiskohtaiset hyötykuorman kaappaukset SQLitessa (`request_detail_logs`), kun `settings.detailed_logs_enabled=true`
-- tekstimuotoisen pyynnön tilaloki tiedostossa "log.txt" (valinnainen/compat)
-- valinnaiset syvät pyyntö-/käännöslokit lokit/-kohdassa, kun ENABLE_REQUEST_LOGS=true
-- hallintapaneelin käyttöpäätepisteet (`/api/usage/*`) käyttöliittymän käyttöä varten
+- sync errors are surfaced but local runtime continues
+- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default
-Yksityiskohtainen pyyntöhyötykuormakaappaus tallentaa jopa neljä JSON-hyötykuorman vaihetta reititettyä puhelua kohden:
+## 5) Data Integrity
-- Asiakkaalta saatu raakapyyntö
-- käännetty pyyntö todella lähetetty alkupäässä
-- palveluntarjoajan vastaus rekonstruoitu JSON-muodossa; suoratoistovastaukset tiivistetään lopulliseksi yhteenvedoksi ja virran metadataksi
-- OmniRouten palauttama lopullinen asiakkaan vastaus; suoratoistovastaukset tallennetaan samaan kompaktiin tiivistelmään## Security-Sensitive Boundaries
+- SQLite schema migrations and auto-upgrade hooks at startup
+- legacy JSON → SQLite migration compatibility path
-- JWT-salaisuus (`JWT_SECRET`) suojaa hallintapaneelin istunnon evästeen vahvistuksen/allekirjoituksen
-- Alkuperäisen salasanan käynnistys (`INITIAL_PASSWORD`) on määritettävä eksplisiittisesti ensiajoa varten
-- API-avaimen HMAC-salaisuus (`API_KEY_SECRET`) suojaa luodun paikallisen API-avainmuodon
-- Tarjoajan salaisuudet (API-avaimet/tunnisteet) säilyvät paikallisessa tietokannassa, ja ne tulee suojata tiedostojärjestelmätasolla
-- Pilvisynkronoinnin päätepisteet perustuvat API-avaimen todennus + konetunnuksen semantiikkaan## Environment and Runtime Matrix
+## Observability and Operational Signals
-Koodin aktiivisesti käyttämät ympäristömuuttujat:
+Runtime visibility sources:
-- Sovellus/todennus: "JWT_SECRET", "INITIAL_PASSWORD"
-- Tallennustila: "DATA_DIR".
-- Yhteensopivan solmun toiminta: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
-- Valinnainen tallennuskannan ohitus (Linux/macOS, kun "DATA_DIR" ei ole asetettu): "XDG_CONFIG_HOME"
-- Suojaushajautus: `API_KEY_SECRET`, `MACHINE_ID_SALT`
-- Kirjaaminen: "ENABLE_REQUEST_LOGS".
-- Synkronointi/pilvi-URL-osoite: NEXT_PUBLIC_BASE_URL, NEXT_PUBLIC_CLOUD_URL
-- Lähtevä välityspalvelin: "HTTP_PROXY", "HTTPS_PROXY", "ALL_PROXY", "NO_PROXY" ja pienet versiot
-- SOCKS5-ominaisuuden liput: "ENABLE_SOCKS5_PROXY", "NEXT_PUBLIC_ENABLE_SOCKS5_PROXY"
-- Alusta/ajonaikaiset apuohjelmat (ei sovelluskohtaiset asetukset): "APPDATA", "NODE_ENV", "PORTTI", "HOSTNAME"## Known Architectural Notes
+- console logs from `src/sse/utils/logger.ts`
+- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`)
+- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true`
+- textual request status log in `log.txt` (optional/compat)
+- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true`
+- dashboard usage endpoints (`/api/usage/*`) for UI consumption
-1. `usageDb` ja `localDb` jakavat saman perushakemistokäytännön (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) vanhojen tiedostojen siirrolla.
-2. "/api/v1/route.ts" siirtää samaan yhdistetyn luettelon rakennustyökaluun, jota "/api/v1/models" ("src/app/api/v1/models/catalog.ts") käyttää semanttisen ajautumisen välttämiseksi.
-3. Pyyntöloggeri kirjoittaa täydet otsikot/runko, kun se on käytössä; käsittele lokihakemistoa arkaluontoisena.
-4. Pilven toiminta riippuu oikeasta NEXT_PUBLIC_BASE_URL-osoitteesta ja pilvipäätepisteen saavutettavuudesta.
-5. Hakemisto "open-sse/" julkaistaan @omniroute/open-sse**npm-työtilapaketina**. Lähdekoodi tuo sen @omniroute/open-sse/...-tiedoston kautta (ratkaisi Next.js `transpilePackages`). Tämän asiakirjan tiedostopolut käyttävät edelleen hakemistonimeä `open-sse/` johdonmukaisuuden vuoksi.
-6. Hallintapaneelin kaaviot käyttävät**Uudelleenkaavioita**(SVG-pohjainen) helppokäyttöisten, interaktiivisten analytiikkavisualisoinnit (mallien käyttöpalkkikaaviot, toimittajien erittelytaulukot onnistumisprosentteineen) varten.
-7. E2E-testit käyttävät**Playwrightia**(`tests/e2e/`), suoritetaan komennolla "npm run test:e2e". Yksikkötesteissä käytetään**Node.js-testirunneria**(`tests/unit/`), suoritetaan komennolla "npm run test:unit". Lähdekoodi kohdassa `src/` on**TypeScript**(`.ts`/`.tsx`); `open-sse/`-työtila pysyy JavaScriptina (`.js`).
-8. Asetukset-sivu on järjestetty viiteen välilehteen: Suojaus, Reititys (6 globaalia strategiaa: täytä ensin, round-robin, p2c, satunnainen, vähiten käytetty, kustannusoptimoitu), Resilience (muokattavat nopeusrajoitukset, katkaisija, käytännöt), AI (ajattelubudjetti, järjestelmäkehote, kehote välimuisti), Advanced (välityspalvelin).## Operational Verification Checklist
+Detailed request payload capture stores up to four JSON payload stages per routed call:
-- Koonti lähteestä: `npm run build`
-- Build Docker -kuva: `docker build -t omniroute .`
-- Aloita huolto ja varmista:
-- "HAE /api/settings".
-- "GET /api/v1/models".
-- CLI-kohteen perus-URL-osoitteen tulee olla "http://:20128/v1", kun PORT=20128
+- raw request received from the client
+- translated request actually sent upstream
+- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata
+- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form
+
+## Security-Sensitive Boundaries
+
+- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing
+- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning
+- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format
+- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level
+- Cloud sync endpoints rely on API key auth + machine id semantics
+
+## Environment and Runtime Matrix
+
+Environment variables actively used by code:
+
+- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD`
+- Storage: `DATA_DIR`
+- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
+- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME`
+- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT`
+- Logging: `ENABLE_REQUEST_LOGS`
+- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
+- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants
+- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
+- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`
+
+## Known Architectural Notes
+
+1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration.
+2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift.
+3. Request logger writes full headers/body when enabled; treat log directory as sensitive.
+4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability.
+5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency.
+6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates).
+7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`).
+8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy).
+9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed.
+10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22.
+11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible.
+
+## Operational Verification Checklist
+
+- Build from source: `npm run build`
+- Build Docker image: `docker build -t omniroute .`
+- Start service and verify:
+- `GET /api/settings`
+- `GET /api/v1/models`
+- CLI target base URL should be `http://:20128/v1` when `PORT=20128`
diff --git a/docs/i18n/fi/docs/FEATURES.md b/docs/i18n/fi/docs/FEATURES.md
index cf0668989f..cc7e406c0b 100644
--- a/docs/i18n/fi/docs/FEATURES.md
+++ b/docs/i18n/fi/docs/FEATURES.md
@@ -4,102 +4,168 @@
---
-Visuaalinen opas OmniRoute-hallintapaneelin jokaiseen osioon.---
+
+
+Visual guide to every section of the OmniRoute dashboard.
+
+---
## 🔌 Providers
-Hallinnoi AI-palveluntarjoajan yhteyksiä: OAuth-palveluntarjoajat (Claude Code, Codex, Gemini CLI), API-avaintoimittajat (Groq, DeepSeek, OpenRouter) ja ilmaiset palveluntarjoajat (Qoder, Qwen, Kiro). Kiro-tilit sisältävät luottosaldon seurannan – jäljellä olevat saldot, kokonaisrahoitus ja uusimispäivä näkyvät kohdassa Dashboard → Käyttö.
+Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage.
+
+
---
## 🎨 Combos
-Luo mallin reitityskomboja kuudella strategialla: prioriteetti, painotettu, kiertävä, satunnainen, vähiten käytetty ja kustannusoptimoitu. Jokainen yhdistelmä ketjuttaa useita malleja automaattisilla varauksilla ja sisältää nopeat mallit ja valmiustarkistukset.
+Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks.
+
+
---
## 📊 Analytics
-Kattava käyttöanalytiikka tunnuksen kulutuksella, kustannusarvioilla, aktiivisuuslämpökartoilla, viikoittaisilla jakelukaavioilla ja palveluntarjoajakohtaisilla erittelyillä.
+Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns.
+
+
---
## 🏥 System Health
-Reaaliaikainen seuranta: käyttöaika, muisti, versio, latenssiprosenttipisteet (p50/p95/p99), välimuistitilastot ja palveluntarjoajan katkaisijan tilat.
+Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states.
+
+
---
## 🔧 Translator Playground
-Neljä tilaa API-käännösten virheenkorjaukseen:**Playground**(muodonmuunnin),**Chat Tester**(livepyynnöt),**Test Bench**(erätestit) ja**Live Monitor**(reaaliaikainen suoratoisto).
+Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream).
+
+
---
## 🎮 Model Playground _(v2.0.9+)_
-Testaa mitä tahansa mallia suoraan kojelaudalta. Valitse palveluntarjoaja, malli ja päätepiste, kirjoita kehotteita Monaco Editorilla, suoratoista vastaukset reaaliajassa, keskeytä kesken stream ja tarkastele ajoitusmittauksia.---
+Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics.
+
+---
## 🎨 Themes _(v2.0.5+)_
-Muokattavat väriteemat koko kojelautaan. Valitse 7 esiasetetusta väristä (koralli, sininen, punainen, vihreä, violetti, oranssi, syaani) tai luo mukautettu teema valitsemalla mikä tahansa kuusioväri. Tukee vaaleaa, tummaa ja järjestelmätilaa.---
+Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode.
+
+---
## ⚙️ Settings
-Kattava asetuspaneeli välilehdillä:
+Comprehensive settings panel with tabs:
--**Yleistä**- Järjestelmän tallennus, varmuuskopioiden hallinta (vienti/tuonti tietokanta) -**Ulkoasu**- Teeman valitsin (tumma/vaalea/järjestelmä), väriteeman esiasetukset ja mukautetut värit, terveyslokin näkyvyys, sivupalkin kohteiden näkyvyyden säätimet -**Turvallisuus**— API-päätepisteiden suojaus, mukautetun palveluntarjoajan esto, IP-suodatus, istuntotiedot -**Reititys**— Mallin aliakset, taustatehtävän huononeminen -**Kestävyys**— Hintarajoituksen pysyvyys, katkaisijan viritys, estettyjen tilien automaattinen poistaminen käytöstä, palveluntarjoajan vanhenemisen valvonta -**Lisäasetukset**— Kokoonpanon ohitukset, määrityksen kirjausketju, varatilan heikkenemistila
+- **General** — System storage, backup management (export/import database)
+- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls
+- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info
+- **Routing** — Model aliases, background task degradation
+- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration
+- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode
+
+
---
## 🔧 CLI Tools
-Yhden napsautuksen konfigurointi AI-koodaustyökaluille: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor ja Factory Droid. Sisältää automaattisen konfiguroinnin käyttöönotto/nollaus, yhteysprofiilit ja mallikartoituksen.
+One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping.
+
+
---
## 🤖 CLI Agents _(v2.0.11+)_
-Kojelauta CLI-agenttien löytämiseen ja hallintaan. Näyttää 14 sisäänrakennetun agentin (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) ruudukon, jossa on:
+Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with:
--**Asennustila**— Asennettu / Ei löydy versiontunnistuksen kanssa -**Protokollamerkit**— stdio, HTTP jne. -**Muokatut agentit**— Rekisteröi mikä tahansa CLI-työkalu lomakkeella (nimi, binaari, versiokomento, spawn args) -**CLI-sormenjälkien vastaavuus**– Palveluntarjoajakohtainen kytkin vastaamaan alkuperäisten CLI-pyyntöjen allekirjoituksia, mikä vähentää eston riskiä ja säilyttää välityspalvelimen IP-osoitteen---
+- **Installation status** — Installed / Not Found with version detection
+- **Protocol badges** — stdio, HTTP, etc.
+- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args)
+- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP
+
+---
+
+## 🔗 Context Relay _(v3.5.5+)_
+
+A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context.
+
+Configurable via combo-level or global settings:
+- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%)
+- **Max Messages For Summary** — How much recent history to condense
+- **Summary Model** — Optional override model for generating the handoff summary
+
+Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md).
+
+---
+
+## 🛡️ Proxy Hardening _(v3.5.5+)_
+
+Comprehensive proxy configuration enforcement across the entire request pipeline:
+
+- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments
+- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings
+- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22
+- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS
+
+---
## 🖼️ Media _(v2.0.3+)_
-Luo kuvia, videoita ja musiikkia kojelaudalta. Tukee OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open ja MusicGen.---
+Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen.
+
+---
## 📝 Request Logs
-Reaaliaikainen pyyntöjen kirjaaminen suodatuksella palveluntarjoajan, mallin, tilin ja API-avaimen mukaan. Näyttää tilakoodit, tunnuksen käytön, viiveen ja vastaustiedot.
+Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details.
+
+
---
## 🌐 API Endpoint
-Yhdistetty API-päätepisteesi ominaisuuksien erittelyllä: Chat Completions, Responses API, upotukset, kuvan luominen, uudelleensijoitus, äänen transkriptio, tekstistä puheeksi, moderaatiot ja rekisteröidyt API-avaimet. Cloudflare Quick Tunnel -integraatio ja pilvivälityspalvelintuki etäkäyttöä varten.
+Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access.
+
+
---
## 🔑 API Key Management
-Luo, laajenna ja peruuta API-avaimia. Jokainen avain voidaan rajoittaa tiettyihin malleihin/palveluntarjoajiin, joilla on täydet käyttöoikeudet tai vain lukuoikeudet. Visuaalinen avainten hallinta käytön seurannalla.---
+Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking.
+
+---
## 📋 Audit Log
-Hallinnollinen toimintojen seuranta suodatuksella toimintotyypin, toimijan, kohteen, IP-osoitteen ja aikaleiman mukaan. Täydellinen tietoturvatapahtumahistoria.---
+Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history.
+
+---
## 🖥️ Desktop Application
-Native Electron -työpöytäsovellus Windowsille, macOS:lle ja Linuxille. Suorita OmniRoute itsenäisenä sovelluksena, jossa on järjestelmälokeron integrointi, offline-tuki, automaattinen päivitys ja asennus yhdellä napsautuksella.
+Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install.
-Tärkeimmät ominaisuudet:
+Key features:
-- Palvelimen valmiuskysely (ei tyhjää näyttöä kylmäkäynnistyksen yhteydessä)
-- Järjestelmälokero portinhallinnan kanssa
-- Sisällön suojauskäytäntö
-- Yksiosainen lukko
-- Automaattinen päivitys uudelleenkäynnistyksen yhteydessä
-- Alustan ehdollinen käyttöliittymä (macOS-liikennevalot, Windowsin/Linuxin oletusotsikkopalkki)
-- Hardened Electron build -pakkaus – itsenäisen nipun symlinkoidut "solmumoduulit" tunnistetaan ja hylätään ennen pakkausta, mikä estää ajonaikaisen riippuvuuden rakennuskoneesta (v2.5.5+)
+- Server readiness polling (no blank screen on cold start)
+- System tray with port management
+- Content Security Policy
+- Single-instance lock
+- Auto-update on restart
+- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar)
+- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+)
-📖 Katso täydelliset asiakirjat osoitteesta [`electron/README.md`](../electron/README.md).
+📖 See [`electron/README.md`](../electron/README.md) for full documentation.
diff --git a/docs/i18n/fi/docs/TROUBLESHOOTING.md b/docs/i18n/fi/docs/TROUBLESHOOTING.md
index d301f4f000..c2fbb68c02 100644
--- a/docs/i18n/fi/docs/TROUBLESHOOTING.md
+++ b/docs/i18n/fi/docs/TROUBLESHOOTING.md
@@ -4,68 +4,142 @@
---
-OmniRouten yleisiä ongelmia ja ratkaisuja.---
+
+
+Common problems and solutions for OmniRoute.
+
+---
## Quick Fixes
-| Ongelma | Ratkaisu |
-| ---------------------------------- | ------------------------------------------------------------------------------- | --- |
-| Ensimmäinen kirjautuminen ei toimi | Aseta 'INITIAL_PASSWORD' .env:ssä (ei kovakoodattua oletusarvoa) |
-| Kojelauta avautuu väärään porttiin | Aseta `PORT=20128` ja `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
-| Ei pyyntölokeja kohdassa "lokit/" | Aseta ENABLE_REQUEST_LOGS=true |
-| EACCES: lupa evätty | Aseta "DATA_DIR=/polku/kirjoitettavaan/hakemistoon" ohittaaksesi "~/.omniroute" |
-| Reititysstrategia ei tallennu | Päivitys versioon 1.4.11+ (Zod-skeeman korjaus asetusten pysyvyyttä varten) | --- |
+| Problem | Solution |
+| ----------------------------- | ------------------------------------------------------------------ |
+| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) |
+| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
+| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` |
+| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` |
+| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) |
+| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below |
+| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below |
+
+---
+
+## Node.js Compatibility
+
+
+
+### Login page crashes or shows "Module self-registration" error
+
+**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database.
+
+**Symptoms:**
+- Login page shows a blank screen or a server error
+- Console shows `Error: Module did not self-register` or similar native binding errors
+- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected
+
+**Fix:**
+
+1. Install Node.js 22 LTS (recommended):
+ ```bash
+ nvm install 22
+ nvm use 22
+ ```
+2. Verify your version: `node --version` should show `v22.x.x`
+3. Reinstall OmniRoute: `npm install -g omniroute`
+4. Restart: `omniroute`
+
+> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**.
+
+---
+
+## Proxy Issues
+
+
+
+### Provider validation shows "fetch failed"
+
+**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing.
+
+**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically.
+
+### Token health check fails with "fetch failed"
+
+**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection.
+
+**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+.
+
+### SOCKS5 proxy returns "invalid onRequestStart method"
+
+**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation.
+
+**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+.
+
+---
## Provider Issues
### "Language model did not provide messages"
-**Syy:**Palveluntarjoajan kiintiö käytetty.
+**Cause:** Provider quota exhausted.
-**Korjaa:**
+**Fix:**
-1. Tarkista kojelaudan kiintiöiden seuranta
-2. Käytä yhdistelmää varatasoilla
-3. Vaihda halvempaan/ilmaiseen tasoon### Rate Limiting
+1. Check dashboard quota tracker
+2. Use a combo with fallback tiers
+3. Switch to cheaper/free tier
-**Syy:**Tilauskiintiö käytetty.
+### Rate Limiting
-**Korjaa:**
+**Cause:** Subscription quota exhausted.
-- Lisää vara: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-- Käytä GLM/MiniMaxia halvana varmuuskopiona### OAuth Token Expired
+**Fix:**
-OmniRoute päivittää tunnukset automaattisesti. Jos ongelmat jatkuvat:
+- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Use GLM/MiniMax as cheap backup
-1. Kojelauta → Palveluntarjoaja → Yhdistä uudelleen
-2. Poista ja lisää palveluntarjoajan yhteys uudelleen---
+### OAuth Token Expired
+
+OmniRoute auto-refreshes tokens. If issues persist:
+
+1. Dashboard → Provider → Reconnect
+2. Delete and re-add the provider connection
+
+---
## Cloud Issues
### Cloud Sync Errors
-1. Varmista, että BASE_URL osoittaa käynnissä olevaan esiintymääsi (esim. http://localhost:20128)
-2. Varmista, että CLOUD_URL-osoite osoittaa pilvipäätepisteeseesi (esim. https://omniroute.dev).
-3. Pidä NEXT*PUBLIC*\*-arvot kohdakkain palvelinpuolen arvojen kanssa### Cloud `stream=false` Returns 500
+1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`)
+2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`)
+3. Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**Oire:**"Odottamaton tunnus "d"..." pilvipäätepisteessä ei-suoratoistopuheluille.
+### Cloud `stream=false` Returns 500
-**Syy:**Upstream palauttaa SSE-hyötykuorman, kun asiakas odottaa JSONia.
+**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls.
-**Ratkaisu:**Käytä "stream=true" pilvisuorapuheluissa. Paikallinen suoritusaika sisältää SSE→JSON-varavaihtoehdon.### Cloud Says Connected but "Invalid API key"
+**Cause:** Upstream returns SSE payload while client expects JSON.
-1. Luo uusi avain paikallisesta hallintapaneelista (`/api/keys`)
-2. Suorita pilvisynkronointi: Ota pilvi käyttöön → Synkronoi nyt
-3. Vanhat/synkronoimattomat avaimet voivat edelleen palauttaa 401:n pilvessä---
+**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback.
+
+### Cloud Says Connected but "Invalid API key"
+
+1. Create a fresh key from local dashboard (`/api/keys`)
+2. Run cloud sync: Enable Cloud → Sync Now
+3. Old/non-synced keys can still return `401` on cloud
+
+---
## Docker Issues
### CLI Tool Shows Not Installed
-1. Tarkista ajonaikaiset kentät: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
-2. Kannettava tila: käytä kuvakohdetta "runner-cli" (niputetut CLI:t)
-3. Isäntäliitostila: aseta CLI_EXTRA_PATHS ja liitä isäntäalustahakemisto vain luku -muotoiseksi
-4. Jos "installed=true" ja "runnable=false": binaari löytyi, mutta kuntotarkastus epäonnistui### Quick Runtime Validation
+1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
+2. For portable mode: use image target `runner-cli` (bundled CLIs)
+3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only
+4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck
+
+### Quick Runtime Validation
```bash
curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
@@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,
### High Costs
-1. Tarkista käyttötilastot kohdassa Dashboard → Usage
-2. Vaihda ensisijaiseksi malliksi GLM/MiniMax
-3. Käytä ilmaista tasoa (Gemini CLI, Qoder) ei-kriittisiin tehtäviin
-4. Aseta kustannusbudjetit API-avainta kohti: Dashboard → API Keys → Budget---
+1. Check usage stats in Dashboard → Usage
+2. Switch primary model to GLM/MiniMax
+3. Use free tier (Gemini CLI, Qoder) for non-critical tasks
+4. Set cost budgets per API key: Dashboard → API Keys → Budget
+
+---
## Debugging
### Enable Request Logs
-Aseta ENABLE_REQUEST_LOGS=true .env-tiedostoosi. Lokit näkyvät lokit/hakemistossa.### Check Provider Health
+Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory.
+
+### Check Provider Health
```bash
# Health dashboard
@@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health
### Runtime Storage
-- Päätila: `${DATA_DIR}/storage.sqlite` (palveluntarjoajat, yhdistelmät, aliakset, avaimet, asetukset)
-- Käyttö: SQLite-taulukot tiedostossa "storage.sqlite" ("usage_history", "call_logs", "proxy_logs") + valinnainen "${DATA_DIR}/log.txt" ja "${DATA_DIR}/call_logs/"
-- Pyydä lokeja: `/logs/...` (kun `ENABLE_REQUEST_LOGS=true`)---
+- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings)
+- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/`
+- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`)
+
+---
## Circuit Breaker Issues
### Provider stuck in OPEN state
-Kun palveluntarjoajan katkaisija on AUKI, pyynnöt estetään, kunnes jäähdytys päättyy.
+When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires.
-**Korjaa:**
+**Fix:**
-1. Siirry kohtaan**Käyttöpaneeli → Asetukset → Resilience**
-2. Tarkista asianomaisen palveluntarjoajan katkaisijakortti
-3. Napsauta**Nollaa kaikki**tyhjentääksesi kaikki katkaisijat tai odota jäähdytysajan päättymistä
-4. Varmista, että palveluntarjoaja on todella saatavilla, ennen kuin nollaat### Provider keeps tripping the circuit breaker
+1. Go to **Dashboard → Settings → Resilience**
+2. Check the circuit breaker card for the affected provider
+3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire
+4. Verify the provider is actually available before resetting
-Jos palveluntarjoaja siirtyy toistuvasti OPEN-tilaan:
+### Provider keeps tripping the circuit breaker
-1. Tarkista vikakuvio kohdasta**Dashboard → Health → Provider Health**
-2. Siirry kohtaan**Settings → Resilience → Provider Profiles**ja nosta vikakynnystä.
-3. Tarkista, onko palveluntarjoaja muuttanut API-rajoja tai vaatiiko todennuksen uudelleen
-4. Tarkista viiveen telemetria — korkea latenssi voi aiheuttaa aikakatkaisuun perustuvia virheitä---
+If a provider repeatedly enters OPEN state:
+
+1. Check **Dashboard → Health → Provider Health** for the failure pattern
+2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold
+3. Check if the provider has changed API limits or requires re-authentication
+4. Review latency telemetry — high latency may cause timeout-based failures
+
+---
## Audio Transcription Issues
### "Unsupported model" error
-- Varmista, että käytät oikeaa etuliitettä: "deepgram/nova-3" tai "assemblyai/best"
-- Varmista, että palveluntarjoaja on yhdistetty kohdassa**Dashboard → Providers**### Transcription returns empty or fails
+- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best`
+- Verify the provider is connected in **Dashboard → Providers**
-- Tarkista tuetut äänimuodot: "mp3", "wav", "m4a", "flac", "ogg", "webm"
-- Varmista, että tiedostokoko on palveluntarjoajan rajoissa (yleensä < 25 Mt)
-- Tarkista palveluntarjoajan API-avaimen voimassaolo toimittajakortista---
+### Transcription returns empty or fails
+
+- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm`
+- Verify file size is within provider limits (typically < 25MB)
+- Check provider API key validity in the provider card
+
+---
## Translator Debugging
-Käytä**Käyttöpaneeli → Kääntäjä**muotojen käännösongelmien korjaamiseen:
+Use **Dashboard → Translator** to debug format translation issues:
-| Tila | Milloin käyttää |
-| ------------------------- | ------------------------------------------------------------------------------------------------------- | ------------------------ |
-| **Leikkikenttä** | Vertaa syöttö-/tulostusmuotoja vierekkäin – liitä epäonnistunut pyyntö nähdäksesi, miten se käännetään |
-| **Pikaviestien testaaja** | Lähetä reaaliaikaisia viestejä ja tarkasta koko pyynnön/vastauksen hyötykuorma, mukaan lukien otsikot |
-| **Testipenkki** | Suorita erätestejä muotoyhdistelmille selvittääksesi, mitkä käännökset ovat rikki |
-| **Live Monitor** | Tarkkaile reaaliaikaista pyyntövirtaa havaitaksesi ajoittaiset käännösongelmat | ### Common format issues |
+| Mode | When to Use |
+| ---------------- | -------------------------------------------------------------------------------------------- |
+| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates |
+| **Chat Tester** | Send live messages and inspect the full request/response payload including headers |
+| **Test Bench** | Run batch tests across format combinations to find which translations are broken |
+| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues |
--**Ajattelevat tunnisteet eivät näy**— Tarkista, tukeeko kohdetoimittaja ajattelua ja ajattelun budjettiasetusta -**Työkalukutsujen pudottaminen**— Jotkin muotokäännökset voivat poistaa ei-tuetut kentät. vahvista leikkikenttätilassa -**Järjestelmäkehote puuttuu**— Claude ja Gemini kahvajärjestelmä kehottaa eri tavalla; tarkista käännöstulos -**SDK palauttaa raakamerkkijonon objektin sijaan**— Korjattu versiossa 1.1.0: vastauspuhdistin poistaa nyt standardista poikkeavat kentät ("x_groq", "usage_breakdown" jne.), jotka aiheuttavat OpenAI SDK Pydantic -tarkistusvirheitä -**GLM/ERNIE hylkää "järjestelmän" roolin**- Korjattu versiossa 1.1.0: roolin normalisoija yhdistää automaattisesti järjestelmäviestit käyttäjäviesteiksi yhteensopimattomissa malleissa -**"kehittäjäroolia" ei tunnistettu**- Korjattu versiossa 1.1.0: muunnetaan automaattisesti "järjestelmäksi" muille kuin OpenAI-palveluntarjoajille -**`json_schema` ei toimi Geminin kanssa**— Korjattu versiossa 1.1.0: `response_format` muunnetaan nyt Geminin `responseMimeType` + `responseSchema` -muotoon.---
+### Common format issues
+
+- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting
+- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode
+- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output
+- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures
+- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models
+- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers
+- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema`
+
+---
## Resilience Settings
### Auto rate-limit not triggering
-- Automaattinen nopeusrajoitus koskee vain API-avainten toimittajia (ei OAuth-tilausta)
-- Varmista, että**Asetukset → Resilienssi → Palveluntarjoajan profiilit**on automaattinen rajoitus käytössä
-- Tarkista, palauttaako palveluntarjoaja "429"-tilakoodit tai "Retry-After"-otsikot### Tuning exponential backoff
+- Auto rate-limit only applies to API key providers (not OAuth/subscription)
+- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled
+- Check if the provider returns `429` status codes or `Retry-After` headers
-Palveluntarjoajan profiilit tukevat näitä asetuksia:
+### Tuning exponential backoff
--**Perusviive**— Ensimmäinen odotusaika ensimmäisen epäonnistumisen jälkeen (oletus: 1 s) -**Maksimiviive**- Odotusajan enimmäisraja (oletus: 30 s) -**Kerroin**— Kuinka paljon viivettä lisätään peräkkäistä vikaa kohti (oletus: 2x)### Anti-thundering herd
+Provider profiles support these settings:
-Kun monet samanaikaiset pyynnöt osuvat nopeusrajoitettuun palveluntarjoajaan, OmniRoute käyttää mutex + automaattista nopeuden rajoitusta sarjoittamaan pyynnöt ja estämään peräkkäiset epäonnistumiset. Tämä on automaattinen API-avainten tarjoajille.---
+- **Base delay** — Initial wait time after first failure (default: 1s)
+- **Max delay** — Maximum wait time cap (default: 30s)
+- **Multiplier** — How much to increase delay per consecutive failure (default: 2x)
+
+### Anti-thundering herd
+
+When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers.
+
+---
## Optional RAG / LLM failure taxonomy (16 problems)
-Jotkut OmniRouten käyttäjät sijoittavat yhdyskäytävän RAG- tai agenttipinojen eteen. Näissä asetuksissa on tavallista nähdä outo kuvio: OmniRoute näyttää terveeltä (palveluntarjoajat valmiina, reititysprofiilit kunnossa, ei nopeusrajoitushälytyksiä), mutta lopullinen vastaus on silti väärä.
+Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong.
-Käytännössä nämä tapaukset tulevat yleensä loppupään RAG-putkistosta, eivät itse yhdyskäytävästä.
+In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself.
-Jos haluat jaetun sanaston kuvaamaan näitä vikoja, voit käyttää WFGY ProblemMapia, ulkoista MIT-lisenssitekstiresurssia, joka määrittelee kuusitoista toistuvaa RAG/LLM-vikamallia. Korkealla tasolla se kattaa:
+If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers:
-- haun ajautuminen ja rikotut kontekstin rajat
-- tyhjät tai vanhentuneet indeksit ja vektorivarastot
-- upottaminen vs. semanttinen yhteensopivuus
-- Nopeat kokoonpano- ja kontekstiikkuna-ongelmat
-- logiikka romahtaa ja liian itsevarmat vastaukset
-- pitkän ketjun ja agenttien koordinaatiohäiriöt
-- monen agentin muisti ja roolien siirtyminen
-- käyttöönotto- ja käynnistystilausongelmat
+- retrieval drift and broken context boundaries
+- empty or stale indexes and vector stores
+- embedding versus semantic mismatch
+- prompt assembly and context window issues
+- logic collapse and overconfident answers
+- long chain and agent coordination failures
+- multi agent memory and role drift
+- deployment and bootstrap ordering problems
-Idea on yksinkertainen:
+The idea is simple:
-1. Kun tutkit huonoa vastausta, tallenna:
- - käyttäjän tehtävä ja pyyntö
- - reitti- tai tarjoajayhdistelmä OmniRoutessa
- - mikä tahansa loppupäässä käytetty RAG-konteksti (haettu asiakirjat, työkalukutsut jne.)
-2. Kartoita tapahtuma yhteen tai kahteen WFGY-ongelmakarttanumeroon (`No.1` … `No.16`).
-3. Tallenna numero omaan kojelautaan, runbookiin tai tapahtumaseurantaan OmniRoute-lokien viereen.
-4. Käytä vastaavaa WFGY-sivua päättääksesi, onko sinun muutettava RAG-pinoa, noutajaa tai reititysstrategiaa.
+1. When you investigate a bad response, capture:
+ - user task and request
+ - route or provider combo in OmniRoute
+ - any RAG context used downstream (retrieved documents, tool calls, etc)
+2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`).
+3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs.
+4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy.
-Koko teksti ja konkreettiset reseptit löytyvät täältä (MIT-lisenssi, vain teksti):
+Full text and concrete recipes live here (MIT license, text only):
[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md)
-Voit jättää tämän osion huomioimatta, jos et käytä RAG- tai agenttiputkia OmniRouten takana.---
+You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute.
+
+---
## Still Stuck?
--**GitHub-ongelmat**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Arkkitehtuuri**: Katso sisäiset tiedot osoitteesta [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) -**API-viite**: Katso [`docs/API_REFERENCE.md`](API_REFERENCE.md) kaikista päätepisteistä -**Health Dashboard**: Tarkista järjestelmän reaaliaikainen tila kohdasta**Dashboard → Health** -**Kääntäjä**: Käytä**Käyttöpaneeli → Kääntäjä**muotoongelmien korjaamiseen
+- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details
+- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints
+- **Health Dashboard**: Check **Dashboard → Health** for real-time system status
+- **Translator**: Use **Dashboard → Translator** to debug format issues
diff --git a/docs/i18n/fi/llm.txt b/docs/i18n/fi/llm.txt
new file mode 100644
index 0000000000..e0b98c94a8
--- /dev/null
+++ b/docs/i18n/fi/llm.txt
@@ -0,0 +1,476 @@
+# OmniRoute (Suomi)
+
+🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt)
+
+---
+
+
+> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app.
+
+## Yleiskatsaus
+
+OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free.
+
+**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost.
+
+**Current version:** 3.5.5
+
+## Tech Stack
+
+- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`)
+- **Framework:** Next.js 16 (App Router) with TypeScript 5.9
+- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations)
+- **State management:** Zustand (client), SQLite (server persistence)
+- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons
+- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth
+- **Schemas:** Zod v4 for all API / MCP input validation
+- **Background jobs:** Custom token health check scheduler, 24h model auto-sync
+- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses
+- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine
+- **i18n:** next-intl with 30 languages
+- **Desktop:** Electron (cross-platform: Windows, macOS, Linux)
+- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`)
+
+## Project Structure
+
+```
+/
+├── src/ # Main application source
+│ ├── app/ # Next.js App Router pages and API routes
+│ │ ├── (dashboard)/ # Dashboard UI pages
+│ │ │ └── dashboard/
+│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents)
+│ │ │ ├── analytics/ # Usage analytics and charts
+│ │ │ ├── api-manager/ # API key management
+│ │ │ ├── audit/ # Audit logs
+│ │ │ ├── auto-combo/ # Auto-combo engine dashboard
+│ │ │ ├── cache/ # Cache dashboard (semantic cache stats)
+│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.)
+│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates)
+│ │ │ ├── costs/ # Cost tracking per provider/model
+│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs
+│ │ │ ├── health/ # System health (uptime, circuit breakers, latency)
+│ │ │ ├── limits/ # Rate limits dashboard
+│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed)
+│ │ │ ├── media/ # Image/video/music generation + transcription
+│ │ │ ├── memory/ # Memory system dashboard
+│ │ │ ├── onboarding/ # Onboarding wizard
+│ │ │ ├── playground/ # Model playground (Monaco editor, streaming)
+│ │ │ ├── providers/ # Provider management (OAuth + API key + free)
+│ │ │ ├── search-tools/ # Search tools configuration
+│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced)
+│ │ │ ├── skills/ # Skills system dashboard
+│ │ │ ├── translator/ # Format translator + debug tools
+│ │ │ └── usage/ # Usage history
+│ │ ├── api/ # REST API endpoints (51 route directories)
+│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings,
+│ │ │ │ # images, audio, videos, music, moderations, rerank, search,
+│ │ │ │ # responses, messages, registered-keys, quotas, accounts)
+│ │ │ ├── v1beta/ # Gemini-compatible API
+│ │ │ ├── a2a/ # A2A agent management API
+│ │ │ ├── acp/ # ACP agent management API
+│ │ │ ├── oauth/ # OAuth flows per provider
+│ │ │ ├── providers/ # Provider CRUD and batch testing
+│ │ │ ├── models/ # Dashboard model listing and aliases
+│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains)
+│ │ │ ├── memory/ # Memory system API
+│ │ │ ├── skills/ # Skills system API
+│ │ │ ├── evals/ # Eval runner API
+│ │ │ ├── mcp/ # MCP HTTP transport API
+│ │ │ ├── search/ # Search provider API
+│ │ │ ├── webhooks/ # Webhook management
+│ │ │ ├── tunnels/ # Cloudflare tunnel management
+│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.)
+│ │ ├── landing/ # Landing page
+│ │ ├── login/ # Login page
+│ │ ├── forgot-password/ # Password recovery
+│ │ ├── status/ # Status page
+│ │ └── docs/ # In-app documentation
+│ ├── domain/ # Domain types and policy engine
+│ │ ├── policyEngine.ts # Central policy engine
+│ │ ├── comboResolver.ts # Combo resolution logic
+│ │ ├── costRules.ts # Cost calculation rules
+│ │ ├── degradation.ts # Graceful degradation
+│ │ ├── fallbackPolicy.ts # Fallback behavior
+│ │ ├── lockoutPolicy.ts # Account lockout logic
+│ │ ├── modelAvailability.ts # Model availability checks
+│ │ ├── providerExpiration.ts # Provider credential expiration
+│ │ ├── quotaCache.ts # Quota caching layer
+│ │ ├── configAudit.ts # Configuration auditing
+│ │ └── responses.ts # Domain response types
+│ ├── i18n/ # Internationalization
+│ │ └── messages/ # 30 language JSON files
+│ ├── lib/ # Core libraries
+│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server
+│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting)
+│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup
+│ │ │ └── streaming.ts # SSE streaming for A2A
+│ │ ├── acp/ # Agent Communication Protocol registry and manager
+│ │ ├── compliance/ # Compliance policy engine
+│ │ ├── db/ # SQLite database layer (21 modules + migrations)
+│ │ │ ├── core.ts # Database initialization, connection, schema
+│ │ │ ├── providers.ts # Provider connection CRUD
+│ │ │ ├── models.ts # Model catalog management
+│ │ │ ├── combos.ts # Combo configuration
+│ │ │ ├── apiKeys.ts # API key management
+│ │ │ ├── settings.ts # Settings persistence
+│ │ │ ├── backup.ts # Database backup/restore
+│ │ │ ├── proxies.ts # Proxy registry
+│ │ │ ├── prompts.ts # Prompt templates
+│ │ │ ├── webhooks.ts # Webhook subscriptions
+│ │ │ ├── detailedLogs.ts # Detailed request logging
+│ │ │ ├── domainState.ts # Domain state persistence
+│ │ │ ├── registeredKeys.ts # Registered API keys with quotas
+│ │ │ ├── quotaSnapshots.ts # Quota snapshot history
+│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings
+│ │ │ ├── cliToolState.ts # CLI tool state tracking
+│ │ │ ├── encryption.ts # Data encryption
+│ │ │ ├── readCache.ts # Read-through cache layer
+│ │ │ ├── secrets.ts # Secrets management
+│ │ │ ├── stateReset.ts # State reset utilities
+│ │ │ ├── migrationRunner.ts # Schema migration runner
+│ │ │ └── migrations/ # 16 SQL migration files
+│ │ ├── evals/ # Eval runner and scheduler
+│ │ ├── memory/ # Persistent conversational memory
+│ │ │ ├── extraction.ts # Memory extraction from conversations
+│ │ │ ├── injection.ts # Memory injection into context
+│ │ │ ├── retrieval.ts # Memory retrieval/search
+│ │ │ ├── store.ts # Memory persistence layer
+│ │ │ └── summarization.ts # Memory summarization
+│ │ ├── oauth/ # OAuth providers, services, and utilities
+│ │ │ ├── constants/ # Default OAuth credentials (overridable via env)
+│ │ │ ├── providers/ # Provider-specific OAuth configs
+│ │ │ ├── services/ # Provider-specific token exchange logic
+│ │ │ └── utils/ # PKCE, callback server, token helpers
+│ │ ├── plugins/ # Plugin system
+│ │ ├── skills/ # Extensible skill framework
+│ │ │ ├── registry.ts # Skill registration
+│ │ │ ├── executor.ts # Skill execution engine
+│ │ │ ├── sandbox.ts # Skill sandbox environment
+│ │ │ ├── builtin/ # Built-in skills
+│ │ │ ├── interception.ts # Skill request interception
+│ │ │ └── injection.ts # Skill context injection
+│ │ ├── usage/ # Usage tracking system
+│ │ │ ├── callLogs.ts # Call log persistence
+│ │ │ ├── costCalculator.ts # Cost calculation engine
+│ │ │ └── usageHistory.ts # Usage history queries
+│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers
+│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management
+│ │ ├── pricingSync.ts # LiteLLM pricing data sync
+│ │ ├── semanticCache.ts # Semantic caching layer
+│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler
+│ │ ├── webhookDispatcher.ts # Webhook event dispatcher
+│ │ └── localDb.ts # Unified re-export layer for all DB modules
+│ ├── middleware/ # Request middleware
+│ │ └── promptInjectionGuard.ts # Prompt injection detection
+│ ├── mitm/ # MITM proxy capability
+│ │ ├── cert/ # Certificate management
+│ │ ├── dns/ # DNS handling
+│ │ ├── targets/ # Target routing
+│ │ └── manager.ts # MITM proxy manager
+│ ├── shared/ # Shared utilities, components, and constants
+│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.)
+│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes
+│ │ ├── contracts/ # Shared API contracts
+│ │ ├── hooks/ # React hooks
+│ │ ├── middleware/ # Shared middleware utilities
+│ │ ├── schemas/ # Shared Zod schemas
+│ │ ├── services/ # Shared services
+│ │ ├── types/ # Shared TypeScript types
+│ │ ├── validation/ # Zod schemas (settings, providers, routes)
+│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID)
+│ ├── sse/ # SSE proxy pipeline
+│ │ ├── services/ # Auth resolution, format translation, response handling
+│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency
+│ ├── store/ # Zustand client-side stores (theme, providers, etc.)
+│ └── types/ # TypeScript type definitions
+├── open-sse/ # Standalone SSE server (npm workspace)
+│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video,
+│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models)
+│ ├── executors/ # Provider-specific request executors (14 executors)
+│ │ ├── base.ts # Base executor with shared logic
+│ │ ├── default.ts # Default OpenAI-compatible executor
+│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum)
+│ │ ├── codex.ts # OpenAI Codex CLI
+│ │ ├── antigravity.ts # Antigravity IDE
+│ │ ├── github.ts # GitHub Copilot
+│ │ ├── gemini-cli.ts # Gemini CLI
+│ │ ├── kiro.ts # Kiro AI
+│ │ ├── qoder.ts # Qoder AI
+│ │ ├── vertex.ts # Vertex AI (Service Account JSON)
+│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI
+│ │ ├── opencode.ts # OpenCode Zen/Go
+│ │ ├── pollinations.ts # Pollinations AI
+│ │ └── puter.ts # Puter AI
+│ ├── handlers/ # Request handlers per API type (11 handlers)
+│ │ ├── chatCore.ts # Main chat completions handler
+│ │ ├── responsesHandler.ts # OpenAI Responses API handler
+│ │ ├── embeddings.ts # Embedding generation
+│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.)
+│ │ ├── videoGeneration.ts # Video generation
+│ │ ├── musicGeneration.ts # Music generation
+│ │ ├── audioSpeech.ts # Text-to-speech
+│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI)
+│ │ ├── moderations.ts # Content moderation
+│ │ ├── rerank.ts # Reranking API
+│ │ └── search.ts # Web search API
+│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP)
+│ │ ├── server.ts # MCP server core (tool registration, scope enforcement)
+│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools)
+│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a)
+│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes)
+│ │ ├── audit.ts # Tool call audit logging
+│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat
+│ │ └── httpTransport.ts # HTTP transport handler
+│ ├── services/ # 36+ service modules
+│ │ ├── combo.ts # Core routing engine
+│ │ ├── usage.ts # Usage tracking
+│ │ ├── tokenRefresh.ts # OAuth token refresh
+│ │ ├── rateLimitManager.ts # Rate limit management
+│ │ ├── accountFallback.ts # Multi-account fallback
+│ │ ├── sessionManager.ts # Session management
+│ │ ├── wildcardRouter.ts # Wildcard model routing
+│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration)
+│ │ ├── intentClassifier.ts # Request intent classification
+│ │ ├── taskAwareRouter.ts # Task-aware routing
+│ │ ├── thinkingBudget.ts # Thinking budget management
+│ │ ├── contextManager.ts # Context window management
+│ │ ├── modelDeprecation.ts # Model deprecation handling
+│ │ ├── modelFamilyFallback.ts # Intra-family model fallback
+│ │ ├── emergencyFallback.ts # Emergency fallback
+│ │ ├── workflowFSM.ts # Workflow state machine
+│ │ ├── backgroundTaskDetector.ts # Background task detection
+│ │ ├── ipFilter.ts # IP-based access control
+│ │ ├── signatureCache.ts # CLI signature caching
+│ │ ├── volumeDetector.ts # Request volume detection
+│ │ ├── contextHandoff.ts # Context relay handoff generation and injection
+│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay
+│ │ └── ... # Additional services (14 more modules)
+│ ├── transformer/ # Responses API transformer
+│ │ └── responsesTransformer.ts
+│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek)
+│ │ ├── request/ # Request translators per provider
+│ │ ├── response/ # Response translators per provider
+│ │ ├── helpers/ # Translation helpers
+│ │ └── image/ # Image format translation
+│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.)
+├── electron/ # Electron desktop app (cross-platform)
+│ ├── main.js # Electron main process
+│ ├── preload.js # Preload script (IPC bridge)
+│ └── assets/ # App icons and assets
+├── tests/ # Test suites
+│ ├── unit/ # 122 unit test files
+│ ├── integration/ # Integration tests
+│ ├── e2e/ # Playwright E2E tests
+│ ├── security/ # Security tests
+│ ├── translator/ # Translator-specific tests
+│ └── load/ # Load tests
+├── docs/ # Documentation
+│ ├── i18n/ # 30-language translated docs
+│ ├── ARCHITECTURE.md # Full architecture documentation
+│ ├── API_REFERENCE.md # API reference
+│ ├── USER_GUIDE.md # User guide
+│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview
+│ ├── CLI-TOOLS.md # CLI tools integration guide
+│ ├── A2A-SERVER.md # A2A agent protocol documentation
+│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring)
+│ ├── MCP-SERVER.md # MCP server (25 tools)
+│ ├── TROUBLESHOOTING.md # Troubleshooting guide
+│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide
+│ ├── openapi.yaml # OpenAPI specification
+│ └── screenshots/ # Dashboard screenshots
+├── bin/ # CLI entry points (omniroute, reset-password)
+├── scripts/ # Build and utility scripts
+└── .env.example # Environment variable template
+```
+
+## Key Features (v3.5.5)
+
+### Core Proxy
+- **60+ AI providers** with automatic format translation
+- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible)
+- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay
+- **4-tier fallback**: Subscription → API Key → Cheap → Free
+- **Context Relay strategy**: Session handoff summaries on account rotation for continuity
+- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown
+- **Semantic caching** with cache hit/miss headers
+- **Idempotency** with configurable dedup window
+- **Circuit breaker** per provider with configurable thresholds
+- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback
+- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers
+- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement
+- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization
+- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills
+- **Prompt Injection Guard**: Middleware-level prompt injection detection
+- **MITM Proxy**: Certificate management, DNS handling, and target routing
+- **Cloudflare Tunnels**: Managed tunnel creation for remote access
+- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches)
+
+### Turvallisuus
+- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs)
+- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()`
+- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()`
+- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection
+- **CLI Fingerprint Matching** — Per-provider request signature matching
+- **Prompt injection guard** — Request middleware detection
+- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`)
+- **PII sanitizer** — Sensitive data scrubbing in logs
+
+### Dashboard Pages (23 sections)
+- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons
+- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies
+- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics
+- **Analytics** — Token consumption, cost, heatmaps, distributions
+- **Health** — Uptime, memory, latency percentiles, circuit breakers
+- **Logs** — Request, Proxy, Audit, Console (tabbed)
+- **Audit** — Audit trail and compliance logging
+- **Costs** — Cost tracking per provider/model
+- **Limits** — Rate limit monitoring
+- **Cache** — Semantic cache statistics and management
+- **CLI Tools** — One-click configuration for 10+ AI CLI tools
+- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration
+- **Playground** — Test any model with Monaco editor, streaming responses
+- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files)
+- **Search Tools** — Search provider configuration and testing
+- **Memory** — Memory system management and visualization
+- **Skills** — Skills framework management and execution
+- **Translator** — Format debugging: playground, chat tester, test bench, live monitor
+- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced
+- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed)
+- **Onboarding** — Setup wizard for new users
+- **Usage** — Usage history and analytics
+- **API Manager** — API key management with scoped permissions
+
+### Protocol Support
+- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations`
+- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens`
+- **OpenAI Responses** — `/v1/responses`
+- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}`
+- **Ollama** — `/v1/api/chat`, `/api/tags`
+- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily)
+- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP)
+- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills)
+- **ACP** — Agent Communication Protocol registry and manager
+
+### MCP Server (25 Tools)
+| Category | Tools |
+|-----------|-------|
+| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` |
+| Memory (3) | `memory_search`, `memory_add`, `memory_clear` |
+| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` |
+
+**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience`
+
+### Provider Categories
+
+**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI
+
+**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline
+
+**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan
+
+**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs
+
+### Internationalization
+- 30 languages for UI (all dashboard pages)
+- 30 translated documentation sets in docs/i18n/
+- Language switcher in documentation
+
+## Key Architectural Decisions
+
+1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints.
+
+2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently.
+
+3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation.
+
+4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity.
+
+5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client.
+
+6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes.
+
+7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`).
+
+8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages.
+
+9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations.
+
+10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`.
+
+11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation.
+
+12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline.
+
+13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast.
+
+## Main Flows
+
+### Proxy Request Flow
+1. Client sends OpenAI-format request to `/v1/chat/completions`
+2. API key validation
+3. Model resolution: direct model or combo lookup
+4. For combos: iterate through models with selected strategy
+5. Auth resolution: get credentials for the target provider
+6. Format translation: OpenAI → provider native format
+7. CLI fingerprint matching (if enabled for provider)
+8. Upstream request with circuit breaker and rate limiting
+9. Response translation: provider → OpenAI format
+10. omniModel tag sanitization (strip internal tags)
+11. SSE streaming back to client
+12. Memory extraction (if memory system enabled)
+13. Usage logging and cost calculation
+
+### OAuth Flow
+1. Dashboard initiates `/api/oauth/[provider]/authorize`
+2. User completes OAuth login in browser
+3. Callback hits `/api/oauth/[provider]/exchange`
+4. Tokens stored as a provider connection in SQLite
+5. Background job refreshes tokens before expiry
+
+## Important Notes for LLMs
+
+1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only).
+
+2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`).
+
+3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services.
+
+4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`.
+
+5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module.
+
+6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`).
+
+7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes.
+
+8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB.
+
+9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown.
+
+10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130).
+
+11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux).
+
+12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint.
+
+13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API.
+
+14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API.
+
+15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time.
+
+16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`.
+
+17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22.
+
+18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false.
+
+## Links
+
+- Repository: https://github.com/diegosouzapw/OmniRoute
+- Website: https://omniroute.online
+- npm: https://www.npmjs.com/package/omniroute
+- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute
+- Documentation: See `/docs/` directory
diff --git a/docs/i18n/fr/README.md b/docs/i18n/fr/README.md
index 5399e19643..362bdd077a 100644
--- a/docs/i18n/fr/README.md
+++ b/docs/i18n/fr/README.md
@@ -4,11 +4,14 @@
---
+
### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback.
-_Votre proxy API universel : un point de terminaison, plus de 60 fournisseurs, aucun temps d'arrêt. Désormais avec**Serveur MCP (25 outils)**,**Protocole A2A**,**Systèmes de mémoire/compétences**et**Application de bureau Electron**._
+_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._
-**Fin de discussions • Intégrations • Génération d'images • Vidéo • Musique • Audio • Reclassement •**Recherche sur le Web**• Serveur MCP • Protocole A2A • 100 % TypeScript**---
+**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript**
+
+---
@@ -39,9 +42,13 @@ _Votre proxy API universel : un point de terminaison, plus de 60 fournisseurs,
[](https://omniroute.online)
[](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-[🌐 Site Web](https://omniroute.online) • [🚀 Démarrage rapide](#-démarrage rapide) • [💡 Fonctionnalités](#-fonctionnalités-clés) • [📖 Documents](#-documentation) • [💰 Tarification](#-tarification-en un coup d'œil) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-🌐**Disponible en :**🇺🇸 [anglais](README.md) | 🇧🇷 [Português (Brésil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italien](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonésie](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Pays-Bas](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Philippin](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)---
+
+
+🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)
+
+---
## 🖼️ Main Dashboard
@@ -53,29 +60,30 @@ _Votre proxy API universel : un point de terminaison, plus de 60 fournisseurs,
## 📸 Dashboard Preview
-
+
+Click to see dashboard screenshots
-Cliquez pour voir les captures d'écran du tableau de bord
+| Page | Screenshot |
+| -------------- | ------------------------------------------------- |
+| **Providers** |  |
+| **Combos** |  |
+| **Analytics** |  |
+| **Health** |  |
+| **Translator** |  |
+| **Settings** |  |
+| **CLI Tools** |  |
+| **Usage Logs** |  |
+| **Endpoints** |  |
-| Pages | Capture d'écran |
-| -------------------------- | ---------------------------------------------------------- | ---------- |
-| **Fournisseurs** |  |
-| **Combinaisons** |  |
-| **Analyses** |  |
-| **Santé** |  |
-| **Traducteur** |  |
-| **Paramètres** |  |
-| **Outils CLI** |  |
-| **Journaux d'utilisation** |  |
-| **Points de terminaison** |  | |
+
---
### 🤖 Free AI Provider for your favorite coding agents
-_Connectez n'importe quel outil IDE ou CLI alimenté par l'IA via OmniRoute — une passerelle API gratuite pour un codage illimité._
+_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._
-
+
@@ -89,28 +97,28 @@ _Connectez n'importe quel outil IDE ou CLI alimenté par l'IA via OmniRoute —

NanoBot
- ⭐ 20,9K
+ ⭐ 20.9K
|

- PicoGriffe
+ PicoClaw
- ⭐ 14,6K
+ ⭐ 14.6K
|

- ZéroClaw
+ ZeroClaw
- ⭐ 9,9K
+ ⭐ 9.9K
|

- Griffe de Fer
+ IronClaw
- ⭐ 2,1K
+ ⭐ 2.1K
|
@@ -124,483 +132,557 @@ _Connectez n'importe quel outil IDE ou CLI alimenté par l'IA via OmniRoute —

- CLI Codex
+ Codex CLI
- ⭐ 60,8K
+ ⭐ 60.8K
|

Claude Code
- ⭐ 67,3K
+ ⭐ 67.3K
|

- CLI Gemini
+ Gemini CLI
- ⭐ 94,7K
+ ⭐ 94.7K
|
- 
- Code kilo
+ 
+ Kilo Code
- ⭐ 15,5K
+ ⭐ 15.5K
|
-📡 Tous les agents se connectent via http://localhost:20128/v1 ou http://cloud.omniroute.online/v1 — une configuration, des modèles et un quota illimités---
+📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota
+
+---
## 🤔 Why OmniRoute?
-**Arrêtez de gaspiller de l'argent et d'atteindre vos limites :**
+**Stop wasting money and hitting limits:**
--
Le quota d'abonnement expire chaque mois sans être utilisé
--
Les limites de débit vous empêchent de coder
--
API coûteuses (20-50 $/mois par fournisseur)
--
Commutation manuelle entre les fournisseurs
+-
Subscription quota expires unused every month
+-
Rate limits stop you mid-coding
+-
Expensive APIs ($20-50/month per provider)
+-
Manual switching between providers
-**OmniRoute résout ce problème :**
+**OmniRoute solves this:**
-- ✅**Maximiser les abonnements**- Suivez le quota, utilisez chaque bit avant la réinitialisation
-- ✅**Repli automatique**- Abonnement → Clé API → Pas cher → Gratuit, aucun temps d'arrêt
-- ✅**Multi-compte**- Round-robin entre les comptes par fournisseur
-- ✅**Universel**- Fonctionne avec Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, n'importe quel outil CLI---
+- ✅ **Maximize subscriptions** - Track quota, use every bit before reset
+- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime
+- ✅ **Multi-account** - Round-robin between accounts per provider
+- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool
+
+---
## 📧 Support
-> 💬**Rejoignez notre communauté !**[Groupe WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Obtenez de l'aide, partagez des conseils et restez informé.
+> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated.
--**Site Internet**: [omniroute.online](https://omniroute.online) -**GitHub** : [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problèmes** : [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp** : [Groupe communautaire](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Contribuer** : voir [CONTRIBUTING.md](CONTRIBUTING.md), ouvrir un PR ou choisir un « bon premier numéro » -**Projet original** : [9router par decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug?
+- **Website**: [omniroute.online](https://omniroute.online)
+- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute)
+- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue`
+- **Original Project**: [9router by decolua](https://github.com/decolua/9router)
-Lors de l'ouverture d'un ticket, veuillez exécuter la commande system-info et joindre le fichier généré :```bash
+### 🐛 Reporting a Bug?
+
+When opening an issue, please run the system-info command and attach the generated file:
+
+```bash
npm run system-info
-
```
-Cela génère un `system-info.txt` avec votre version de Node.js, la version d'OmniRoute, les détails du système d'exploitation, les outils CLI installés (qoder, gemini, claude, codex, antigravity, droid, etc.), l'état Docker/PM2 et les packages système — tout ce dont nous avons besoin pour reproduire rapidement votre problème. Joignez le fichier directement à votre problème GitHub.---
+This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue.
+
+---
## 🔄 How It Works
```
-
┌─────────────┐
-│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
-│ Tool │
+│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
+│ Tool │
└──────┬──────┘
-│ http://localhost:20128/v1
-↓
+ │ http://localhost:20128/v1
+ ↓
┌─────────────────────────────────────────┐
-│ OmniRoute (Smart Router) │
-│ • Format translation (OpenAI ↔ Claude) │
-│ • Quota tracking + Embeddings + Images │
-│ • Auto token refresh │
+│ OmniRoute (Smart Router) │
+│ • Format translation (OpenAI ↔ Claude) │
+│ • Quota tracking + Embeddings + Images │
+│ • Auto token refresh │
└──────┬──────────────────────────────────┘
-│
-├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
-│ ↓ quota exhausted
-├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
-│ ↓ budget limit
-├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
-│ ↓ budget limit
-└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
+ │
+ ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
+ │ ↓ quota exhausted
+ ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
+ │ ↓ budget limit
+ ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
+ │ ↓ budget limit
+ └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
Result: Never stop coding, minimal cost
-
-````
+```
---
## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases
->**Tous les développeurs utilisant des outils d'IA sont confrontés quotidiennement à ces problèmes.**OmniRoute a été conçu pour tous les résoudre : des dépassements de coûts aux blocages régionaux, des flux OAuth interrompus aux opérations de protocole et à l'observabilité de l'entreprise.
+> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability.
-
-💸 1. "Je paie un abonnement coûteux mais je suis quand même interrompu par des limites"
+
+💸 1. "I pay for an expensive subscription but still get interrupted by limits"
-Les développeurs paient entre 20 et 200 $/mois pour Claude Pro, Codex Pro ou GitHub Copilot. Même payant, le quota est plafonné : 5 heures d'utilisation, limites hebdomadaires ou limites de tarif à la minute. En cours de session de codage, le fournisseur ne répond plus et le développeur perd en fluidité et en productivité.
+Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity.
-**Comment OmniRoute le résout :**
+**How OmniRoute solves it:**
--**Smart 4-Tier Fallback**— Si le quota d'abonnement est épuisé, redirige automatiquement vers la clé API → Pas cher → Gratuit sans intervention manuelle
--**Suivi des limites du fournisseur**— Actualisation des instantanés de quotas mis en cache selon une planification côté serveur (par défaut `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) avec actualisation manuelle disponible dans l'interface utilisateur
--**Support multi-comptes**— Plusieurs comptes par fournisseur avec tourniquet automatique — lorsqu'un compte est épuisé, passe au suivant
--**Combos personnalisés**— Chaînes de secours personnalisables avec 9 stratégies d'équilibrage (priorité, pondérée, remplissage en premier, round-robin, P2C, aléatoire, la moins utilisée, optimisée en termes de coût, strictement aléatoire)
--**Codex Business Quotas**— Surveillance des quotas d'espace de travail Business/Équipe directement dans le tableau de bord
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention
+- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI
+- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next
+- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**)
+- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard
-
-🔌 2. "Je dois utiliser plusieurs fournisseurs mais chacun a une API différente"
+
-OpenAI utilise un format, Claude (Anthropic) en utilise un autre, Gemini encore un autre. Si un développeur souhaite tester des modèles de différents fournisseurs ou utiliser un modèle de secours entre eux, il doit reconfigurer les SDK, modifier les points de terminaison et gérer les formats incompatibles. Les fournisseurs personnalisés (FriendLI, NIM) ont des points de terminaison de modèle non standard.
+
+🔌 2. "I need to use multiple providers but each has a different API"
-**Comment OmniRoute le résout :**
+OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints.
--**Unified Endpoint**— Un seul « http://localhost:20128/v1 » sert de proxy pour plus de 60 fournisseurs.
--**Traduction de format**— Automatique et transparente : OpenAI ↔ Claude ↔ Gemini ↔ API Responses
--**Response Sanitization**— Supprime les champs non standard (`x_groq`, `usage_breakdown`, `service_tier`) qui cassent OpenAI SDK v1.83+
--**Role Normalization**— Convertit « développeur » → « système » pour les fournisseurs non OpenAI ; `système` → `utilisateur` pour GLM/ERNIE
--**Think Tag Extraction**— Extrait les blocs `` de modèles comme DeepSeek R1 dans un `reasoning_content` standardisé
--**Sortie structurée pour Gemini**— Conversion automatique `json_schema` → `responseMimeType`/`responseSchema`
--**`stream` est par défaut `false`**— S'aligne sur les spécifications OpenAI, évitant ainsi le SSE inattendu dans les SDK Python/Rust/Go
+**How OmniRoute solves it:**
-
-🌐 3. "Mon fournisseur d'IA bloque ma région/mon pays"
+- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers
+- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
+- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+
+- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE
+- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content`
+- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion
+- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs
-Des fournisseurs comme OpenAI/Codex bloquent l’accès depuis certaines régions géographiques. Les utilisateurs obtiennent des erreurs telles que « unsupported_country_region_territory » lors des connexions OAuth et API. Ceci est particulièrement frustrant pour les développeurs des pays en développement.
+
-**Comment OmniRoute le résout :**
+
+🌐 3. "My AI provider blocks my region/country"
--**Configuration proxy à 3 niveaux**— Proxy configurable à 3 niveaux : global (tout le trafic), par fournisseur (un seul fournisseur) et par connexion/clé
--**Badges proxy à code couleur**— Indicateurs visuels : 🟢 proxy global, 🟡 proxy fournisseur, 🔵 proxy de connexion, affichant toujours l'adresse IP
--**Échange de jetons OAuth via proxy**— Le flux OAuth passe également par le proxy, résolvant `unsupported_country_region_territory`
--**Tests de connexion via proxy**— Les tests de connexion utilisent le proxy configuré (plus de contournement direct)
--**Support SOCKS5**— Prise en charge complète du proxy SOCKS5 pour le routage sortant
--**TLS Fingerprint Spoofing**— Empreinte digitale TLS de type navigateur via `wreq-js` pour contourner la détection des robots
--**🔏 Correspondance d'empreintes digitales CLI**— Réorganise les en-têtes et les champs de corps pour qu'ils correspondent aux signatures binaires CLI natives, réduisant ainsi considérablement le risque de signalement de compte. L'adresse IP du proxy est préservée : vous bénéficiez simultanément du masquage furtif**et**IP
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries.
-
-🆓 4. "Je veux utiliser l'IA pour coder mais je n'ai pas d'argent"
+**How OmniRoute solves it:**
-Tout le monde ne peut pas payer entre 20 et 200 $/mois pour des abonnements à l’IA. Les étudiants, les développeurs des pays émergents, les amateurs et les indépendants doivent avoir accès à des modèles de qualité à un coût nul.
+- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key
+- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP
+- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory`
+- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass)
+- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection
+- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously
-**Comment OmniRoute le résout :**
+
--**Fournisseurs gratuits intégrés**— Prise en charge native des fournisseurs 100 % gratuits : Qoder (5 modèles illimités via OAuth : kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 modèles illimités : qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + ID AWS Builder gratuits), Gemini CLI (180 000 jetons/mois gratuits)
--**Ollama Cloud**— Modèles Ollama hébergés dans le cloud sur « api.ollama.com » avec niveau gratuit « Utilisation légère » ; utilisez le préfixe `ollamacloud/`
--**Combos gratuits uniquement**— Chaîne `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 $/mois sans temps d'arrêt
--**NVIDIA NIM Free Access**— ~ 40 RPM d'accès gratuit pour toujours à plus de 70 modèles sur build.nvidia.com (passage des crédits aux limites de débit pures)
--**Stratégie d'optimisation des coûts**— Stratégie de routage qui choisit automatiquement le fournisseur disponible le moins cher
+
+🆓 4. "I want to use AI for coding but I have no money"
-
-🔒 5. "Je dois protéger ma passerelle IA contre tout accès non autorisé"
+Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost.
-Lors de l'exposition d'une passerelle IA au réseau (LAN, VPS, Docker), toute personne possédant l'adresse peut consommer les jetons/quota du développeur. Sans protection, les API sont vulnérables aux utilisations abusives, aux injections rapides et aux abus.
+**How OmniRoute solves it:**
-**Comment OmniRoute le résout :**
+- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free)
+- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix
+- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime
+- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits)
+- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider
--**Gestion des clés API**— Génération, rotation et portée par fournisseur avec une page dédiée `/dashboard/api-manager`
--**Autorisations au niveau du modèle**— Restreindre les clés API à des modèles spécifiques (`openai/*`, modèles génériques), avec la bascule Autoriser tout/Restreindre
--**API Endpoint Protection**— Exiger une clé pour `/v1/models` et bloquer des fournisseurs spécifiques de la liste
--**Auth Guard + Protection CSRF**— Toutes les routes du tableau de bord protégées avec le middleware `withAuth` + les jetons CSRF
--**Rate Limiter**— Limitation du débit par IP avec fenêtres configurables
--**Filtrage IP** – Liste autorisée/liste de blocage pour le contrôle d'accès
--**Prompt Injection Guard**— Nettoyage contre les modèles d'invite malveillants
--**Chiffrement AES-256-GCM**— Informations d'identification chiffrées au repos
+
-
-🛑 6. "Mon fournisseur est tombé en panne et j'ai perdu mon flux de codage"
+
+🔒 5. "I need to protect my AI gateway from unauthorized access"
-Les fournisseurs d’IA peuvent devenir instables, renvoyer des erreurs 5xx ou atteindre des limites de débit temporaires. Si un développeur dépend d'un seul fournisseur, il est interrompu. Sans disjoncteurs, des tentatives répétées peuvent faire planter l’application.
+When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse.
-**Comment OmniRoute le résout :**
+**How OmniRoute solves it:**
--**Disjoncteur par modèle**— Ouverture/fermeture automatique avec seuils et temps de recharge configurables (Fermé/Ouvert/Semi-ouvert), limités par modèle pour éviter les blocages en cascade
--**Exponential Backoff**— Délais progressifs entre les tentatives
--**Anti-Thundering Herd**— Protection mutex + sémaphore contre les tempêtes de nouvelles tentatives simultanées
--**Chaînes de secours combinées**— Si le fournisseur principal échoue, passe automatiquement à travers la chaîne sans intervention
--**Combo Circuit Breaker** – Désactive automatiquement les fournisseurs défaillants au sein d'une chaîne combo
--**Tableau de bord de santé**— Surveillance de la disponibilité, états des disjoncteurs, verrouillages, statistiques du cache, latence p50/p95/p99
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page
+- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle
+- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing
+- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens
+- **Rate Limiter** — Per-IP rate limiting with configurable windows
+- **IP Filtering** — Allowlist/blocklist for access control
+- **Prompt Injection Guard** — Sanitization against malicious prompt patterns
+- **AES-256-GCM Encryption** — Credentials encrypted at rest
-
-🔧 7. "La configuration de chaque outil d'IA est fastidieuse et répétitive"
+
-Les développeurs utilisent Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Chaque outil nécessite une configuration différente (point de terminaison API, clé, modèle). La reconfiguration lors du changement de fournisseur ou de modèle est une perte de temps.
+
+🛑 6. "My provider went down and I lost my coding flow"
-**Comment OmniRoute le résout :**
+AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application.
--**CLI Tools Dashboard**— Page dédiée avec configuration en un clic pour Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
--**GitHub Copilot Config Generator**— Génère `chatLanguageModels.json` pour VS Code avec sélection groupée de modèles
--**Assistant d'intégration**— Configuration guidée en 4 étapes pour les nouveaux utilisateurs
--**Un point de terminaison, tous les modèles**— Configurez `http://localhost:20128/v1` une fois, accédez à plus de 60 fournisseurs
+**How OmniRoute solves it:**
-
-🔑 8. "Gérer les jetons OAuth de plusieurs fournisseurs est un enfer"
+- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks
+- **Exponential Backoff** — Progressive retry delays
+- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms
+- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention
+- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain
+- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency
-Claude Code, Codex, Gemini CLI, Copilot — tous utilisent OAuth 2.0 avec des jetons expirant. Les développeurs doivent se réauthentifier constamment, gérer « client_secret manquant », « redirect_uri_mismatch » et les échecs sur les serveurs distants. OAuth sur LAN/VPS est particulièrement problématique.
+
-**Comment OmniRoute le résout :**
+
+🔧 7. "Configuring each AI tool is tedious and repetitive"
--**Actualisation automatique des jetons** : les jetons OAuth sont actualisés en arrière-plan avant leur expiration.
--**OAuth 2.0 (PKCE) intégré**— Flux automatique pour Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
--**Multi-Account OAuth**— Plusieurs comptes par fournisseur via l'extraction de jetons JWT/ID
--**OAuth LAN/Remote Fix**— Détection d'adresse IP privée pour `redirect_uri` + mode URL manuel pour les serveurs distants
--**OAuth derrière Nginx**— Utilise `window.location.origin` pour la compatibilité du proxy inverse
--**Guide OAuth à distance**— Guide étape par étape pour les informations d'identification Google Cloud sur VPS/Docker
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time.
-
-📊 9. "Je ne sais pas combien je dépense ni où"
+**How OmniRoute solves it:**
-Les développeurs utilisent plusieurs fournisseurs payants mais n'ont pas de vue unifiée des dépenses. Chaque fournisseur dispose de son propre tableau de bord de facturation, mais il n'existe pas de vue consolidée. Les coûts inattendus peuvent s’accumuler.
+- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
+- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection
+- **Onboarding Wizard** — Guided 4-step setup for first-time users
+- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers
-**Comment OmniRoute le résout :**
+
--**Cost Analytics Dashboard**— Suivi des coûts par jeton et gestion du budget par fournisseur
--**Limites budgétaires par niveau**— Plafond de dépenses par niveau qui déclenche un repli automatique
--**Configuration de tarification par modèle**— Prix configurables par modèle
--**Statistiques d'utilisation par clé API**— Nombre de demandes et horodatage de la dernière utilisation par clé
--**Tableau de bord Analytics**— Cartes statistiques, tableau d'utilisation du modèle, tableau des fournisseurs avec taux de réussite et latence
+
+🔑 8. "Managing OAuth tokens from multiple providers is hell"
-
-🐛 10. "Je ne peux pas diagnostiquer les erreurs et les problèmes dans les appels IA"
+Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic.
-Lorsqu'un appel échoue, le développeur ne sait pas s'il s'agit d'une limite de débit, d'un jeton expiré, d'un format incorrect ou d'une erreur du fournisseur. Journaux fragmentés sur différents terminaux. Sans observabilité, le débogage est un essai et une erreur.
+**How OmniRoute solves it:**
-**Comment OmniRoute le résout :**
+- **Auto Token Refresh** — OAuth tokens refresh in background before expiration
+- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
+- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction
+- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers
+- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility
+- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker
--**Tableau de bord des journaux unifiés**— 4 onglets : journaux de requêtes, journaux proxy, journaux d'audit, console
--**Console Log Viewer**— Visualiseur de style terminal en temps réel avec niveaux de code couleur, défilement automatique, recherche, filtre
--**Journaux du proxy SQLite**— Journaux persistants qui survivent aux redémarrages du serveur
--**Translator Playground**— 4 modes de débogage : Playground (traduction de format), Chat Tester (aller-retour), Test Bench (batch), Live Monitor (temps réel)
--**Demande de télémétrie**— latence p50/p95/p99 + traçage X-Request-Id
--**Journalisation basée sur des fichiers avec rotation**— Les journaux d'applications alternent en fonction de la taille, des jours de conservation et du nombre d'archives ; les artefacts du journal des appels alternent en fonction des jours de conservation et du nombre de fichiers
--**Rapport d'informations système**— `npm run system-info` génère `system-info.txt` avec votre environnement complet (version Node, version OmniRoute, système d'exploitation, outils CLI, statut Docker/PM2). Joignez-le lorsque vous signalez des problèmes pour un tri instantané.
+
-
-🏗️ 11. "Le déploiement et la maintenance de la passerelle sont complexes"
+
+📊 9. "I don't know how much I'm spending or where"
-L'installation, la configuration et la maintenance d'un proxy IA dans différents environnements (local, VPS, Docker, cloud) demandent beaucoup de main-d'œuvre. Des problèmes tels que les chemins codés en dur, les « EACCES » sur les répertoires, les conflits de ports et les versions multiplateformes ajoutent des frictions.
+Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up.
-**Comment OmniRoute le résout :**
+**How OmniRoute solves it:**
--**npm global install**— `npm install -g omniroute && omniroute` — terminé
--**Docker Multi-Platform**— AMD64 + ARM64 natif (Apple Silicon, AWS Graviton, Raspberry Pi)
--**Profils Docker Compose**— `base` (pas d'outils CLI) et `cli` (avec Claude Code, Codex, OpenClaw)
--**Electron Desktop App**— Application native pour Windows/macOS/Linux avec barre d'état système, démarrage automatique et mode hors ligne
--**Mode Split-Port**— API et tableau de bord sur des ports séparés pour des scénarios avancés (proxy inverse, réseau de conteneurs)
--**Cloud Sync** – Configurez la synchronisation entre les appareils via Cloudflare Workers
--**Sauvegardes DB**— Sauvegarde, restauration, exportation et importation automatiques de tous les paramètres, avec `DISABLE_SQLITE_AUTO_BACKUP` pour les sauvegardes gérées en externe
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider
+- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback
+- **Per-Model Pricing Configuration** — Configurable prices per model
+- **Usage Statistics Per API Key** — Request count and last-used timestamp per key
+- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency
-
-🌍 12. "L'interface est uniquement en anglais et mon équipe ne parle pas anglais"
+
-Les équipes des pays non anglophones, notamment en Amérique latine, en Asie et en Europe, ont du mal à utiliser des interfaces uniquement en anglais. Les barrières linguistiques réduisent l’adoption et augmentent les erreurs de configuration.
+
+🐛 10. "I can't diagnose errors and problems in AI calls"
-**Comment OmniRoute le résout :**
+When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error.
--**Tableau de bord i18n — 30 langues**— Plus de 500 touches traduites, dont arabe, bulgare, danois, allemand, espagnol, finnois, français, hébreu, hindi, hongrois, indonésien, italien, japonais, coréen, malais, néerlandais, norvégien, polonais, portugais (PT/BR), roumain, russe, slovaque, suédois, thaï, ukrainien, vietnamien, chinois, philippin, anglais.
--**Support RTL**— Prise en charge de droite à gauche pour l'arabe et l'hébreu
--**README multilingues**— 30 traductions complètes de la documentation
--**Sélecteur de langue**— Icône de globe dans l'en-tête pour une commutation en temps réel
+**How OmniRoute solves it:**
-
-🔄 13. "J'ai besoin de plus que du chat : j'ai besoin d'intégrations, d'images, d'audio"
+- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console
+- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter
+- **SQLite Proxy Logs** — Persistent logs that survive server restarts
+- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time)
+- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing
+- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count
+- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage.
-L'IA ne se limite pas à la réalisation de discussions. Les développeurs doivent générer des images, transcrire l'audio, créer des intégrations pour RAG, reclasser les documents et modérer le contenu. Chaque API a un point de terminaison et un format différents.
+
-**Comment OmniRoute le résout :**
+
+🏗️ 11. "Deploying and maintaining the gateway is complex"
--**Embeddings**— `/v1/embeddings` avec 6 fournisseurs et plus de 9 modèles
--**Génération d'images**— `/v1/images/generations` avec 10 fournisseurs et plus de 20 modèles (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
--**Text-to-Video**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) et SD WebUI
--**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
--**Transcription audio**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
--**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + fournisseurs existants
--**Modérations**— `/v1/moderations` — Contrôles de sécurité du contenu
--**Reclassement**— `/v1/rerank` — Reclassement de la pertinence du document
--**API Responses**— Prise en charge complète de `/v1/responses` pour le Codex
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction.
-
-🧪 14. "Je n'ai aucun moyen de tester et de comparer la qualité des différents modèles"
+**How OmniRoute solves it:**
-Les développeurs veulent savoir quel modèle convient le mieux à leur cas d'utilisation (code, traduction, raisonnement) mais la comparaison manuelle est lente. Il n’existe aucun outil d’évaluation intégré.
+- **npm global install** — `npm install -g omniroute && omniroute` — done
+- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi)
+- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw)
+- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode
+- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking)
+- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers
+- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups
-**Comment OmniRoute le résout :**
+
--**Évaluations LLM**— Tests Golden Set avec 10 cas préchargés couvrant les salutations, les mathématiques, la géographie, la génération de code, la conformité JSON, la traduction, la démarque, le refus de sécurité
--**4 stratégies de correspondance**— `exact`, `contains`, `regex`, `custom` (fonction JS)
--**Banc de test Translator Playground**— Tests par lots avec plusieurs entrées et sorties attendues, comparaison entre fournisseurs
--**Chat Tester**— Aller-retour complet avec rendu de réponse visuelle
--**Live Monitor**— Flux en temps réel de toutes les requêtes transitant par le proxy
+
+🌍 12. "The interface is English-only and my team doesn't speak English"
-
-📈 15. "J'ai besoin d'évoluer sans perdre en performances"
+Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors.
-À mesure que le volume de demandes augmente, sans mettre en cache les mêmes questions, cela génère des coûts en double. Sans idempotence, les demandes en double gaspillent le traitement. Les limites tarifaires par fournisseur doivent être respectées.
+**How OmniRoute solves it:**
-**Comment OmniRoute le résout :**
+- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English
+- **RTL Support** — Right-to-left support for Arabic and Hebrew
+- **Multi-Language READMEs** — 30 complete documentation translations
+- **Language Selector** — Globe icon in header for real-time switching
--**Cache sémantique**— Le cache à deux niveaux (signature + sémantique) réduit les coûts et la latence
--**Request Idempotency**— Fenêtre de déduplication de 5 s pour des requêtes identiques
--**Détection de limite de débit**— RPM par fournisseur, écart minimum et suivi simultané maximum
--**Limites de débit modifiables**— Valeurs par défaut configurables dans Paramètres → Résilience avec persistance
--**Cache de validation de clé API**— Cache à 3 niveaux pour les performances de production
--**Tableau de bord de santé avec télémétrie**— latence p50/p95/p99, statistiques de cache, disponibilité
+
-
-🤖 16. "Je veux contrôler le comportement du modèle à l'échelle mondiale"
+
+🔄 13. "I need more than chat — I need embeddings, images, audio"
-Les développeurs qui souhaitent que toutes les réponses soient dans une langue spécifique, avec un ton spécifique, ou qui souhaitent limiter les jetons de raisonnement. Configurer cela dans chaque outil/demande n’est pas pratique.
+AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format.
-**Comment OmniRoute le résout :**
+**How OmniRoute solves it:**
--**Injection d'invite système**— Invite globale appliquée à toutes les requêtes
--**Thinking Budget Validation**— Contrôle d'allocation de jetons de raisonnement par requête (passthrough, automatique, personnalisé, adaptatif)
--**9 stratégies de routage** – Stratégies globales qui déterminent la manière dont les demandes sont distribuées
--**Wildcard Router**— Les modèles `provider/*` acheminent dynamiquement vers n'importe quel fournisseur
--**Combo Enable/Disable Toggle**— Basculez les combos directement depuis le tableau de bord
--**Provider Toggle**— Activer/désactiver toutes les connexions pour un fournisseur en un seul clic
--**Fournisseurs bloqués**— Exclure des fournisseurs spécifiques de la liste `/v1/models`
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models
+- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
+- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI
+- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
+- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
+- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers
+- **Moderations** — `/v1/moderations` — Content safety checks
+- **Reranking** — `/v1/rerank` — Document relevance reranking
+- **Responses API** — Full `/v1/responses` support for Codex
-
-🧰 17. "J'ai besoin d'outils MCP en tant que fonctionnalités de produit de premier ordre"
+
-De nombreuses passerelles IA exposent MCP uniquement en tant que détail d'implémentation caché. Les équipes ont besoin d’une couche opérationnelle visible et gérable.
+
+🧪 14. "I have no way to test and compare quality across models"
-**Comment OmniRoute le résout :**
+Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist.
-- MCP apparaît dans l'onglet de navigation du tableau de bord et de protocole de point de terminaison
-- Page de gestion MCP dédiée avec processus, outils, portées et audit
-- Démarrage rapide intégré pour `omniroute --mcp` et l'intégration des clients
+**How OmniRoute solves it:**
-
-🧠 18. "J'ai besoin d'une orchestration A2A avec des chemins de tâches de synchronisation et de flux"
+- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal
+- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function)
+- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison
+- **Chat Tester** — Full round-trip with visual response rendering
+- **Live Monitor** — Real-time stream of all requests flowing through the proxy
-Les flux de travail des agents nécessitent à la fois des réponses directes et une exécution en continu de longue durée avec contrôle du cycle de vie.
+
-**Comment OmniRoute le résout :**
+
+📈 15. "I need to scale without losing performance"
-- Point de terminaison A2A JSON-RPC (`POST /a2a`) avec `message/send` et `message/stream`
-- Streaming SSE avec propagation de l'état terminal
-- API de cycle de vie des tâches pour "tasks/get" et "tasks/cancel"
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected.
-
-🛰️ 19. "J'ai besoin d'un véritable état de santé du processus MCP, et non d'un état deviné"
+**How OmniRoute solves it:**
-Les équipes opérationnelles doivent savoir si MCP est réellement actif, et pas seulement si une API est accessible.
+- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency
+- **Request Idempotency** — 5s deduplication window for identical requests
+- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking
+- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence
+- **API Key Validation Cache** — 3-tier cache for production performance
+- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime
-**Comment OmniRoute le résout :**
+
-- Fichier de battement de cœur d'exécution avec PID, horodatages, transport, nombre d'outils et mode de portée
-- API de statut MCP combinant battement de coeur + activité récente
-- Cartes d'état de l'interface utilisateur pour la fraîcheur des processus/disponibilité/battement de cœur
+
+🤖 16. "I want to control model behavior globally"
-
-📋 20. "J'ai besoin d'une exécution vérifiable de l'outil MCP"
+Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical.
-Lorsque les outils modifient la configuration ou déclenchent des actions opérationnelles, les équipes ont besoin d'une traçabilité médico-légale.
+**How OmniRoute solves it:**
-**Comment OmniRoute le résout :**
+- **System Prompt Injection** — Global prompt applied to all requests
+- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive)
+- **9 Routing Strategies** — Global strategies that determine how requests are distributed
+- **Wildcard Router** — `provider/*` patterns route dynamically to any provider
+- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard
+- **Provider Toggle** — Enable/disable all connections for a provider with one click
+- **Blocked Providers** — Exclude specific providers from `/v1/models` listing
-- Journalisation d'audit basée sur SQLite pour les appels d'outils MCP
-- Filtres par outil, succès/échec, clé API et pagination
-- Tableau d'audit du tableau de bord + points de terminaison de statistiques pour l'automatisation
+
-
-🔐 21. "J'ai besoin d'autorisations MCP limitées par intégration"
+
+🧰 17. "I need MCP tools as first-class product capabilities"
-Différents clients doivent avoir le moindre privilège d’accès aux catégories d’outils.
+Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer.
-**Comment OmniRoute le résout :**
+**How OmniRoute solves it:**
-- 10 étendues MCP granulaires pour un accès contrôlé aux outils
-- Application de la portée et visibilité dans l'interface utilisateur de gestion MCP
-- Posture par défaut sécurisée pour les outils opérationnels
+- MCP appears in the dashboard navigation and endpoint protocol tab
+- Dedicated MCP management page with process, tools, scopes, and audit
+- Built-in quick-start for `omniroute --mcp` and client onboarding
-
-⚙️ 22. "J'ai besoin de contrôles opérationnels sans redéploiement"
+
-Les équipes ont besoin de changements d'exécution rapides lors d'incidents ou d'événements de coûts.
+
+🧠 18. "I need A2A orchestration with sync + stream task paths"
-**Comment OmniRoute le résout :**
+Agent workflows need both direct replies and long-running streamed execution with lifecycle control.
-- Activer le combo de commutation directement depuis le tableau de bord MCP
-- Appliquer des profils de résilience à partir de packs de politiques prédéfinis
-- Réinitialiser l'état du disjoncteur à partir du même panneau de commande
+**How OmniRoute solves it:**
-
-🔄 23. "J'ai besoin d'une visibilité et d'une annulation en direct du cycle de vie des tâches A2A"
+- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream`
+- SSE streaming with terminal state propagation
+- Task lifecycle APIs for `tasks/get` and `tasks/cancel`
-Sans visibilité sur le cycle de vie, les incidents de tâches deviennent difficiles à trier.
+
-**Comment OmniRoute le résout :**
+
+🛰️ 19. "I need real MCP process health, not guessed status"
-- Liste des tâches/filtrage par état/compétence avec pagination
-- Analyse approfondie des métadonnées, des événements et des artefacts des tâches
-- Point de terminaison d'annulation de tâche et action de l'interface utilisateur avec confirmation
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable.
-
-🌊 24. "J'ai besoin de métriques de flux actif pour la charge A2A"
+**How OmniRoute solves it:**
-Les flux de travail de streaming nécessitent une vision opérationnelle de la concurrence et des connexions en direct.
+- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode
+- MCP status API combining heartbeat + recent activity
+- UI status cards for process/uptime/heartbeat freshness
-**Comment OmniRoute le résout :**
+
-- Compteurs de flux actifs intégrés au statut A2A
-- Horodatage de la dernière tâche et nombre par état
-- Cartes de tableau de bord A2A pour la surveillance des opérations en temps réel
+
+📋 20. "I need auditable MCP tool execution"
-
-🪪 25. "J'ai besoin d'une découverte d'agent standard pour les clients"
+When tools mutate config or trigger ops actions, teams need forensic traceability.
-Les clients et orchestrateurs externes ont besoin de métadonnées lisibles par machine pour l'intégration.
+**How OmniRoute solves it:**
-**Comment OmniRoute le résout :**
+- SQLite-backed audit logging for MCP tool calls
+- Filters by tool, success/failure, API key, and pagination
+- Dashboard audit table + stats endpoints for automation
-- Carte d'agent exposée dans `/.well-known/agent.json`
-- Capacités et compétences affichées dans l'interface utilisateur de gestion
-- L'API de statut A2A inclut des métadonnées de découverte pour l'automatisation
+
-
-🧭 26. "J'ai besoin de la possibilité de découvrir le protocole dans l'UX du produit"
+
+🔐 21. "I need scoped MCP permissions per integration"
-Si les utilisateurs ne peuvent pas découvrir les surfaces de protocole, l’adoption et la qualité du support chutent.
+Different clients should have least-privilege access to tool categories.
-**Comment OmniRoute le résout :**
+**How OmniRoute solves it:**
-- Page**Points de terminaison**consolidée avec des onglets pour les points de terminaison Proxy, MCP, A2A et API
-- Basculement de l'état du service en ligne (en ligne/hors ligne) pour MCP et A2A
-- Liens depuis l'aperçu vers les onglets de gestion dédiés
+- 10 granular MCP scopes for controlled tool access
+- Scope enforcement and visibility in MCP management UI
+- Safe default posture for operational tooling
-
-🧪 27. "J'ai besoin d'une validation de protocole de bout en bout avec de vrais clients"
+
-Les tests simulés ne suffisent pas pour valider la compatibilité des protocoles avant la publication.
+
+⚙️ 22. "I need operational controls without redeploying"
-**Comment OmniRoute le résout :**
+Teams need quick runtime changes during incidents or cost events.
-- Suite E2E qui démarre l'application et utilise un véritable transport client MCP SDK
-- Tests client A2A pour les flux de découverte, d'envoi, de streaming, d'obtention et d'annulation
-- Vérifier les assertions par rapport aux API d'audit MCP et de tâches A2A
+**How OmniRoute solves it:**
-
-📡 28. "J'ai besoin d'une observabilité unifiée sur toutes les interfaces"
+- Switch combo activation directly from MCP dashboard
+- Apply resilience profiles from pre-defined policy packs
+- Reset circuit breaker state from the same operations panel
-Le fractionnement de l'observabilité par protocole crée des angles morts et un MTTR plus long.
+
-**Comment OmniRoute le résout :**
+
+🔄 23. "I need live A2A task lifecycle visibility and cancellation"
-- Tableaux de bord/journaux/analyses unifiés dans un seul produit
-- Santé + audit + télémétrie des demandes sur les couches OpenAI, MCP et A2A
-- API opérationnelles pour le statut et l'automatisation
+Without lifecycle visibility, task incidents become hard to triage.
-
-💼 29. "J'ai besoin d'un environnement d'exécution pour l'orchestration proxy + outils + agent"
+**How OmniRoute solves it:**
-L’exécution de nombreux services distincts augmente les coûts opérationnels et les modes de défaillance.
+- Task listing/filtering by state/skill with pagination
+- Drill-down on task metadata, events, and artifacts
+- Task cancellation endpoint and UI action with confirmation
-**Comment OmniRoute le résout :**
+
-- Proxy compatible OpenAI, serveur MCP et serveur A2A dans une seule pile
-- Authentification partagée, résilience, stockage de données et observabilité
-- Modèle de politique cohérent sur toutes les surfaces d'interaction
+
+🌊 24. "I need active stream metrics for A2A load"
-
-🚀 30. "J'ai besoin d'expédier des flux de travail agentiques sans prolifération de code collant"
+Streaming workflows require operational insight into concurrency and live connections.
-Les équipes perdent de la vitesse lors de l’assemblage de plusieurs services et scripts ad hoc.
+**How OmniRoute solves it:**
-**Comment OmniRoute le résout :**
+- Active stream counters integrated into A2A status
+- Last task timestamp and per-state counts
+- A2A dashboard cards for real-time ops monitoring
-- Stratégie de point de terminaison unifiée pour les clients et les agents
-- Interfaces utilisateur de gestion de protocole intégrées et chemins de validation de fumée
-- Bases prêtes pour la production (sécurité, journalisation, résilience, sauvegarde)
+
+
+
+🪪 25. "I need standard agent discovery for clients"
+
+External clients and orchestrators need machine-readable metadata for onboarding.
+
+**How OmniRoute solves it:**
+
+- Agent Card exposed at `/.well-known/agent.json`
+- Capabilities and skills shown in management UI
+- A2A status API includes discovery metadata for automation
+
+
+
+
+🧭 26. "I need protocol discoverability in the product UX"
+
+If users cannot discover protocol surfaces, adoption and support quality drop.
+
+**How OmniRoute solves it:**
+
+- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints
+- Inline service status toggles (Online/Offline) for MCP and A2A
+- Links from overview to dedicated management tabs
+
+
+
+
+🧪 27. "I need end-to-end protocol validation with real clients"
+
+Mock tests are not enough to validate protocol compatibility before release.
+
+**How OmniRoute solves it:**
+
+- E2E suite that boots app and uses real MCP SDK client transport
+- A2A client tests for discovery, send, stream, get, and cancel flows
+- Cross-check assertions against MCP audit and A2A tasks APIs
+
+
+
+
+📡 28. "I need unified observability across all interfaces"
+
+Splitting observability by protocol creates blind spots and longer MTTR.
+
+**How OmniRoute solves it:**
+
+- Unified dashboards/logs/analytics in one product
+- Health + audit + request telemetry across OpenAI, MCP, and A2A layers
+- Operational APIs for status and automation
+
+
+
+
+💼 29. "I need one runtime for proxy + tools + agent orchestration"
+
+Running many separate services increases operational cost and failure modes.
+
+**How OmniRoute solves it:**
+
+- OpenAI-compatible proxy, MCP server, and A2A server in one stack
+- Shared auth, resilience, data store, and observability
+- Consistent policy model across all interaction surfaces
+
+
+
+
+🚀 30. "I need to ship agentic workflows without glue-code sprawl"
+
+Teams lose velocity when stitching multiple ad-hoc services and scripts.
+
+**How OmniRoute solves it:**
+
+- Unified endpoint strategy for clients and agents
+- Built-in protocol management UIs and smoke validation paths
+- Production-ready foundations (security, logging, resilience, backup)
+
+
### Example Playbooks (Integrated Use Cases)
-**Playbook A : Maximisez l'abonnement payant + sauvegarde bon marché**```txt
+**Playbook A: Maximize paid subscription + cheap backup**
+
+```txt
Combo: "maximize-claude"
1. cc/claude-opus-4-6
2. glm/glm-4.7
@@ -608,21 +690,23 @@ Combo: "maximize-claude"
Monthly cost: $20 + small backup spend
Outcome: higher quality, near-zero interruption
-````
+```
-**Playbook B : pile de codage à coût nul**```txt
+**Playbook B: Zero-cost coding stack**
+
+```txt
Combo: "free-forever"
-
-1. gc/gemini-3-flash
-2. if/kimi-k2-thinking
-3. qw/qwen3-coder-plus
+ 1. gc/gemini-3-flash
+ 2. if/kimi-k2-thinking
+ 3. qw/qwen3-coder-plus
Monthly cost: $0
Outcome: stable free coding workflow
+```
-````
+**Playbook C: 24/7 always-on fallback chain**
-**Playbook C : chaîne de secours toujours active 24h/24 et 7j/7**```txt
+```txt
Combo: "always-on"
1. cc/claude-opus-4-6
2. cx/gpt-5.2-codex
@@ -631,122 +715,134 @@ Combo: "always-on"
5. if/kimi-k2-thinking
Outcome: deep fallback depth for deadline-critical workloads
-````
+```
-**Playbook D : Opérations d'agent avec MCP + A2A**```txt
+**Playbook D: Agent ops with MCP + A2A**
-1. Start MCP transport (`omniroute --mcp`) for tool-driven operations
-2. Run A2A tasks via `message/send` and `message/stream`
-3. Observe via /dashboard/endpoint (MCP and A2A tabs)
-4. Toggle services via inline status controls
-
-````
+```txt
+1) Start MCP transport (`omniroute --mcp`) for tool-driven operations
+2) Run A2A tasks via `message/send` and `message/stream`
+3) Observe via /dashboard/endpoint (MCP and A2A tabs)
+4) Toggle services via inline status controls
+```
---
## 🆓 Start Free — Zero Configuration Cost
-> Configurez le codage IA en quelques minutes à**0 $/mois**. Connectez ces comptes gratuits et utilisez le combo**Free Stack**intégré.
+> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo.
-| Étape | Actions | Fournisseurs débloqués |
+| Step | Action | Providers Unlocked |
| ---- | -------------------------------------------------- | ------------------------------------------------------------------ |
-| 1 | Connectez**Kiro**(ID AWS Builder OAuth) | Claude Sonnet 4.5, Haïku 4.5 —**illimité**|
-| 2 | Connectez**Qoder**(Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... —**illimité**|
-| 3 | Connectez**Qwen**(code de l'appareil) | qwen3-coder-plus, qwen3-coder-flash... —**illimité**|
-| 4 | Connectez**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro —**180 000/mois gratuits**|
-| 5 | `/dashboard/combos` → Modèle**Free Stack ($0)**| Faites un tourniquet automatique entre tous les fournisseurs gratuits |
+| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** |
+| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** |
+| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** |
+| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** |
+| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically |
-**Pointez n'importe quel IDE/CLI vers :**`http://localhost:20128/v1` · Clé API : `any-string` · Terminé.
+**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done.
->**Couverture supplémentaire facultative (également gratuite) :**Clé API Groq (30 RPM gratuits), NVIDIA NIM (40 RPM gratuits, plus de 70 modèles), Cerebras (1 million de tok/jour), clé API LongCat (50 millions de jetons/jour !), Cloudflare Workers AI (10 000 neurones/jour, plus de 50 modèles).## Démarrage Rapide
+> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models).
+
+## Démarrage Rapide
### 1) Install and run
```bash
npm install -g omniroute
omniroute
-````
+```
-> **Utilisateurs pnpm :**Exécutez `pnpm approved-builds -g` après l'installation pour activer les scripts de build natifs requis par `better-sqlite3` et `@swc/core` :
+> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`:
>
> ```bash
> pnpm install -g omniroute
-> pnpm approuver-builds -g # Sélectionner tous les packages → approuver
+> pnpm approve-builds -g # Select all packages → approve
> omniroute
> ```
-Le tableau de bord s'ouvre sur « http://localhost:20128 » et l'URL de base de l'API est « http://localhost:20128/v1 ».
+Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`.
-| Commande | Descriptif |
-| ----------------------- | --------------------------------------------------------------------------- |
-| `omniroute` | Démarrer le serveur (`PORT=20128`, API et tableau de bord sur le même port) |
-| `omniroute --port 3000` | Définir le port canonique/API sur 3000 |
-| `omniroute --mcp` | Démarrer le serveur MCP (transport stdio) |
-| `omniroute --no-open` | Ne pas ouvrir automatiquement le navigateur |
-| `omniroute --help` | Afficher l'aide |
+| Command | Description |
+| ----------------------- | ----------------------------------------------------------- |
+| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) |
+| `omniroute --port 3000` | Set canonical/API port to 3000 |
+| `omniroute --mcp` | Start MCP server (stdio transport) |
+| `omniroute --no-open` | Don't auto-open browser |
+| `omniroute --help` | Show help |
-Mode de port partagé en option :```bash
+Optional split-port mode:
+
+```bash
PORT=20128 DASHBOARD_PORT=20129 omniroute
-
-# API: http://localhost:20128/v1
-
+# API: http://localhost:20128/v1
# Dashboard: http://localhost:20129
-
-````
+```
### Long-Running Streaming Timeouts
-Pour la plupart des déploiements, vous n'avez besoin que de :
+For most deployments, you only need:
-| Variables | Par défaut | Objectif |
-| -------------------- | ----------------------------- | -------------------------------------------------------------------------------------------------------------------------------- |
-| `REQUEST_TIMEOUT_MS` | «600 000» | Base de référence partagée pour la récupération en amont, les délais d'attente Undici cachés, les demandes d'empreintes digitales TLS et les délais d'attente des demandes de pont API/proxy |
-| `STREAM_IDLE_TIMEOUT_MS` | hérite de `REQUEST_TIMEOUT_MS` | Écart maximum entre les morceaux de streaming avant qu'OmniRoute n'abandonne le flux SSE |
+| Variable | Default | Purpose |
+| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- |
+| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts |
+| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream |
-La compatibilité ascendante est préservée : `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS` et d'autres variables de délai d'expiration par couche fonctionnent toujours et remplacent la ligne de base partagée.
+Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline.
-Des remplacements avancés sont disponibles si vous avez besoin d'un contrôle plus précis :| Variables | Par défaut | Objectif |
-| --------------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- |
-| `FETCH_TIMEOUT_MS` | hérite de `REQUEST_TIMEOUT_MS` | Délai d'expiration total de la demande en amont utilisé par le signal d'abandon de récupération principal |
-| `FETCH_HEADERS_TIMEOUT_MS` | hérite de `FETCH_TIMEOUT_MS` | Délai Undici pour la réception des en-têtes de réponse en amont |
-| `FETCH_BODY_TIMEOUT_MS` | hérite de `FETCH_TIMEOUT_MS` | Limite de temps Undici entre les morceaux de corps en amont (`0` le désactive) |
-| `FETCH_CONNECT_TIMEOUT_MS` | '30 000' | Undici Délai d'expiration de la connexion TCP |
-| `FETCH_KEEPALIVE_TIMEOUT_MS` | '4000' | Undici délai d'attente du socket keep-alive inactif |
-| `TLS_CLIENT_TIMEOUT_MS` | hérite de `FETCH_TIMEOUT_MS` | Délai d'expiration pour les demandes d'empreintes digitales TLS effectuées via `wreq-js` |
-| `API_BRIDGE_PROXY_TIMEOUT_MS` | hérite de `REQUEST_TIMEOUT_MS` ou `30000` | Délai d'expiration pour le transfert du proxy `/v1` du port API vers le port du tableau de bord |
-| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300 000)` | Délai d'expiration des requêtes entrantes sur le serveur de pont API |
-| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | «60 000» | Délai d'expiration de l'en-tête entrant sur le serveur de pont API |
-| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | '5000' | Délai d'expiration de conservation sur le serveur de pont API |
-| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Délai d'inactivité du socket sur le serveur de pont API (`0` le désactive) |
+Advanced overrides are available if you need finer control:
-Si vous exécutez OmniRoute derrière Nginx, Caddy, Cloudflare ou un autre proxy inverse, assurez-vous que le proxy
-les délais d'attente sont également supérieurs aux délais d'attente de votre flux/récupération OmniRoute.### 2) Connect providers and create your API key
+| Variable | Default | Purpose |
+| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- |
+| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal |
+| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers |
+| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) |
+| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout |
+| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout |
+| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` |
+| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port |
+| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server |
+| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server |
+| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server |
+| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) |
-1. Ouvrez le tableau de bord → « Fournisseurs » et connectez au moins un fournisseur (OAuth ou clé API).
-2. Ouvrez le tableau de bord → « Endpoints » et créez une clé API.
-3. (Facultatif) Ouvrez le tableau de bord → « Combos » et définissez votre chaîne de secours.### 3) Point your coding tool to OmniRoute
+If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy
+timeouts are also higher than your OmniRoute stream/fetch timeouts.
+
+### 2) Connect providers and create your API key
+
+1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key).
+2. Open Dashboard → `Endpoints` and create an API key.
+3. (Optional) Open Dashboard → `Combos` and set your fallback chain.
+
+### 3) Point your coding tool to OmniRoute
```txt
Base URL: http://localhost:20128/v1
API Key: [copy from Endpoint page]
Model: if/kimi-k2-thinking (or any provider/model prefix)
-````
+```
-Fonctionne avec les SDK compatibles Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode et OpenAI.### 4) Enable and validate protocols (v2.0)
+Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs.
-**MCP (pour les opérations pilotées par outils) :**```bash
+### 4) Enable and validate protocols (v2.0)
+
+**MCP (for tool-driven operations):**
+
+```bash
omniroute --mcp
+```
-````
-
-Connectez ensuite votre client MCP via `stdio` et testez des outils tels que :
+Then connect your MCP client over `stdio` and test tools like:
- `omniroute_get_health`
- `omniroute_list_combos`
-**A2A (pour les flux de travail d'agent à agent) :**```bash
+**A2A (for agent-to-agent workflows):**
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
```bash
curl -X POST http://localhost:20128/a2a \
@@ -760,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \
npm run test:protocols:e2e
```
-Cette suite valide les flux clients MCP et A2A réels par rapport à une application en cours d'exécution.### Alternative: run from source
+This suite validates real MCP and A2A client flows against a running app.
+
+### Alternative: run from source
```bash
cp .env.example .env
@@ -768,14 +866,13 @@ npm install
PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev
```
-
+
+Void Linux (`xbps-src` template)
-Annuler Linux (modèle `xbps-src`)
-
-Pour les utilisateurs de Void Linux, vous pouvez créer un package natif en utilisant `xbps-src`. Enregistrez ce bloc sous `srcpkgs/omniroute/template` :```bash
+For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`:
+```bash
# Template file for 'omniroute'
-
pkgname=omniroute
version=3.4.1
revision=1
@@ -787,7 +884,7 @@ license="MIT"
homepage="https://github.com/diegosouzapw/OmniRoute"
distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz"
checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b
-system_accounts="\_omniroute"
+system_accounts="_omniroute"
omniroute_homedir="/var/lib/omniroute"
export NODE_ENV=production
export npm_config_engine_strict=false
@@ -795,71 +892,70 @@ export npm_config_loglevel=error
export npm_config_fund=false
export npm_config_audit=false
-do_build() { # Determine target CPU arch for node-gyp
-local \_gyp_arch
-case "$XBPS_TARGET_MACHINE" in
-aarch64*) \_gyp_arch=arm64 ;;
-armv7*|armv6*) \_gyp_arch=arm ;;
-i686*) \_gyp_arch=ia32 ;;
-\*) \_gyp_arch=x64 ;;
-esac
+do_build() {
+ # Determine target CPU arch for node-gyp
+ local _gyp_arch
+ case "$XBPS_TARGET_MACHINE" in
+ aarch64*) _gyp_arch=arm64 ;;
+ armv7*|armv6*) _gyp_arch=arm ;;
+ i686*) _gyp_arch=ia32 ;;
+ *) _gyp_arch=x64 ;;
+ esac
- # 1) Install all deps – skip scripts (no network in do_build, native modules
- # compiled separately below; better-sqlite3 is serverExternalPackage so
- # Next.js does not execute it during next build)
- NODE_ENV=development npm ci --ignore-scripts
+ # 1) Install all deps – skip scripts (no network in do_build, native modules
+ # compiled separately below; better-sqlite3 is serverExternalPackage so
+ # Next.js does not execute it during next build)
+ NODE_ENV=development npm ci --ignore-scripts
- # 2) Build the Next.js standalone bundle
- npm run build
+ # 2) Build the Next.js standalone bundle
+ npm run build
- # 3) Copy static assets into standalone
- cp -r .next/static .next/standalone/.next/static
- [ -d public ] && cp -r public .next/standalone/public || true
+ # 3) Copy static assets into standalone
+ cp -r .next/static .next/standalone/.next/static
+ [ -d public ] && cp -r public .next/standalone/public || true
- # 4) Compile better-sqlite3 native binding for the target architecture.
- # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
- # without npm altering them.
- local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
- (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
+ # 4) Compile better-sqlite3 native binding for the target architecture.
+ # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
+ # without npm altering them.
+ local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
+ (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
- # 5) Place the compiled binding into the standalone bundle
- local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
- mkdir -p "$_bs3_release"
- cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
+ # 5) Place the compiled binding into the standalone bundle
+ local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
+ mkdir -p "$_bs3_release"
+ cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
- # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
- # so sharp is not used at runtime; x64 .so files would break aarch64 strip
- rm -rf .next/standalone/node_modules/@img
-
- # 7) Copy pino runtime deps omitted by Next.js static analysis:
- # pino-abstract-transport – required by pino's worker thread
- # split2 – dep of pino-abstract-transport
- # process-warning – dep of pino itself
- for _mod in pino-abstract-transport split2 process-warning; do
- cp -r "node_modules/$_mod" .next/standalone/node_modules/
- done
+ # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
+ # so sharp is not used at runtime; x64 .so files would break aarch64 strip
+ rm -rf .next/standalone/node_modules/@img
+ # 7) Copy pino runtime deps omitted by Next.js static analysis:
+ # pino-abstract-transport – required by pino's worker thread
+ # split2 – dep of pino-abstract-transport
+ # process-warning – dep of pino itself
+ for _mod in pino-abstract-transport split2 process-warning; do
+ cp -r "node_modules/$_mod" .next/standalone/node_modules/
+ done
}
do_check() {
-npm run test:unit
+ npm run test:unit
}
do_install() {
-vmkdir usr/lib/omniroute/.next
+ vmkdir usr/lib/omniroute/.next
- vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
+ vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
- # Prevent removal of empty Next.js app router dirs by the post-install hook
- for _d in \
- .next/standalone/.next/server/app/dashboard \
- .next/standalone/.next/server/app/dashboard/settings \
- .next/standalone/.next/server/app/dashboard/providers; do
- touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
- done
-
- cat > "${WRKDIR}/omniroute" <<'EOF'
+ # Prevent removal of empty Next.js app router dirs by the post-install hook
+ for _d in \
+ .next/standalone/.next/server/app/dashboard \
+ .next/standalone/.next/server/app/dashboard/settings \
+ .next/standalone/.next/server/app/dashboard/providers; do
+ touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
+ done
+ cat > "${WRKDIR}/omniroute" <<'EOF'
#!/bin/sh
export PORT="${PORT:-20128}"
export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}"
@@ -871,10 +967,9 @@ EOF
}
post_install() {
-vlicense LICENSE
+ vlicense LICENSE
}
-
-````
+```
@@ -882,9 +977,11 @@ vlicense LICENSE
## 🐳 Docker
-OmniRoute est disponible sous forme d'image Docker publique sur [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
+OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
-**Exécution rapide :**```bash
+**Quick run:**
+
+```bash
docker run -d \
--name omniroute \
--restart unless-stopped \
@@ -892,85 +989,96 @@ docker run -d \
-p 20128:20128 \
-v omniroute-data:/app/data \
diegosouzapw/omniroute:latest
-````
+```
-**Avec fichier d'environnement :**```bash
+**With environment file:**
+```bash
# Copy and edit .env first
-
cp .env.example .env
docker run -d \
- --name omniroute \
- --restart unless-stopped \
- --stop-timeout 40 \
- --env-file .env \
- -p 20128:20128 \
- -v omniroute-data:/app/data \
- diegosouzapw/omniroute:latest
+ --name omniroute \
+ --restart unless-stopped \
+ --stop-timeout 40 \
+ --env-file .env \
+ -p 20128:20128 \
+ -v omniroute-data:/app/data \
+ diegosouzapw/omniroute:latest
+```
-````
+**Using Docker Compose:**
-**Utilisation de Docker Compose :**```bash
+```bash
# Base profile (no CLI tools)
docker compose --profile base up -d
# CLI profile (Claude Code, Codex, OpenClaw built-in)
docker compose --profile cli up -d
-````
+```
-La prise en charge des tableaux de bord pour les déploiements Docker inclut désormais un**Cloudflare Quick Tunnel**en un clic sur « Tableau de bord → Points de terminaison ». La première activation télécharge « cloudflared » uniquement en cas de besoin, démarre un tunnel temporaire vers votre point de terminaison « /v1 » actuel et affiche l'URL « https://\*.trycloudflare.com/v1 » générée directement sous votre URL publique normale.
+Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL.
-Remarques :
+Notes:
-- Les URL du tunnel rapide sont temporaires et changent après chaque redémarrage.
-- Les tunnels rapides ne sont pas automatiquement restaurés après un redémarrage d'OmniRoute ou d'un conteneur. Réactivez-les depuis le tableau de bord si nécessaire.
-- L'installation gérée prend actuellement en charge Linux, macOS et Windows sur `x64` / `arm64`.
-- Les tunnels rapides gérés utilisent par défaut le transport HTTP/2 pour éviter les avertissements de tampon QUIC UDP bruyants dans les environnements de conteneurs contraints. Définissez `CLOUDFLARED_PROTOCOL=quic` ou `auto` si vous souhaitez un transport différent.
-- Les images Docker regroupent les racines de l'autorité de certification du système et les transmettent au « cloudflared » géré, ce qui évite les échecs de confiance TLS lorsque le tunnel s'amorce à l'intérieur du conteneur.
-- SQLite fonctionne en mode WAL. `docker stop` doit être autorisé à se terminer afin qu'OmniRoute puisse vérifier les dernières modifications dans `storage.sqlite`.
-- Les fichiers Compose fournis définissent déjà un délai de grâce d'arrêt de 40 s. Si vous exécutez l'image directement, conservez « --stop-timeout 40 » (ou similaire) afin que les arrêts manuels n'interrompent pas le nettoyage de l'arrêt.
-- Définissez `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` si vous souhaitez qu'OmniRoute utilise un binaire existant au lieu d'en télécharger un.
+- Quick Tunnel URLs are temporary and change after every restart.
+- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed.
+- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`.
+- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport.
+- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container.
+- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`.
+- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup.
+- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one.
-**Utilisation de Docker Compose avec Caddy (HTTPS Auto-TLS) :**
+**Using Docker Compose with Caddy (HTTPS Auto-TLS):**
-OmniRoute peut être exposé en toute sécurité grâce au provisionnement SSL automatique de Caddy. Assurez-vous que l'enregistrement DNS A de votre domaine pointe vers l'adresse IP de votre serveur.```yaml
+OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP.
+
+```yaml
services:
-omniroute:
-image: diegosouzapw/omniroute:latest
-container_name: omniroute
-restart: unless-stopped
-volumes: - omniroute-data:/app/data
-environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com
+ omniroute:
+ image: diegosouzapw/omniroute:latest
+ container_name: omniroute
+ restart: unless-stopped
+ volumes:
+ - omniroute-data:/app/data
+ environment:
+ - PORT=20128
+ - NEXT_PUBLIC_BASE_URL=https://your-domain.com
-caddy:
-image: caddy:latest
-container_name: caddy
-restart: unless-stopped
-ports: - "80:80" - "443:443"
-command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
+ caddy:
+ image: caddy:latest
+ container_name: caddy
+ restart: unless-stopped
+ ports:
+ - "80:80"
+ - "443:443"
+ command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
volumes:
-omniroute-data:
+ omniroute-data:
+```
-````
+| Image | Tag | Size | Description |
+| ------------------------ | -------- | ------ | --------------------- |
+| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release |
+| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version |
-| Images | Étiquette | Taille | Descriptif |
-| -------------------- | -------- | ------ | ------------------------------------ |
-| `diegosouzapw/omniroute` | `dernier` | ~250 Mo | Dernière version stable |
-| `diegosouzapw/omniroute` | `1.0.3` | ~250 Mo | Version actuelle |---
+---
## 🖥️ Desktop App — Offline & Always-On
-> 🆕**NOUVEAU !**OmniRoute est désormais disponible en tant qu'**application de bureau native**pour Windows, macOS et Linux.
+> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux.
-Exécutez OmniRoute en tant qu'application de bureau autonome : aucun terminal, aucun navigateur, aucune connexion Internet requise pour les modèles locaux. L'application basée sur Electron comprend :
+Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes:
-- 🖥️**Fenêtre native**— Fenêtre d'application dédiée avec intégration dans la barre d'état système
-- 🔄**Démarrage automatique**— Lancez OmniRoute lors de la connexion au système
-- 🔔**Notifications natives**— Recevez des alertes en cas d'épuisement de quota ou de problèmes de fournisseur
-- ⚡**Installation en un clic**— NSIS (Windows), DMG (macOS), AppImage (Linux)
-- 🌐**Mode hors ligne**— Fonctionne entièrement hors ligne avec le serveur fourni### Démarrage Rapide
+- 🖥️ **Native Window** — Dedicated app window with system tray integration
+- 🔄 **Auto-Start** — Launch OmniRoute on system login
+- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues
+- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux)
+- 🌐 **Offline Mode** — Works fully offline with bundled server
+
+### Démarrage Rapide
```bash
# Development mode
@@ -981,308 +1089,370 @@ npm run electron:build # Current platform
npm run electron:build:win # Windows (.exe)
npm run electron:build:mac # macOS (.dmg) — x64 & arm64
npm run electron:build:linux # Linux (.AppImage)
-````
+```
### System Tray
-Lorsqu'il est réduit, OmniRoute réside dans votre barre d'état système avec des actions rapides :
+When minimized, OmniRoute lives in your system tray with quick actions:
-- Ouvrir le tableau de bord
-- Changer le port du serveur
-- Quitter l'application
+- Open dashboard
+- Change server port
+- Quit application
-📖 Documentation complète : [`electron/README.md`](electron/README.md)---
+📖 Full documentation: [`electron/README.md`](electron/README.md)
+
+---
## 💰 Pricing at a Glance
-| Niveau | Fournisseur | Coût | Réinitialisation des quotas | Idéal pour |
-| ----------------- | --------------------------------- | ---------------------------------------- | --------------------------- | ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| **💳 ABONNEMENT** | Claude Code (Pro) | 20 $/mois | 5h + hebdomadaire | Déjà abonné |
-| | Codex (Plus/Pro) | 20-200 $/mois | 5h + hebdomadaire | Utilisateurs d'OpenAI |
-| | CLI Gémeaux | **GRATUIT** | 180K/mois + 1K/jour | Tout le monde! |
-| | Copilote GitHub | 10-19 $/mois | Mensuel | Utilisateurs GitHub |
-| **🔑 CLÉ API** | NIM NVIDIA | **GRATUIT**(développement pour toujours) | ~40 tr/min | Plus de 70 modèles ouverts |
-| | Cérébraux | **GRATUIT**(1 million de tok/jour) | 60 000 TPM / 30 tr/min | Le plus rapide du monde |
-| | Groq | **GRATUIT**(30 TR/MIN) | 14,4 000 tr/min | Lama/Gemma ultra-rapide |
-| | DeepSeek V3.2 | 0,27 $/1,10 $ par 1 million | Aucun | Raisonnement meilleur prix/qualité |
-| | xAI Grok-4 Rapide | **0,20$/0,50$ par 1M**🆕 | Aucun | Appel d'outil le plus rapide +, ultra lent |
-| | xAI Grok-4 (standard) | 0,20 $/1,50 $ par 1 M 🆕 | Aucun | Produit phare du raisonnement de xAI |
-| | Mistral | Essai gratuit + payant | Tarif limité | IA européenne |
-| | OuvrirRouter | Paiement à l'utilisation | Aucun | Plus de 100 modèles agrégés. |
-| **💰 BON MARCHÉ** | GLM-5 (via Z.AI) 🆕 | 0,5 $/1 M | Tous les jours 10h | Sortie 128K, nouveau produit phare |
-| | GLM-4.7 | 0,6 $/1 M | Tous les jours 10h | Sauvegarde budgétaire |
-| | MiniMax M2.5 🆕 | Entrée de 0,3 $/1 M | 5 heures roulantes | Raisonnement + tâches agentiques |
-| | MiniMax M2.1 | 0,2 $/1 M | 5 heures roulantes | Option la moins chère |
-| | Kimi K2.5 (API Moonshot) 🆕 | Paiement à l'utilisation | Aucun | Accès direct à l'API Moonshot |
-| | Kimi K2 | 9 $/mois plat | 10 millions de jetons/mois | Coût prévisible |
-| **🆓 GRATUIT** | Qoder | **0$** | Unlimited | 5 modèles illimités |
-| | Qwen | **0$** | Illimité | 4 modèles illimités |
-| | Kiro | **0$** | Illimité | Claude Sonnet/Haïku (AWS Builder) |
-| | LongCat Flash-Lite 🆕 | **$0**(50M tok/jour 🔥) | 1 RPS | Le plus grand quota gratuit sur Terre |
-| | Pollinisations IA 🆕 | **0$**(aucune clé requise) | 1 demande/15s | GPT-5, Claude, DeepSeek, Lama 4 |
-| | IA des travailleurs Cloudflare 🆕 | **0$**(10 000 neurones/jour) | ~150 resp/jour | Plus de 50 modèles, avantage mondial |
-| | IA Scaleway 🆕 | **0 $**(1 million de jetons au total) | Tarif limité | UE/RGPD, Qwen3 235B, Lama 70B | > 🆕**Nouveaux modèles ajoutés (mars 2026) :**Famille Grok-4 Fast à 0,20 $/0,50 $/M (référence à 1 143 ms – 30 % plus rapide que Gemini 2.5 Flash), GLM-5 via Z.AI avec sortie 128K, raisonnement MiniMax M2.5, tarification mise à jour DeepSeek V3.2, Kimi K2.5 via l'API directe Moonshot. |
+| Tier | Provider | Cost | Quota Reset | Best For |
+| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- |
+| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed |
+| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users |
+| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! |
+| | GitHub Copilot | $10-19/mo | Monthly | GitHub users |
+| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models |
+| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest |
+| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma |
+| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning |
+| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow |
+| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI |
+| | Mistral | Free trial + paid | Rate limited | European AI |
+| | OpenRouter | Pay-per-use | None | 100+ models aggr. |
+| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship |
+| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup |
+| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks |
+| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option |
+| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access |
+| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost |
+| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited |
+| | Qwen | **$0** | Unlimited | 4 models unlimited |
+| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) |
+| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth |
+| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 |
+| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge |
+| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B |
-**💡 Pile combinée à 0 $ — La configuration gratuite complète :**```
+> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API.
+**💡 $0 Combo Stack — The Complete Free Setup:**
+
+```
# 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever
+Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
+Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
+Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
+Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
+NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+```
-Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
-Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
-LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
-Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
-Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
-Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
-Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
-Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
-Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
-NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
-Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever.
-````
-
-**Zéro coût. N'arrête jamais de coder.**Configurez-le comme un combo OmniRoute et toutes les solutions de secours se produisent automatiquement — sans jamais de commutation manuelle.---
+---
---
## 🆓 Free Models — What You Actually Get
-> Tous les modèles ci-dessous sont**100 % gratuits et aucune carte de crédit requise**. OmniRoute effectue un acheminement automatique entre eux lorsqu'un quota est épuisé : combinez-les tous pour un combo incassable à 0 $.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
+> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo.
-| Modèle | Préfixe | Limite | Limite de taux |
-| ------------------- | ------ | ------------- | ------------------------------------ |
-| `claude-sonnet-4.5` | `kr/` |**Illimité**| Aucun plafond quotidien signalé |
-| `claude-haïku-4.5` | `kr/` |**Illimité**| Aucun plafond quotidien signalé |
-| `claude-opus-4.6` | `kr/` |**Illimité**| Dernier Opus via Kiro |### 🟢 QODER MODELS (Free PAT via qodercli)
+### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
-| Modèle | Préfixe | Limite | Limite de taux |
+| Model | Prefix | Limit | Rate Limit |
+| ------------------- | ------ | ------------- | --------------------- |
+| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro |
+
+### 🟢 QODER MODELS (Free PAT via qodercli)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------ | ------ | ------------- | --------------- |
-| `kimi-k2-pensée` | `si/` |**Illimité**| Aucun plafond signalé |
-| `qwen3-coder-plus` | `si/` |**Illimité**| Aucun plafond signalé |
-| `deepseek-r1` | `si/` |**Illimité**| Aucun plafond signalé |
-| `minimax-m2.1` | `si/` |**Illimité**| Aucun plafond signalé |
-| `kimi-k2` | `si/` |**Illimité**| Aucun plafond signalé |
+| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap |
+| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap |
+| `deepseek-r1` | `if/` | **Unlimited** | No reported cap |
+| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap |
+| `kimi-k2` | `if/` | **Unlimited** | No reported cap |
-> Méthode de connexion recommandée :**Jeton d'accès personnel + `qodercli`**. Le navigateur OAuth est
-> expérimental et désactivé par défaut sauf si les variables d'environnement `QODER_OAUTH_*` sont configurées.### 🟡 QWEN MODELS (Device Code Auth)
+> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is
+> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured.
-| Modèle | Préfixe | Limite | Limite de taux |
+### 🟡 QWEN MODELS (Device Code Auth)
+
+| Model | Prefix | Limit | Rate Limit |
| ------------------- | ------ | ------------- | ------------------- |
-| `qwen3-coder-plus` | `qw/` |**Illimité**| Aucun plafond signalé |
-| `qwen3-coder-flash` | `qw/` |**Illimité**| Aucun plafond signalé |
-| `qwen3-coder-suivant` | `qw/` |**Illimité**| Aucun plafond signalé |
-| `modèle-vision` | `qw/` |**Illimité**| Multimodal (images) |### 🟣 GEMINI CLI (Google OAuth)
+| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap |
+| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) |
-| Modèle | Préfixe | Limite | Limite de taux |
-| -------------------- | ------ | -------------------------------- | ------------- |
-| `gemini-3-flash-aperçu` | `gc/` |**180K tok/mois**+ 1K/jour | Réinitialisation mensuelle |
-| `gemini-2.5-pro` | `gc/` | 180K/mois (piscine partagée) | Haute qualité |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
+### 🟣 GEMINI CLI (Google OAuth)
-| Niveau | Limite quotidienne | Limite de taux | Remarques |
+| Model | Prefix | Limit | Rate Limit |
+| ------------------------ | ------ | --------------------------- | ------------- |
+| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset |
+| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality |
+
+### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
+
+| Tier | Daily Limit | Rate Limit | Notes |
| ---------- | ------------ | ----------- | ------------------------------------------------------ |
-| Gratuit (développement) | Pas de plafond de jetons |**~40 tr/min**| Plus de 70 modèles ; transition vers des limites de taux pures mi-2025 |
+| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 |
-Modèles gratuits populaires : `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
+Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`
-| Niveau | Limite quotidienne | Limite de taux | Remarques |
-| ---- | ----------------- | ---------------- | ------------------------------------------------ |
-| Gratuit |**1 million de jetons/jour**| 60 000 TPM / 30 tr/min | L'inférence LLM la plus rapide au monde ; resets daily |
+### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
-Disponible gratuitement : `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`### 🔴 GROQ (Free API Key — console.groq.com)
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---- | ----------------- | ---------------- | ------------------------------------------- |
+| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily |
-| Niveau | Limite quotidienne | Limite de taux | Remarques |
+Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`
+
+### 🔴 GROQ (Free API Key — console.groq.com)
+
+| Tier | Daily Limit | Rate Limit | Notes |
| ---- | ------------- | ---------------- | ----------------------------------------- |
-| Gratuit |**14 400 RPJ**| 30 tr/min par modèle | Pas de carte de crédit ; 429 en limite, non facturé |
+| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged |
-Disponible gratuitement : `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`
-| Modèle | Préfixe | Quota quotidien gratuit | Remarques |
+### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+
+| Model | Prefix | Daily Free Quota | Notes |
| ----------------------------- | ------ | ----------------- | ----------------------- |
-| `LongCat-Flash-Lite` | `lc/` |**50 millions de jetons**💥 | Le plus grand quota gratuit jamais vu |
-| `LongCat-Flash-Chat` | `lc/` | 500 000 jetons | Chat multi-tours |
-| `LongCat-Flash-Pensée` | `lc/` | 500 000 jetons | Raisonnement / CoT |
-| `LongCat-Flash-Pensée-2601` | `lc/` | 500 000 jetons | Version janvier 2026 |
-| `LongCat-Flash-Omni-2603` | `lc/` | 500 000 jetons | Multimodal |
+| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever |
+| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat |
+| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT |
+| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version |
+| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal |
-> 100 % gratuit en version bêta publique. Inscrivez-vous sur [longcat.chat](https://longcat.chat) par e-mail ou par téléphone. Se réinitialise quotidiennement à 00h00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC.
-| Modèle | Préfixe | Limite de taux | Fournisseur derrière |
+### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+
+| Model | Prefix | Rate Limit | Provider Behind |
| ---------- | ------ | ---------- | ------------------ |
-| `openai` | `pol/` | 1 demande/15s | GPT-5 |
-| `claude` | `pol/` | 1 demande/15s | Claude Anthropique |
-| `Gémeaux` | `pol/` | 1 demande/15s | Google Gémeaux |
-| `recherche profonde` | `pol/` | 1 demande/15s | Recherche profonde V3 |
-| `lama` | `pol/` | 1 demande/15s | Meta Lama 4 Scout |
-| 'mistral' | `pol/` | 1 demande/15s | Mistral IA |
+| `openai` | `pol/` | 1 req/15s | GPT-5 |
+| `claude` | `pol/` | 1 req/15s | Anthropic Claude |
+| `gemini` | `pol/` | 1 req/15s | Google Gemini |
+| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 |
+| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout |
+| `mistral` | `pol/` | 1 req/15s | Mistral AI |
-> ✨**Zéro friction :**Pas d'inscription, pas de clé API. Ajoutez le fournisseur Pollinations avec un champ clé vide et cela fonctionne immédiatement.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately.
-| Niveau | Neurones quotidiens | Utilisation équivalente | Remarques |
+### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+
+| Tier | Daily Neurons | Equivalent Usage | Notes |
| ---- | ------------- | --------------------------------------- | ----------------------- |
-| Gratuit |**10 000**| ~ 150 LLM resp / 500 s audio / 15 000 intégrations | Avantage mondial, plus de 50 modèles |
+| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models |
-Modèles gratuits populaires : `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (audio gratuit !), `@cf/qwen/qwen2.5-coder-15b-instruct`
+Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct`
-> Nécessite un jeton API + un identifiant de compte de [dash.cloudflare.com](https://dash.cloudflare.com). Stockez l’ID de compte dans les paramètres du fournisseur.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings.
-| Niveau | Quotas gratuits | Localisation | Remarques |
+### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+
+| Tier | Free Quota | Location | Notes |
| ---- | ------------- | ------------ | ----------------------------------- |
-| Gratuit |**1 million de jetons**| 🇫🇷 Paris, UE | Aucune carte de crédit nécessaire dans certaines limites |
+| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits |
-Disponible gratuitement : `qwen3-235b-a22b-instruct-2507` (Qwen3 235B !), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
+Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
-> Conforme UE/RGPD. Obtenez la clé API sur [console.scaleway.com](https://console.scaleway.com).
+> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com).
->**💡 The Ultimate Free Stack (11 fournisseurs, 0 $ pour toujours) :**
+> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):**
>
> ```
-> Kiro (kr/) → Claude Sonnet/Haiku ILLIMITÉ
-> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 ILLIMITÉ
-> LongCat Lite (lc/) → LongCat-Flash-Lite — 50 millions de jetons/jour 🔥
-> Pollinisations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — aucune clé nécessaire
-> Qwen (qw/) → modèles qwen3-coder ILLIMITÉS
-> Gemini (gemini/) → Gemini 2.5 Flash — 1 500 req/jour gratuits
-> Cloudflare AI (cf/) → 50+ modèles — 10K Neurons/jour
-> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 million de jetons gratuits (UE)
-> Groq (groq/) → Lama/Gemma — 14,4K req/jour ultra-rapide
-> NVIDIA NIM (nvidia/) → Plus de 70 modèles ouverts — 40 RPM pour toujours
-> Cerebras (cerebras/) → Lama/Qwen le plus rapide au monde — 1 million de tok/jour
-> ```## 🎙️ Free Transcription Combo
+> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+> Qwen (qw/) → qwen3-coder models UNLIMITED
+> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free
+> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day
+> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast
+> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+> ```
-> Transcrivez n'importe quel audio/vidéo pour**0 $**— Deepgram mène avec 200 $ gratuits, AssemblyAI 50 $ de secours, Groq Whisper comme sauvegarde d'urgence illimitée.
+## 🎙️ Free Transcription Combo
-| Fournisseur | Crédits gratuits | Meilleur modèle | Limite de taux |
+> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup.
+
+| Provider | Free Credits | Best Model | Rate Limit |
| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- |
-| 🟢**Deepgram**|**200 $ gratuits**(inscription) | `nova-3` — meilleure précision, plus de 30 langues | Aucune limite de RPM sur les crédits gratuits |
-| 🔵**AssemblyAI**|**50 $ gratuits**(inscription) | `universal-3-pro` — chapitres, sentiments, PII | Aucune limite de RPM sur les crédits gratuits |
-| 🔴**Groq**|**Gratuit pour toujours**| `chuchotement-large-v3` — OpenAI Whisper | 30 tr/min (taux limité) |
+| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits |
+| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits |
+| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) |
-**Combo suggéré dans `/dashboard/combos` :**```
+**Suggested combo in `/dashboard/combos`:**
+
+```
Name: free-transcription
Strategy: Priority
Nodes:
[1] deepgram/nova-3 → uses $200 free first
[2] assemblyai/universal-3-pro → fallback when Deepgram credits run out
[3] groq/whisper-large-v3 → free forever, emergency fallback
-````
+```
-Ensuite, dans `/dashboard/media` → onglet**Transcription**: téléchargez n'importe quel fichier audio ou vidéo → sélectionnez votre point de terminaison combo → obtenez la transcription dans les formats pris en charge.## 💡 Key Features
+Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats.
-OmniRoute v2.0 est conçu comme une plate-forme opérationnelle et non comme un simple proxy relais.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
+## 💡 Key Features
-| Fonctionnalité | Ce qu'il fait |
-| --------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ |
-| ⚡**Famille rapide Grok-4** | Modèles xAI à 0,20 $/0,50 $/M — 1 143 ms de référence (30 % plus rapide que Gemini 2.5 Flash) |
-| 🧠**GLM-5 via Z.AI** | Contexte de sortie de 128 000 $, 0,5 / 1 million de dollars – le dernier produit phare de la famille GLM |
-| 🔮**MiniMax M2.5** | Raisonnement + tâches agentiques à 0,30 $/1 million — mise à niveau significative depuis M2.1 |
-| 🎯**toolCalling Flag par modèle** | `toolCalling : true/false` par modèle dans le registre — AutoCombo ignore les modèles non compatibles avec les outils |
-| 🌍**Détection d'intention multilingue** | Mots-clés PT/ZH/ES/AR dans la notation AutoCombo — meilleure sélection de modèles pour le contenu non anglais |
-| 📊**Replis basés sur des benchmarks** | Latence p95 réelle à partir de la notation combinée des flux de requêtes en direct — AutoCombo apprend à partir des données réelles |
-| 🔁**Demander une déduplication** | Fenêtre de déduplication basée sur le hachage de contenu — sécurisée multi-agents, évite les frais en double |
-| 🔌**Stratégie de routeur enfichable** | Interface extensible `RouterStrategy` — ajoutez une logique de routage personnalisée sous forme de plugins | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP |
+OmniRoute v3.5 is built as an operational platform, not just a relay proxy.
-| Fonctionnalité | Ce qu'il fait |
-| -------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |
-| 🎮**Aire de jeux modèle** | Page de tableau de bord pour tester n'importe quel modèle directement — sélecteurs de fournisseur/modèle/point de terminaison, éditeur de Monaco, streaming, abandon, timing |
-| 🔏**Correspondance d'empreintes digitales CLI** | Ordre des en-têtes/corps par fournisseur pour correspondre aux signatures CLI natives : basculez par fournisseur dans Paramètres > Sécurité.**Votre IP proxy est préservée** |
-| 🤝**Prise en charge ACP (Protocole Agent Client)** | Découverte d'agent CLI (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 de plus), générateur de processus, point de terminaison `/api/acp/agents` |
-| 🤖**Tableau de bord des agents ACP** | Page Débogage > Agents — grille de 14 agents avec état d'installation, version, formulaire d'agent personnalisé pour n'importe quel outil CLI. Les utilisateurs**OpenCode**bénéficient d'un bouton "Télécharger opencode.json" qui génère automatiquement une configuration prête à l'emploi avec tous les modèles disponibles. |
-| 🔧**Modèle personnalisé de routage `apiFormat`** | Les modèles personnalisés avec `apiFormat : "responses"` sont désormais correctement acheminés vers le traducteur de l'API Responses |
-| 🏢**Isolement de l'espace de travail Codex** | Plusieurs espaces de travail Codex par e-mail — OAuth sépare correctement les connexions par ID d'espace de travail |
-| 🔄**Mise à jour automatique électronique** | L'application de bureau vérifie les mises à jour + installation automatique au redémarrage | ### 🤖 Agent & Protocol Operations (v2.0) |
+### 🆕 New — v3.5.5 Highlights (Apr 2026)
-| Fonctionnalité | Ce qu'il fait |
-| ---------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- |
-| 🔧**Serveur MCP (25 outils)** | Outils IDE/agent via 3 transports : stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 cœurs + 3 mémoires + 4 outils de compétences |
-| 🤝**Serveur A2A (JSON-RPC + SSE)** | Exécution de tâches d'agent à agent avec flux de synchronisation et de streaming |
-| 🧭**Page des points de terminaison consolidés** | Page de gestion à onglets avec les onglets Endpoint Proxy, MCP, A2A et API Endpoints |
-| 🎚️**Bascules d'activation/désactivation du service** | Interrupteurs ON/OFF pour MCP et A2A avec persistance des paramètres (par défaut : OFF) |
-| 🛰️**Battement de coeur d'exécution MCP** | Statut réel du processus (pid, disponibilité, âge du battement de cœur, transport, mode scope) |
-| 📋**Piste d'audit MCP** | Journaux d'audit filtrables avec succès/échec et attribution des clés |
-| 🔐**Application du champ d'application du MCP** | 10 autorisations de portée granulaire pour un accès contrôlé aux outils |
-| 📡**Gestion du cycle de vie des tâches A2A** | Répertorier/filtrer les tâches, inspecter les événements/artefacts, annuler les tâches en cours |
-| 📋**Découverte de la carte d'agent** | `/.well-known/agent.json` pour la découverte automatique du client |
-| 🧪**Harnais de test du protocole E2E** | Le vrai client MCP SDK + A2A circule dans `test:protocols:e2e` |
-| ⚙️**Contrôles opérationnels** | Changer de combo, appliquer des profils de résilience, réinitialiser les disjoncteurs à partir d'une surface de contrôle | ### 🧠 Routing & Intelligence |
+| Feature | What It Does |
+| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation |
+| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs |
+| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions |
+| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection |
+| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts |
-| Fonctionnalité | Ce qu'il fait |
-| --------------------------------------------- | ----------------------------------------------------------------------------------------------- | ----------------------- |
-| 🎯**Repli intelligent à 4 niveaux** | Auto-route : Abonnement → Clé API → Pas cher → Gratuit |
-| 📊**Suivi des quotas en temps réel** | Nombre de jetons en direct + réinitialisation du compte à rebours par fournisseur |
-| 🔄**Traduction de formats** | OpenAI ↔ Claude ↔ Gémeaux ↔ Réponses avec conversions sécurisées |
-| 👥**Support multi-comptes** | Plusieurs comptes par fournisseur avec sélection intelligente |
-| 🔄**Actualisation automatique des jetons** | Les jetons OAuth s'actualisent automatiquement avec une nouvelle tentative |
-| 🎨**Combos personnalisés** | 9 stratégies d'équilibrage + contrôle de la chaîne de repli |
-| 🌐**Routeur générique** | `provider/*` routage dynamique |
-| 🧠**Penser les contrôles budgétaires** | Limites du raisonnement passthrough, automatique, personnalisé et adaptatif |
-| 🔀**Alias de modèle** | Alias de modèle intégré et personnalisé et sécurité de la migration |
-| ⚡**Dégradation de l'arrière-plan** | Acheminer les tâches en arrière-plan de faible priorité vers des modèles moins chers |
-| 🧪**Routage intelligent sensible aux tâches** | Modèle de sélection automatique par type de contenu (codage/vision/analyse/résumé) |
-| 🔄**Flux de travail des agents A2A** | Orchestrateur FSM déterministe pour les exécutions d'agents multi-étapes avec état |
-| 🔀**Routage adaptatif** | Remplacement de stratégie dynamique basé sur le volume de jetons et la complexité des invites |
-| 🎲**Diversité des fournisseurs** | Score d'entropie de Shannon équilibrant la distribution du trafic auto-combo |
-| 💬**Injection d'invite du système** | Contrôles comportementaux globaux appliqués de manière cohérente |
-| 📄**Compatibilité API des réponses** | Prise en charge complète de `/v1/responses` pour le Codex et les flux de travail agents avancés | ### 🎵 Multi-Modal APIs |
+### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
-| Fonctionnalité | Ce qu'il fait |
-| ------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- |
-| 🖼️**Génération d'images** | `/v1/images/generations` avec le cloud et les backends locaux |
-| 📐**Intégrations** | `/v1/embeddings` pour les pipelines de recherche et RAG |
-| 🎤**Transcription audio** | `/v1/audio/transcriptions` — 7 fournisseurs (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), détection automatique de langue, prise en charge MP4/MP3/WAV |
-| 🔊**Texte-parole** | `/v1/audio/speech` — 10 fournisseurs (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) avec des messages d'erreur corrects |
-| 🎬**Génération vidéo** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) |
-| 🎵**Génération musicale** | `/v1/music/generations` (flux de travail ComfyUI) |
-| 🛡️**Modérations** | Contrôles de sécurité `/v1/moderations` |
-| 🔀**Reclassement** | `/v1/rerank` pour la notation de pertinence |
-| 🔍**Recherche Web**🆕 | `/v1/search` — 5 fournisseurs (Serper, Brave, Perplexity, Exa, Tavily), plus de 6 500 gratuits/mois, basculement automatique, cache | ### 🛡️ Resilience, Security & Governance |
+| Feature | What It Does |
+| ------------------------------------ | ------------------------------------------------------------------------------------------- |
+| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) |
+| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family |
+| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 |
+| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models |
+| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content |
+| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data |
+| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges |
+| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins |
-| Fonctionnalité | Ce qu'il fait |
-| --------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------- | -------------------------------- |
-| 🔌**Disjoncteurs** | Déclenchement/récupération par modèle avec contrôles de seuil |
-| 🎯**Modèles prenant en compte les points de terminaison** | Les modèles personnalisés déclarent les points de terminaison pris en charge + le format API |
-| 🛡️**Troupeau anti-tonnerre** | Protections mutex + sémaphore sur les événements de nouvelle tentative/taux |
-| 🧠**Cache sémantique + signature** | Réduction des coûts/latences avec deux couches de cache |
-| ⚡**Demande d'idempotence** | Fenêtre de protection contre les doubles |
-| 🔒**Usurpation d'empreintes digitales TLS** | Empreinte digitale TLS de type navigateur —**réduit la détection des robots et le signalement des comptes** |
-| 🔏**Correspondance d'empreintes digitales CLI** | Correspond aux signatures de requêtes CLI natives —**réduit le risque d'interdiction tout en préservant l'adresse IP du proxy** |
-| 🌐**Filtrage IP** | Contrôle des listes autorisées/bloquées pour les déploiements exposés |
-| 📊**Limites de taux modifiables** | Limites configurables au niveau global/fournisseur avec persistance |
-| 📉**Dégradation gracieuse** | Capacités de secours multicouches protégeant les opérations principales de la passerelle |
-| 📜**Piste d'audit de configuration** | Suivi des modifications basé sur les différences empêchant la dérive opérationnelle avec de simples restaurations |
-| ⏳**Synchronisation de la santé du fournisseur** | Surveillance proactive de l'expiration des jetons déclenchant des alertes avant les échecs d'autorisation |
-| 🚪**Désactivation automatique des comptes interdits** | Disjoncteur opérationnel scellant automatiquement les comptes de jetons bloqués de manière permanente |
-| 🔑**Gestion des clés API + Cadrage** | Émission/rotation des clés sécurisées et contrôles des modèles/fournisseurs |
-| 👁️**Révélation de clé API étendue**🆕 | Récupération opt-in des clés API via `ALLOW_API_KEY_REVEAL` |
-| 🛡️**Protégé `/models`** | Gating d'authentification et masquage du fournisseur en option pour le catalogue de modèles | ### 📊 Observability & Analytics |
+### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP
-| Fonctionnalité | Ce qu'il fait |
-| ------------------------------------------ | ------------------------------------------------------------------------------------ | ---------------------------- |
-| 📝**Demande + Journalisation proxy** | Journalisation complète des requêtes/réponses et du proxy |
-| 📉**Journaux détaillés diffusés**🆕 | Reconstruit proprement les flux de charge utile SSE dans l'interface utilisateur |
-| 📋**Tableau de bord des journaux unifiés** | Vues de requête, de proxy, d'audit et de console sur une seule page |
-| 🔍**Demander une télémétrie** | Latence p50/p95/p99 et suivi des requêtes |
-| 🏥**Tableau de bord de santé** | Temps de disponibilité, états des disjoncteurs, verrouillages, statistiques du cache |
-| 💰**Suivi des coûts** | Contrôles budgétaires et visibilité des prix par modèle |
-| 📈**Visualisations analytiques** | Informations sur l'utilisation du modèle/fournisseur et vues des tendances |
-| 🧪**Cadre d'évaluation** | Tests du Golden Set avec stratégies de correspondance configurables |
-| 📡**Diagnostics en direct**🆕 | Contournement du cache sémantique pour des tests combo précis en direct | ### ☁️ Deployment & Platform |
+| Feature | What It Does |
+| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing |
+| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** |
+| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint |
+| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. |
+| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator |
+| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID |
+| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart |
-| Fonctionnalité | Ce qu'il fait |
-| ---------------------------------------- | ----------------------------------------------------------------------------------------- | --------------------- |
-| 🌐**Déployer n'importe où** | Localhost, VPS, Docker, environnements Cloud |
-| 🚇**Tunnel Cloudflare**🆕 | Intégration Quick Tunnel en un clic depuis le tableau de bord |
-| 🔑**Filtrage des modèles de clés API** | Réponse native /v1/models filtrée via les rôles contextuels Bearer attribués |
-| ⚡**Contournement intelligent du cache** | Heuristiques TTL configurables et contrôles de récupération forcée |
-| 🔄**Sauvegarde/Restauration** | Flux d'exportation/importation et de reprise après sinistre |
-| 🧙**Assistant d'intégration** | Configuration guidée de première exécution |
-| 🔧**Tableau de bord des outils CLI** | Configuration en un clic pour les outils de codage populaires |
-| 🎮**Aire de jeux modèle** | Testez n’importe quel fournisseur/modèle/point de terminaison à partir du tableau de bord |
-| 🔏**Bascule d'empreinte digitale CLI** | Correspondance des empreintes digitales par fournisseur dans Paramètres > Sécurité |
-| 🌐**i18n (30 langues)** | Tableau de bord complet + prise en charge des langues des documents avec couverture RTL |
-| 🧹**Effacer tous les modèles** | Suppression de la liste de modèles en un clic dans les détails du fournisseur |
-| 👁️**Contrôles de la barre latérale**🆕 | Masquer les composants et les intégrations dans les paramètres d'apparence |
-| 📋**Modèles de problèmes** | Modèles GitHub standardisés pour les bogues et les fonctionnalités |
-| 📂**Répertoire de données personnalisé** | Remplacement de `DATA_DIR` pour l'emplacement de stockage | ### Feature Deep Dive |
+### 🤖 Agent & Protocol Operations (v2.0)
+
+| Feature | What It Does |
+| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
+| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools |
+| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows |
+| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs |
+| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) |
+| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) |
+| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution |
+| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access |
+| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks |
+| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery |
+| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` |
+| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface |
+
+### 🧠 Routing & Intelligence
+
+| Feature | What It Does |
+| ---------------------------------- | ------------------------------------------------------------------------ |
+| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free |
+| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider |
+| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions |
+| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection |
+| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry |
+| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control |
+| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session |
+| 🌐 **Wildcard Router** | `provider/*` dynamic routing |
+| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits |
+| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety |
+| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models |
+| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) |
+| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions |
+| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity |
+| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution |
+| 💬 **System Prompt Injection** | Global behavior controls applied consistently |
+| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows |
+
+### 🎵 Multi-Modal APIs
+
+| Feature | What It Does |
+| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends |
+| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines |
+| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support |
+| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages |
+| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) |
+| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) |
+| 🛡️ **Moderations** | `/v1/moderations` safety checks |
+| 🔀 **Reranking** | `/v1/rerank` for relevance scoring |
+| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache |
+
+### 🛡️ Resilience, Security & Governance
+
+| Feature | What It Does |
+| ----------------------------------- | -------------------------------------------------------------------------------------- |
+| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls |
+| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format |
+| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events |
+| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers |
+| ⚡ **Request Idempotency** | Duplicate protection window |
+| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** |
+| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** |
+| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments |
+| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence |
+| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations |
+| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks |
+| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures |
+| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically |
+| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls |
+| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` |
+| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog |
+
+### 📊 Observability & Analytics
+
+| Feature | What It Does |
+| -------------------------------- | ----------------------------------------------------- |
+| 📝 **Request + Proxy Logging** | Full request/response and proxy logging |
+| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI |
+| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page |
+| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing |
+| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats |
+| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility |
+| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views |
+| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies |
+| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing |
+
+### ☁️ Deployment & Platform
+
+| Feature | What It Does |
+| ------------------------------ | --------------------------------------------------------------------- |
+| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments |
+| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard |
+| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles |
+| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls |
+| 🔄 **Backup/Restore** | Export/import and disaster recovery flows |
+| 🧙 **Onboarding Wizard** | First-run guided setup |
+| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools |
+| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard |
+| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security |
+| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage |
+| 🧹 **Clear All Models** | One-click model list clearing in provider details |
+| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings |
+| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features |
+| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location |
+
+### Feature Deep Dive
#### Smart fallback with practical cost control
@@ -1294,105 +1464,132 @@ Combo: "my-coding-stack"
4. if/kimi-k2-thinking
```
-En cas d'échec d'un quota, d'un taux ou d'un état de santé, OmniRoute passe automatiquement au candidat suivant sans commutation manuelle.#### Protocol management that is visible and operable
+When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching.
-- MCP + A2A sont détectables dans l'interface utilisateur et la documentation (non masquées)
-- Les API d'état du protocole exposent les données opérationnelles en direct (`/api/mcp/*`, `/api/a2a/*`)
-- Les tableaux de bord incluent des actions pour les opérations du jour 2 (basculements combinés, réinitialisations de disjoncteur, annulation de tâches)#### Translator + validation workflow
+#### Protocol management that is visible and operable
-La zone Traducteur comprend :
+- MCP + A2A are discoverable in UI and docs (not hidden)
+- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`)
+- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation)
--**Playground** : demander des contrôles de transformation -**Testeur de chat** : aller-retour complet de requête/réponse -**Banc de test** : plusieurs cas en une seule fois -**Live Monitor** : vue du trafic en temps réel
+#### Translator + validation workflow
-Plus validation du protocole avec de vrais clients via `npm run test:protocols:e2e`.
+The Translator area includes:
-> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Référence de l'outil, configurations IDE et exemples de clients
+- **Playground**: request transformation checks
+- **Chat Tester**: full request/response round-trip
+- **Test Bench**: multiple cases in one run
+- **Live Monitor**: real-time traffic view
+
+Plus protocol validation with real clients via `npm run test:protocols:e2e`.
+
+> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples
>
-> 📖**[A2A Server README](src/lib/a2a/README.md)**— Compétences, méthodes JSON-RPC, streaming et cycle de vie des tâches## 🧪 Evaluations (Evals)
+> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle
-OmniRoute comprend un cadre d'évaluation intégré pour tester la qualité des réponses LLM par rapport à un ensemble de référence. Accédez-y via**Analytics → Evals**dans le tableau de bord.### Built-in Golden Set
+## 🧪 Evaluations (Evals)
-Le « OmniRoute Golden Set » préchargé contient des cas de test pour :
+OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard.
-- Salutations, mathématiques, géographie, génération de code
-- Conformité au format JSON, traduction, génération de démarques
-- Refus de sécurité (contenu nuisible), comptage, logique booléenne### Evaluation Strategies
+### Built-in Golden Set
-| Stratégie | Descriptif | Exemple |
-| ---------------------- | --------------------------------------------------------------- | ---------------------------------- | --- |
-| `exact` | La sortie doit correspondre exactement | `"4"` |
-| `contient` | La sortie doit contenir une sous-chaîne (insensible à la casse) | `"Paris"` |
-| `expression régulière` | La sortie doit correspondre au modèle regex | `"1.*2.*3"` |
-| `personnalisé` | La fonction JS personnalisée renvoie vrai/faux | `(sortie) => sortie.longueur > 10` | --- |
+The pre-loaded "OmniRoute Golden Set" contains test cases for:
+
+- Greetings, math, geography, code generation
+- JSON format compliance, translation, markdown generation
+- Safety refusal (harmful content), counting, boolean logic
+
+### Evaluation Strategies
+
+| Strategy | Description | Example |
+| ---------- | ------------------------------------------------ | -------------------------------- |
+| `exact` | Output must match exactly | `"4"` |
+| `contains` | Output must contain substring (case-insensitive) | `"Paris"` |
+| `regex` | Output must match regex pattern | `"1.*2.*3"` |
+| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` |
+
+---
## 📖 Setup Guide
### Protocol Setup (MCP + A2A)
-
+
+🧩 MCP Setup (Model Context Protocol)
-🧩 Configuration MCP (Model Context Protocol)
+Start MCP transport in stdio mode:
-Démarrez le transport MCP en mode stdio :```bash
+```bash
omniroute --mcp
+```
-````
+Recommended validation flow:
-Flux de validation recommandé :
+1. Connect your MCP client over stdio.
+2. Run `omniroute_get_health`.
+3. Run `omniroute_list_combos`.
+4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit.
-1. Connectez votre client MCP via stdio.
-2. Exécutez `omniroute_get_health`.
-3. Exécutez `omniroute_list_combos`.
-4. Ouvrez « /dashboard/mcp » pour confirmer le rythme cardiaque, l'activité et l'audit.
-
-API utiles pour l'automatisation :
+Useful APIs for automation:
- `GET /api/mcp/status`
- `GET /api/mcp/tools`
- `GET /api/mcp/audit`
-- `GET /api/mcp/audit/stats`
+- `GET /api/mcp/audit/stats`
-
-🤝 Configuration A2A (Agent2Agent)
+
-Découvrez l'agent :```bash
+
+🤝 A2A Setup (Agent2Agent)
+
+Discover the agent:
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
-Envoyer une tâche :```bash
+Send a task:
+
+```bash
curl -X POST http://localhost:20128/a2a \
- -H 'content-type: application/json' \
- -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+ -H 'content-type: application/json' \
+ -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+```
-````
+Manage lifecycle:
-Gérer le cycle de vie :
-
-- `GET /api/a2a/statut`
-- `GET /api/a2a/tâches`
+- `GET /api/a2a/status`
+- `GET /api/a2a/tasks`
- `GET /api/a2a/tasks/:id`
- `POST /api/a2a/tasks/:id/cancel`
-Interface utilisateur opérationnelle :
+Operational UI:
-- `/dashboard/a2a` pour l'observabilité des tâches/états/flux et les actions de fumée
+- `/dashboard/a2a` for task/state/stream observability and smoke actions
-
-🧪 Validation du protocole de bout en bout
+
-Validez les deux protocoles avec de vrais clients :```bash
+
+🧪 End-to-end protocol validation
+
+Validate both protocols with real clients:
+
+```bash
npm run test:protocols:e2e
-````
+```
-Cela vérifie :
+This verifies:
-- Connexion/liste/appel du client MCP SDK
-- Découverte A2A/envoyer/diffuser/obtenir/annuler
-- Recoupement des données dans les API d'audit MCP et de gestion des tâches A2A
+- MCP SDK client connect/list/call
+- A2A discovery/send/stream/get/cancel
+- Cross-check data in MCP audit and A2A task management APIs
-
+
-💳 Fournisseurs d'abonnement### Claude Code (Pro/Max)
+
+💳 Subscription Providers
+
+### Claude Code (Pro/Max)
```bash
Dashboard → Providers → Connect Claude Code
@@ -1405,7 +1602,9 @@ Models:
cc/claude-haiku-4-5-20251001
```
-**Conseil de pro :**Utilisez Opus pour les tâches complexes, Sonnet pour la rapidité. OmniRoute suit le quota par modèle !### OpenAI Codex (Plus/Pro)
+**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model!
+
+### OpenAI Codex (Plus/Pro)
```bash
Dashboard → Providers → Connect Codex
@@ -1419,20 +1618,22 @@ Models:
#### Codex Account Limit Management (5h + Weekly)
-Chaque compte Codex dispose désormais de bascules de politique dans « Tableau de bord -> Fournisseurs » :
+Each Codex account now has policy toggles in `Dashboard -> Providers`:
-- `5h` (ON/OFF) : applique la politique de seuil de fenêtre de 5 heures.
-- `Hebdomadaire` (ON/OFF) : applique la politique de seuil de fenêtre hebdomadaire.
-- Comportement de seuil : lorsqu'une fenêtre activée atteint >=90 % d'utilisation, ce compte est ignoré.
-- Comportement de rotation : OmniRoute achemine automatiquement vers le prochain compte Codex éligible.
-- Comportement de réinitialisation : lorsque le délai "resetAt" du fournisseur est écoulé, le compte redevient automatiquement éligible.
+- `5h` (ON/OFF): enforce the 5-hour window threshold policy.
+- `Weekly` (ON/OFF): enforce the weekly window threshold policy.
+- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped.
+- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically.
+- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically.
-Scénarios :
+Scenarios:
-- `5h ON` + `Weekly ON` : le compte est ignoré lorsque l'une ou l'autre des fenêtres atteint le seuil.
-- `5h OFF` + `Weekly ON` : seule une utilisation hebdomadaire peut bloquer le compte.
-- `5h ON` + `Weekly OFF` : seule une utilisation de 5 heures peut bloquer le compte.
-- `resetAt` réussi : le compte entre à nouveau automatiquement dans la rotation (pas de réactivation manuelle).### Gemini CLI (FREE 180K/month!)
+- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold.
+- `5h OFF` + `Weekly ON`: only weekly usage can block the account.
+- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account.
+- `resetAt` passed: account re-enters rotation automatically (no manual re-enable).
+
+### Gemini CLI (FREE 180K/month!)
```bash
Dashboard → Providers → Connect Gemini CLI
@@ -1444,7 +1645,9 @@ Models:
gc/gemini-2.5-pro
```
-**Meilleur rapport qualité-prix :**Énorme niveau gratuit ! Utilisez-le avant les niveaux payants.### GitHub Copilot
+**Best Value:** Huge free tier! Use this before paid tiers.
+
+### GitHub Copilot
```bash
Dashboard → Providers → Connect GitHub
@@ -1459,74 +1662,91 @@ Models:
-
+
+🔑 API Key Providers
-🔑 Fournisseurs de clés API
### NVIDIA NIM (FREE developer access — 70+ models)
+### NVIDIA NIM (FREE developer access — 70+ models)
-1. Inscrivez-vous : [build.nvidia.com](https://build.nvidia.com)
-2. Obtenez une clé API gratuite (1 000 crédits d'inférence inclus)
-3. Tableau de bord → Ajouter un fournisseur → NVIDIA NIM :
- - Clé API : `nvapi-votre-clé`
+1. Sign up: [build.nvidia.com](https://build.nvidia.com)
+2. Get free API key (1000 inference credits included)
+3. Dashboard → Add Provider → NVIDIA NIM:
+ - API Key: `nvapi-your-key`
-**Modèles :**`nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct` et plus de 50 autres
+**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more
-**Conseil de pro :**API compatible OpenAI — fonctionne de manière transparente avec la traduction de format d'OmniRoute !### DeepSeek
+**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation!
-1. Inscrivez-vous : [platform.deepseek.com](https://platform.deepseek.com)
-2. Obtenez la clé API
-3. Tableau de bord → Ajouter un fournisseur → DeepSeek
+### DeepSeek
-**Modèles :**`deepseek/deepseek-chat`, `deepseek/deepseek-coder`### Groq (Free Tier Available!)
+1. Sign up: [platform.deepseek.com](https://platform.deepseek.com)
+2. Get API key
+3. Dashboard → Add Provider → DeepSeek
-1. Inscrivez-vous : [console.groq.com](https://console.groq.com)
-2. Obtenez la clé API (niveau gratuit inclus)
-3. Tableau de bord → Ajouter un fournisseur → Groq
+**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder`
-**Modèles :**`groq/llama-3.3-70b`, `groq/mixtral-8x7b`
+### Groq (Free Tier Available!)
-**Conseil de pro :**Inférence ultra-rapide : idéale pour le codage en temps réel !### OpenRouter (100+ Models)
+1. Sign up: [console.groq.com](https://console.groq.com)
+2. Get API key (free tier included)
+3. Dashboard → Add Provider → Groq
-1. Inscrivez-vous : [openrouter.ai](https://openrouter.ai)
-2. Obtenez la clé API
-3. Tableau de bord → Ajouter un fournisseur → OpenRouter
+**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b`
-**Modèles :**Accédez à plus de 100 modèles de tous les principaux fournisseurs via une seule clé API.
+**Pro Tip:** Ultra-fast inference — best for real-time coding!
-**Comportement du tableau de bord :**Les modèles OpenRouter sont gérés à partir des**Modèles disponibles**. L'ajout manuel, l'importation et la synchronisation automatique mettent tous à jour la même liste.
+### OpenRouter (100+ Models)
-
+1. Sign up: [openrouter.ai](https://openrouter.ai)
+2. Get API key
+3. Dashboard → Add Provider → OpenRouter
-💰 Fournisseurs bon marché (sauvegarde)### GLM-4.7 (Daily reset, $0.6/1M)
+**Models:** Access 100+ models from all major providers through a single API key.
-1. Inscrivez-vous : [Zhipu AI](https://open.bigmodel.cn/)
-2. Obtenez la clé API du plan de codage
-3. Tableau de bord → Ajouter une clé API :
- - Fournisseur : `glm`
- - Clé API : `votre-clé`
+**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list.
-**Utilisez :**`glm/glm-4.7`
+
-**Conseil de pro :**Le plan de codage offre un quota de 3 × à un coût de 1/7 ! Réinitialisation quotidienne à 10h00.### MiniMax M2.1 (5h reset, $0.20/1M)
+
+💰 Cheap Providers (Backup)
-1. Inscrivez-vous : [MiniMax](https://www.minimax.io/)
-2. Obtenez la clé API
-3. Tableau de bord → Ajouter une clé API
+### GLM-4.7 (Daily reset, $0.6/1M)
-**Utilisez :**`minimax/MiniMax-M2.1`
+1. Sign up: [Zhipu AI](https://open.bigmodel.cn/)
+2. Get API key from Coding Plan
+3. Dashboard → Add API Key:
+ - Provider: `glm`
+ - API Key: `your-key`
-**Conseil de pro :**Option la moins chère pour un contexte long (1 million de jetons) !### Kimi K2 ($9/month flat)
+**Use:** `glm/glm-4.7`
-1. Abonnez-vous : [Moonshot AI](https://platform.moonshot.ai/)
-2. Obtenez la clé API
-3. Tableau de bord → Ajouter une clé API
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM.
-**Utilisez :**`kimi/kimi-latest`
+### MiniMax M2.1 (5h reset, $0.20/1M)
-**Conseil de pro :**Fixe à 9 $/mois pour 10 millions de jetons = 0,90 $/1 M de coût effectif !
+1. Sign up: [MiniMax](https://www.minimax.io/)
+2. Get API key
+3. Dashboard → Add API Key
-
+**Use:** `minimax/MiniMax-M2.1`
-🆓 Fournisseurs GRATUITS (sauvegarde d'urgence)### Qoder (5 FREE models via OAuth)
+**Pro Tip:** Cheapest option for long context (1M tokens)!
+
+### Kimi K2 ($9/month flat)
+
+1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `kimi/kimi-latest`
+
+**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost!
+
+
+
+
+🆓 FREE Providers (Emergency Backup)
+
+### Qoder (5 FREE models via OAuth)
```bash
Dashboard → Connect Qoder
@@ -1567,9 +1787,10 @@ Models:
-
+
+🎨 Create Combos
-🎨 Créer des combos
### Example 1: Maximize Subscription → Cheap Backup
+### Example 1: Maximize Subscription → Cheap Backup
```
Dashboard → Combos → Create New
@@ -1597,9 +1818,10 @@ Cost: $0 forever!
-
+
+🔧 CLI Integration
-🔧 Intégration CLI
### Cursor IDE
+### Cursor IDE
```
Settings → Models → Advanced:
@@ -1610,7 +1832,9 @@ Settings → Models → Advanced:
### Claude Code
-Utilisez la page**CLI Tools**dans le tableau de bord pour une configuration en un clic, ou modifiez manuellement `~/.claude/settings.json`.### Codex CLI
+Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually.
+
+### Codex CLI
```bash
export OPENAI_BASE_URL="http://localhost:20128"
@@ -1621,12 +1845,15 @@ codex "your prompt"
### OpenClaw
-**Option 1 — Tableau de bord (recommandé) :**```
+**Option 1 — Dashboard (recommended):**
+
+```
Dashboard → CLI Tools → OpenClaw → Select Model → Apply
+```
-````
+**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`:
-**Option 2 — Manuel :**Modifiez `~/.openclaw/openclaw.json` :```json
+```json
{
"models": {
"providers": {
@@ -1638,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply
}
}
}
-````
+```
-> **Remarque :**OpenClaw ne fonctionne qu'avec OmniRoute local. Utilisez « 127.0.0.1 » au lieu de « localhost » pour éviter les problèmes de résolution IPv6.### Cline / Continue / RooCode
+> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues.
+
+### Cline / Continue / RooCode
```
Settings → API Configuration:
@@ -1652,15 +1881,17 @@ Settings → API Configuration:
### OpenCode
-**Étape 1 :**Ajoutez OmniRoute en tant que fournisseur personnalisé :```bash
+**Step 1:** Add OmniRoute as a custom provider:
+
+```bash
opencode
/connect
-
# Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key
+```
-````
+**Step 2:** Create/edit `opencode.json` in your project root:
-**Étape 2 :**Créez/modifiez « opencode.json » à la racine de votre projet :```json
+```json
{
"$schema": "https://opencode.ai/config.json",
"provider": {
@@ -1678,117 +1909,130 @@ opencode
}
}
}
-````
+```
-**Étape 3 :**Sélectionnez le modèle dans OpenCode :```bash
+**Step 3:** Select the model in OpenCode:
+
+```bash
/models
-
# Select any OmniRoute model from the list
+```
-````
+> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard.
->**Conseil :**Ajoutez n'importe quel modèle disponible dans votre point de terminaison OmniRoute `/v1/models` à la section `models`. Utilisez le format « fournisseur/modèle-id » de votre tableau de bord OmniRoute.
+
---
## Dépannage
-
-Cliquez pour développer le guide de dépannage
+
+Click to expand troubleshooting guide
-**"Le modèle linguistique n'a pas fourni de messages"**
+**"Language model did not provide messages"**
-- Quota du fournisseur épuisé → Vérifier le suivi des quotas du tableau de bord
-- Solution : utilisez la solution de secours combinée ou passez à un niveau moins cher
+- Provider quota exhausted → Check dashboard quota tracker
+- Solution: Use combo fallback or switch to cheaper tier
**Rate limiting**
-- Quota d'abonnement épuisé → Repli vers GLM/MiniMax
-- Ajouter un combo : `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Subscription quota out → Fallback to GLM/MiniMax
+- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
**OAuth token expired**
-- Actualisé automatiquement par OmniRoute
-- Si les problèmes persistent : Tableau de bord → Fournisseur → Reconnecter
+- Auto-refreshed by OmniRoute
+- If issues persist: Dashboard → Provider → Reconnect
**High costs**
-- Vérifiez les statistiques d'utilisation dans le tableau de bord → Coûts
-- Passer du modèle principal à GLM/MiniMax
-- Utilisez le niveau gratuit (Gemini CLI, Qoder) pour les tâches non critiques
+- Check usage stats in Dashboard → Costs
+- Switch primary model to GLM/MiniMax
+- Use free tier (Gemini CLI, Qoder) for non-critical tasks
-**Les ports du tableau de bord/API sont incorrects**
+**Dashboard/API ports are wrong**
-- `PORT` est le port de base canonique (et le port API par défaut)
-- `API_PORT` remplace uniquement l'écouteur d'API compatible OpenAI
-- `DASHBOARD_PORT` remplace uniquement l'écouteur du tableau de bord/Next.js
-- Définissez `NEXT_PUBLIC_BASE_URL` sur votre tableau de bord/URL publique (pour les rappels OAuth)
+- `PORT` is the canonical base port (and API port by default)
+- `API_PORT` overrides only OpenAI-compatible API listener
+- `DASHBOARD_PORT` overrides only dashboard/Next.js listener
+- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks)
**Cloud sync errors**
-- Vérifiez que `BASE_URL` pointe vers votre instance en cours d'exécution
-- Vérifiez que « CLOUD_URL » pointe vers votre point de terminaison cloud attendu
-- Gardez les valeurs `NEXT_PUBLIC_*` alignées avec les valeurs côté serveur
+- Verify `BASE_URL` points to your running instance
+- Verify `CLOUD_URL` points to your expected cloud endpoint
+- Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**La première connexion ne fonctionne pas**
+**First login not working**
-- Vérifiez `INITIAL_PASSWORD` dans `.env`
-- S'il n'est pas défini, le mot de passe de secours est « 123456 »
+- Check `INITIAL_PASSWORD` in `.env`
+- If unset, fallback password is `123456`
**No request logs**
-- Les artefacts de requête sont écrits dans `DATA_DIR/call_logs/` sous la forme d'un fichier JSON par requête
-- Activez la capture du pipeline depuis le tableau de bord → Journaux → Demander des journaux si vous avez besoin de charges utiles détaillées par étape
-- Définissez `APP_LOG_TO_FILE=true` si vous souhaitez également les journaux de la console d'application dans `logs/application/app.log`
-- Ajustez `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES` et `CALL_LOG_MAX_ENTRIES` selon vos besoins
+- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request
+- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads
+- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log`
+- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed
-**Le test de connexion indique « Invalide » pour les fournisseurs compatibles OpenAI**
+**Connection test shows "Invalid" for OpenAI-compatible providers**
-- De nombreux fournisseurs n'exposent pas de point de terminaison `/models`
-- OmniRoute v1.0.6+ inclut une validation de secours via la complétion du chat
-- Assurez-vous que l'URL de base inclut le suffixe `/v1`### 🔐 OAuth on a Remote Server
+- Many providers don't expose a `/models` endpoint
+- OmniRoute v1.0.6+ includes fallback validation via chat completions
+- Ensure base URL includes `/v1` suffix
+
+### 🔐 OAuth on a Remote Server
->**⚠️ Important pour les utilisateurs exécutant OmniRoute sur un VPS, Docker ou tout autre serveur distant**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
+> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server**
-Les fournisseurs**Antigravity**et**Gemini CLI**utilisent**Google OAuth 2.0**. Google exige que le « redirect_uri » dans le flux OAuth corresponde exactement à l'un des URI préenregistrés dans la Google Cloud Console de l'application.
+#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
-Les informations d'identification OAuth regroupées dans OmniRoute sont enregistrées**pour `localhost` uniquement**. Lorsque vous accédez à OmniRoute sur un serveur distant (par exemple `https://omniroute.myserver.com`), Google rejette l'authentification avec :```
+The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console.
+
+The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solution: Configure your own OAuth credentials
-Vous devez créer un**ID client OAuth 2.0**dans Google Cloud Console avec l'URI de votre serveur.#### Step-by-step
+You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI.
-**1. Ouvrez Google Cloud Console**
+#### Step-by-step
-Accédez à : [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
+**1. Open Google Cloud Console**
+
+Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
**2. Create a new OAuth 2.0 Client ID**
-- Cliquez sur**"+ Créer des informations d'identification"**→**"ID client OAuth"**
-- Type d'application :**"Application Web"**
-- Nom : tout ce que vous voulez (par exemple "OmniRoute Remote")
+- Click **"+ Create Credentials"** → **"OAuth client ID"**
+- Application type: **"Web application"**
+- Name: anything you like (e.g. `OmniRoute Remote`)
-**3. Ajouter des URI de redirection autorisés**
+**3. Add Authorized Redirect URIs**
-Dans le champ**"URI de redirection autorisés"**, ajoutez :```
+In the **"Authorized redirect URIs"** field, add:
+
+```
https://your-server.com/callback
+```
-````
+> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`).
-> Remplacez « votre-serveur.com » par le domaine ou l'IP de votre serveur (incluez le port si nécessaire, par exemple « http://45.33.32.156:20128/callback »).
+**4. Save and copy the credentials**
-**4. Enregistrez et copiez les informations d'identification**
+After creating, Google will show the **Client ID** and **Client Secret**.
-Après la création, Google affichera l'**ID client**et le**Secret client**.
+**5. Set environment variables**
-**5. Définir les variables d'environnement**
+In your `.env` (or Docker environment variables):
-Dans votre `.env` (ou variables d'environnement Docker) :```bash
+```bash
# For Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
@@ -1797,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
-````
+```
-**6. Restart OmniRoute**```bash
+**6. Restart OmniRoute**
+```bash
# npm:
-
npm run dev
# Docker:
-
docker restart omniroute
-
-````
+```
**7. Try connecting again**
-Tableau de bord → Fournisseurs → Antigravity (ou Gemini CLI) → OAuth
+Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth
-Google va désormais rediriger correctement vers « https://your-server.com/callback ».---
+Google will now redirect correctly to `https://your-server.com/callback`.
+
+---
#### Temporary workaround (without custom credentials)
-Si vous ne souhaitez pas configurer vos propres identifiants pour le moment, vous pouvez toujours utiliser le**flux d'URL manuel** :
+If you don't want to set up your own credentials right now, you can still use the **manual URL flow**:
-1. OmniRoute ouvre l'URL d'autorisation Google
-2. Après autorisation, Google tente de rediriger vers « localhost » (ce qui échoue sur le serveur distant)
-3.**Copiez l'URL complète**depuis la barre d'adresse de votre navigateur (même si la page ne se charge pas)
-4. Collez cette URL dans le champ affiché dans le modal de connexion OmniRoute.
-5. Click**"Connect"**
+1. OmniRoute opens the Google authorization URL
+2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server)
+3. **Copy the full URL** from your browser's address bar (even if the page doesn't load)
+4. Paste that URL into the field shown in the OmniRoute connection modal
+5. Click **"Connect"**
-> Cela fonctionne car le code d'autorisation dans l'URL est valide, que la page de redirection soit chargée ou non.---
+> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded.
-
-🇧🇷 Version en portugais
#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+---
-Les fournisseurs**Antigravity**et**Gemini CLI**utilisent**Google OAuth 2.0**pour l'authentification. Google exige que `redirect_uri` soit utilisé pour que le flux OAuth soit**exactement**un URI pré-cadastré dans l'application Google Cloud Console.
+
+🇧🇷 Versão em Português
-Comme les informations d'identification OAuth sont intégrées à OmniRoute, elles sont**spécialisées pour `localhost`**. Lorsque vous accédez à OmniRoute sur un serveur distant (ex : `https://omniroute.meuservidor.com`), Google refuse l'authentification avec :```
+#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+
+Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo.
+
+As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solução: Configure suas próprias credenciais OAuth
-Vous devez précisément créer un**ID client OAuth 2.0**dans Google Cloud Console avec l'URI de votre serveur.#### Passo a passo
+Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor.
-**1. Accéder à Google Cloud Console**
+#### Passo a passo
-Abra : [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
+**1. Acesse o Google Cloud Console**
-**2. Appelez un nouvel ID client OAuth 2.0**
+Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-- Cliquez dessus**"+ Créer des informations d'identification"**→**"ID client OAuth"**
-- Type d'application :**"Application Web"**
-- Nom : escolha qualquer nome (ex : `OmniRoute Remote`)
+**2. Crie um novo OAuth 2.0 Client ID**
-**3. Adicione comme URI de redirection autorisés**
+- Clique em **"+ Create Credentials"** → **"OAuth client ID"**
+- Tipo de aplicativo: **"Web application"**
+- Nome: escolha qualquer nome (ex: `OmniRoute Remote`)
-Pas de champ**"URI de redirection autorisés"**, ajouter :```
+**3. Adicione as Authorized Redirect URIs**
+
+No campo **"Authorized redirect URIs"**, adicione:
+
+```
https://seu-servidor.com/callback
+```
-````
+> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`).
-> Remplacez `seu-servidor.com` par votre domaine ou l'adresse IP de votre serveur (y compris le port si nécessaire, par exemple : `http://45.33.32.156:20128/callback`).
+**4. Salve e copie as credenciais**
-**4. Salve et copie comme credenciais**
+Após criar, o Google mostrará o **Client ID** e o **Client Secret**.
-Après avoir crié, Google affichera le**Client ID**et le**Client Secret**.
+**5. Configure as variáveis de ambiente**
-**5. Configurer comme variables d'ambiance**
+No seu `.env` (ou nas variáveis de ambiente do Docker):
-Pas votre `.env` (ou les variables ambiantes de Docker) :```bash
+```bash
# Para Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
@@ -1876,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
-````
+```
-**6. Reinicie o OmniRoute**```bash
+**6. Reinicie o OmniRoute**
+```bash
# Se usando npm:
-
npm run dev
# Se usando Docker:
-
docker restart omniroute
+```
-````
+**7. Tente conectar novamente**
-**7. Tente de connexion nouvelle**
+Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth
-Tableau de bord → Fournisseurs → Antigravité (ou Gemini CLI) → OAuth
+Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará.
-Agora ou Google redirigera directement vers `https://seu-servidor.com/callback` et la fonction d'authentification.---
+---
#### Workaround temporário (sem configurar credenciais próprias)
-Si vous ne souhaitez pas créer des informations d'identification appropriées il y a peu, vous pouvez également utiliser le flux**manuel d'URL** :
+Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**:
-1. OmniRoute ouvre une URL d'autorisation de Google
-2. Après avoir autorisé, Google tente de rediriger vers `localhost` (qui n'est pas un serveur distant)
-3.**Copiez une URL complète**à partir de la barre d'adresse de votre navigateur (même si la page n'est pas fermée)
-4. Cole est une URL dans le champ qui apparaît dans le modal de connexion à OmniRoute
-5. Cliquez dessus**"Connecter"**
+1. O OmniRoute abrirá a URL de autorização do Google
+2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto)
+3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue)
+4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute
+5. Clique em **"Connect"**
-> Cette solution de contournement fonctionne parce que le code d'autorisation de l'URL est valide indépendamment de la redirection lorsqu'il est chargé ou non.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+
+
---
@@ -1914,64 +2171,73 @@ Si vous ne souhaitez pas créer des informations d'identification appropriées i
## 🛠️ Tech Stack
-
-Cliquez pour développer les détails de la pile technologique
+
+Click to expand tech stack details
--**Exécution** : Node.js 18-22 LTS (⚠️ Node.js 24+ n'est**pas pris en charge**— les binaires natifs `better-sqlite3` sont incompatibles)
--**Langage** : TypeScript 5.9 —**100 % TypeScript**sur `src/` et `open-sse/` (zéro `any` dans les modules principaux depuis la v2.0)
--**Framework** : Next.js 16 + React 19 + Tailwind CSS 4
--**Base de données** : LowDB (JSON) + SQLite (état du domaine + journaux proxy + audit MCP + décisions de routage)
--**Schémas** : Zod (validation des E/S de l'outil MCP, contrats API)
--**Protocoles** : MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
--**Streaming** : événements envoyés par le serveur (SSE)
--**Auth** : OAuth 2.0 (PKCE) + JWT + Clés API + Autorisation étendue MCP
--**Tests** : lanceur de tests Node.js + Vitest (900+ tests incluant unitaire, intégration, E2E)
--**CI/CD** : actions GitHub (publication automatique npm + Docker Hub à la sortie)
--**Site Internet**: [omniroute.online](https://omniroute.online)
--**Package** : [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
--**Docker** : [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
--**Résilience** : disjoncteur, interruption exponentielle, troupeau anti-tonnerre, usurpation d'identité TLS, auto-réparation automatique
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible)
+- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0)
+- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4
+- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions)
+- **Schemas**: Zod (MCP tool I/O validation, API contracts)
+- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
+- **Streaming**: Server-Sent Events (SSE)
+- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization
+- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E)
+- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release)
+- **Website**: [omniroute.online](https://omniroute.online)
+- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
+- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
+- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing
+
+
---
## Documentation
-| Documenter | Descriptif |
+| Document | Description |
| ---------------------------------------------- | --------------------------------------------------- |
-| [Guide de l'utilisateur](docs/USER_GUIDE.md) | Fournisseurs, combos, intégration CLI, déploiement |
-| [Référence API](docs/API_REFERENCE.md) | Tous les points de terminaison avec des exemples |
-| [Serveur MCP](open-sse/mcp-server/README.md) | 16 outils MCP, configurations IDE, clients Python/TS/Go |
-| [Serveur A2A](src/lib/a2a/README.md) | Protocole JSON-RPC 2.0, compétences, streaming, gestion des tâches |
-| [Moteur Auto-Combo](docs/auto-combo.md) | Score à 6 facteurs, packs de modes, auto-guérison |
-| [Dépannage](docs/TROUBLESHOOTING.md) | Problèmes courants et solutions |
-| [Architecture](docs/ARCHITECTURE.md) | Architecture du système et composants internes |
-| [Contribuer](CONTRIBUTING.md) | Configuration et directives de développement |
-| [Spécifications OpenAPI](docs/openapi.yaml) | Spécification OpenAPI 3.0 |
-| [Politique de sécurité](SECURITY.md) | Rapports de vulnérabilité et pratiques de sécurité |
-| [Déploiement de VM](docs/VM_DEPLOYMENT_GUIDE.md) | Guide complet : configuration VM + nginx + Cloudflare |
-| [Galerie de fonctionnalités](docs/FEATURES.md) | Visite visuelle du tableau de bord avec captures d'écran |
-| [Liste de contrôle de publication](docs/RELEASE_CHECKLIST.md) | Étapes de validation avant la publication |---
+| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment |
+| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples |
+| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients |
+| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt |
+| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing |
+| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation |
+| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions |
+| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals |
+| [Contributing](CONTRIBUTING.md) | Development setup and guidelines |
+| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification |
+| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices |
+| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup |
+| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots |
+| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps |
+
+---
## 🗺️ Roadmap
-OmniRoute propose**plus de 210 fonctionnalités prévues**au cours de plusieurs phases de développement. Here are the key areas:
+OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas:
-| Catégorie | Planned Features | Faits saillants |
-| ----------------------------- | ---------------- | ---------------------------------------------------------------------------- |
-| 🧠**Routage & Intelligence**| 25+ | Routage avec la latence la plus faible, routage basé sur des balises, contrôle en amont des quotas, sélection de comptes P2C |
-| 🔒**Sécurité et conformité**| 20+ | Renforcement SSRF, masquage des informations d'identification, limite de débit par point de terminaison, portée des clés de gestion |
-| 📊**Observabilité**| 15+ | Intégration OpenTelemetry, surveillance des quotas en temps réel, suivi des coûts par modèle |
-| 🔄**Intégrations de fournisseurs**| 20+ | Registre de modèles dynamique, temps de recharge des fournisseurs, Codex multi-comptes, analyse des quotas Copilot |
-| ⚡**Performances**| 15+ | Double couche de cache, cache d'invite, cache de réponse, streaming keepalive, API par lots |
-| 🌐**Écosystème**| 10+ | API WebSocket, rechargement à chaud de la configuration, magasin de configuration distribué, mode commercial |### 🔜 Coming Soon
+| Category | Planned Features | Highlights |
+| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- |
+| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection |
+| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping |
+| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model |
+| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing |
+| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API |
+| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode |
-- 🔗**Intégration OpenCode**— Prise en charge par le fournisseur natif pour l'IDE de codage OpenCode AI
-- 🔗**Intégration TRAE**— Prise en charge complète du cadre de développement TRAE AI
-- 📦**Batch API**— Traitement par lots asynchrone pour les demandes groupées
-- 🎯**Routage basé sur des balises**— Acheminez les requêtes en fonction de balises personnalisées et de métadonnées
-- 💰**Stratégie du coût le plus bas**— Sélectionnez automatiquement le fournisseur disponible le moins cher
+### 🔜 Coming Soon
-> 📝 Spécifications complètes des fonctionnalités disponibles dans [`docs/new-features/`](docs/new-features/) (217 spécifications détaillées)---
+- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE
+- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework
+- 📦 **Batch API** — Asynchronous batch processing for bulk requests
+- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata
+- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider
+
+> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs)
+
+---
## 👥 Contributors
@@ -1980,17 +2246,19 @@ OmniRoute propose**plus de 210 fonctionnalités prévues**au cours de plusieurs
### How to Contribute
1. Fork the repository
-2. Créez votre branche de fonctionnalités (`git checkout -b feature/amazing-feature`)
-3. Validez vos modifications (`git commit -m 'Ajouter une fonctionnalité étonnante'`)
-4. Poussez vers la branche (`git push origin feature/amazing-feature`)
+2. Create your feature branch (`git checkout -b feature/amazing-feature`)
+3. Commit your changes (`git commit -m 'Add amazing feature'`)
+4. Push to the branch (`git push origin feature/amazing-feature`)
5. Open a Pull Request
-Voir [CONTRIBUTING.md](CONTRIBUTING.md) pour des directives détaillées.### Releasing a New Version
+See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines.
+
+### Releasing a New Version
```bash
# Create a release — npm publish happens automatically
gh release create v2.0.0 --title "v2.0.0" --generate-notes
-````
+```
---
@@ -2002,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes
## 🙏 Acknowledgments
-Un merci spécial à**[9router](https://github.com/decolua/9router)**de**[decolua](https://github.com/decolua)**— le projet original qui a inspiré ce fork. OmniRoute s'appuie sur cette incroyable base avec des fonctionnalités supplémentaires, des API multimodales et une réécriture complète de TypeScript.
+Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite.
-Un merci spécial à**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**— l'implémentation Go originale qui a inspiré ce port JavaScript.---
+Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port.
+
+---
## Licence
-Licence MIT - voir [LICENSE](LICENSE) pour plus de détails.---
+MIT License - see [LICENSE](LICENSE) for details.
+
+---
Built with ❤️ for developers who code 24/7
diff --git a/docs/i18n/fr/docs/ARCHITECTURE.md b/docs/i18n/fr/docs/ARCHITECTURE.md
index 9697b03066..a6d281af28 100644
--- a/docs/i18n/fr/docs/ARCHITECTURE.md
+++ b/docs/i18n/fr/docs/ARCHITECTURE.md
@@ -4,80 +4,93 @@
---
-_Dernière mise à jour : 2026-03-28_## Executive Summary
-OmniRoute est une passerelle de routage d'IA locale et un tableau de bord construit sur Next.js.
-Il fournit un seul point de terminaison compatible OpenAI (`/v1/*`) et achemine le trafic vers plusieurs fournisseurs en amont avec traduction, secours, actualisation des jetons et suivi de l'utilisation.
-Capacités de base :
+_Last updated: 2026-03-28_
-- Surface API compatible OpenAI pour CLI/outils (28 fournisseurs)
-- Traduction des requêtes/réponses dans tous les formats de fournisseurs
-- Modèle de repli combo (séquence multi-modèles)
-- Repli au niveau du compte (multi-comptes par fournisseur)
-- Gestion des connexions du fournisseur de clé OAuth + API
-- Génération d'embarquement via `/v1/embeddings` (6 fournisseurs, 9 modèles)
-- Génération d'images via `/v1/images/generations` (4 fournisseurs, 9 modèles)
-- Pensez à l'analyse des balises (`
...`) pour les modèles de raisonnement
-- Désinfection des réponses pour une compatibilité stricte avec le SDK OpenAI
-- Normalisation des rôles (développeur → système, système → utilisateur) pour une compatibilité entre fournisseurs
-- Conversion de sortie structurée (json_schema → Gemini ResponseSchema)
-- Persistance locale pour les fournisseurs, les clés, les alias, les combos, les paramètres, les prix
-- Suivi de l'utilisation/des coûts et journalisation des demandes
-- Synchronisation cloud en option pour la synchronisation multi-appareils/états
-- Liste d'autorisation/liste de blocage IP pour le contrôle d'accès aux API
-- Penser la gestion budgétaire (passthrough/auto/custom/adaptatif)
- -Injection rapide du système global
-- Suivi de session et prise d'empreintes digitales
-- Limitation de débit améliorée par compte avec des profils spécifiques au fournisseur
-- Modèle de disjoncteur pour la résilience du fournisseur
-- Protection de troupeau anti-tonnerre avec verrouillage mutex
-- Cache de déduplication de requêtes basé sur les signatures
-- Couche domaine : disponibilité du modèle, règles de coûts, politique de repli, politique de verrouillage
-- Persistance de l'état du domaine (cache en écriture SQLite pour les solutions de repli, les budgets, les verrouillages, les disjoncteurs)
-- Moteur de politique pour l'évaluation centralisée des demandes (verrouillage → budget → repli)
-- Demande de télémétrie avec agrégation de latence p50/p95/p99
-- ID de corrélation (X-Request-Id) pour le traçage de bout en bout
-- Journalisation d'audit de conformité avec désinscription par clé API
-- Cadre d'évaluation pour l'assurance qualité LLM
-- Tableau de bord de l'interface utilisateur de résilience avec l'état du disjoncteur en temps réel
-- Fournisseurs OAuth modulaires (12 modules individuels sous `src/lib/oauth/providers/`)
+## Executive Summary
-Modèle d'exécution principal :
+OmniRoute is a local AI routing gateway and dashboard built on Next.js.
+It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking.
-- Les routes de l'application Next.js sous `src/app/api/*` implémentent à la fois les API de tableau de bord et les API de compatibilité
-- Un noyau SSE/routage partagé dans `src/sse/*` + `open-sse/*` gère l'exécution, la traduction, le streaming, le repli et l'utilisation du fournisseur## Scope and Boundaries
+Core capabilities:
+
+- OpenAI-compatible API surface for CLI/tools (28 providers)
+- Request/response translation across provider formats
+- Model combo fallback (multi-model sequence)
+- Account-level fallback (multi-account per provider)
+- OAuth + API-key provider connection management
+- Embedding generation via `/v1/embeddings` (6 providers, 9 models)
+- Image generation via `/v1/images/generations` (4 providers, 9 models)
+- Think tag parsing (`
...`) for reasoning models
+- Response sanitization for strict OpenAI SDK compatibility
+- Role normalization (developer→system, system→user) for cross-provider compatibility
+- Structured output conversion (json_schema → Gemini responseSchema)
+- Local persistence for providers, keys, aliases, combos, settings, pricing
+- Usage/cost tracking and request logging
+- Optional cloud sync for multi-device/state sync
+- IP allowlist/blocklist for API access control
+- Thinking budget management (passthrough/auto/custom/adaptive)
+- Global system prompt injection
+- Session tracking and fingerprinting
+- Per-account enhanced rate limiting with provider-specific profiles
+- Circuit breaker pattern for provider resilience
+- Anti-thundering herd protection with mutex locking
+- Signature-based request deduplication cache
+- Domain layer: model availability, cost rules, fallback policy, lockout policy
+- Context Relay: session handoff summaries for account rotation continuity
+- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers)
+- Policy engine for centralized request evaluation (lockout → budget → fallback)
+- Request telemetry with p50/p95/p99 latency aggregation
+- Correlation ID (X-Request-Id) for end-to-end tracing
+- Compliance audit logging with opt-out per API key
+- Eval framework for LLM quality assurance
+- Resilience UI dashboard with real-time circuit breaker status
+- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`)
+
+Primary runtime model:
+
+- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs
+- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage
+
+## Scope and Boundaries
### In Scope
-- Runtime de la passerelle locale
-- API de gestion des tableaux de bord
-- Authentification du fournisseur et actualisation du jeton
-- Demander une traduction et un streaming SSE
-- État local + persistance d'utilisation
-- Orchestration de synchronisation cloud en option### Out of Scope
+- Local gateway runtime
+- Dashboard management APIs
+- Provider authentication and token refresh
+- Request translation and SSE streaming
+- Local state + usage persistence
+- Optional cloud sync orchestration
-- Implémentation du service Cloud derrière `NEXT_PUBLIC_CLOUD_URL`
-- SLA/plan de contrôle du fournisseur en dehors du processus local
-- Les binaires CLI externes eux-mêmes (Claude CLI, Codex CLI, etc.)## Dashboard Surface (Current)
+### Out of Scope
-Pages principales sous `src/app/(dashboard)/dashboard/` :
+- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL`
+- Provider SLA/control plane outside local process
+- External CLI binaries themselves (Claude CLI, Codex CLI, etc.)
-- `/dashboard` — démarrage rapide + aperçu du fournisseur
-- `/dashboard/endpoint` — proxy de point de terminaison + MCP + A2A + onglets de point de terminaison API
-- `/dashboard/providers` — connexions et informations d'identification du fournisseur
-- `/dashboard/combos` — stratégies de combo, modèles, règles de routage de modèles
-- `/dashboard/costs` — agrégation des coûts et visibilité sur les prix
-- `/dashboard/analytics` — analyses et évaluations d'utilisation
-- `/dashboard/limits` — contrôles de quotas/taux
-- `/dashboard/cli-tools` — Intégration CLI, détection d'exécution, génération de configuration
-- `/dashboard/agents` — agents ACP détectés + enregistrement d'agent personnalisé
-- `/dashboard/media` — terrain de jeu image/vidéo/musique
-- `/dashboard/search-tools` — tests et historique du moteur de recherche
-- `/dashboard/health` — temps de disponibilité, disjoncteurs, limites de débit
-- `/dashboard/logs` — journaux de requête/proxy/audit/console
-- `/dashboard/settings` — onglets des paramètres système (général, routage, valeurs par défaut des combos, etc.)
-- `/dashboard/api-manager` — Cycle de vie des clés API et autorisations du modèle## High-Level System Context
+## Dashboard Surface (Current)
+
+Main pages under `src/app/(dashboard)/dashboard/`:
+
+- `/dashboard` — quick start + provider overview
+- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs
+- `/dashboard/providers` — provider connections and credentials
+- `/dashboard/combos` — combo strategies, templates, model routing rules
+- `/dashboard/costs` — cost aggregation and pricing visibility
+- `/dashboard/analytics` — usage analytics and evaluations
+- `/dashboard/limits` — quota/rate controls
+- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation
+- `/dashboard/agents` — detected ACP agents + custom agent registration
+- `/dashboard/media` — image/video/music playground
+- `/dashboard/search-tools` — search provider testing and history
+- `/dashboard/health` — uptime, circuit breakers, rate limits
+- `/dashboard/logs` — request/proxy/audit/console logs
+- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.)
+- `/dashboard/api-manager` — API key lifecycle and model permissions
+
+## High-Level System Context
```mermaid
flowchart LR
@@ -129,139 +142,151 @@ flowchart LR
## 1) API and Routing Layer (Next.js App Routes)
-Principaux répertoires :
+Main directories:
-- `src/app/api/v1/*` et `src/app/api/v1beta/*` pour les API de compatibilité
-- `src/app/api/*` pour les API de gestion/configuration
-- Les réécritures suivantes dans `next.config.mjs` mappent `/v1/*` en `/api/v1/*`
+- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs
+- `src/app/api/*` for management/configuration APIs
+- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*`
-Itinéraires de compatibilité importants :
+Important compatibility routes:
- `src/app/api/v1/chat/completions/route.ts`
- `src/app/api/v1/messages/route.ts`
- `src/app/api/v1/responses/route.ts`
-- `src/app/api/v1/models/route.ts` — inclut des modèles personnalisés avec `custom: true`
-- `src/app/api/v1/embeddings/route.ts` — génération d'intégration (6 fournisseurs)
-- `src/app/api/v1/images/generations/route.ts` — génération d'images (4+ fournisseurs dont Antigravity/Nebius)
+- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true`
+- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers)
+- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius)
- `src/app/api/v1/messages/count_tokens/route.ts`
-- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — chat dédié par fournisseur
-- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — intégrations dédiées par fournisseur
-- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — images dédiées par fournisseur
+- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat
+- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings
+- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images
- `src/app/api/v1beta/models/route.ts`
-- `src/app/api/v1beta/models/[...chemin]/route.ts`
+- `src/app/api/v1beta/models/[...path]/route.ts`
-Domaines de gestion :
+Management domains:
-- Authentification/paramètres : `src/app/api/auth/*`, `src/app/api/settings/*`
-- Fournisseurs/connexions : `src/app/api/providers*`
-- Nœuds fournisseurs : `src/app/api/provider-nodes*`
-- Modèles personnalisés : `src/app/api/provider-models` (GET/POST/DELETE)
-- Catalogue de modèles : `src/app/api/models/route.ts` (GET)
-- Configuration du proxy : `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
- -OAuth : `src/app/api/oauth/*`
-- Clés/alias/combos/pricing : `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing`
-- Utilisation : `src/app/api/usage/*`
-- Synchronisation/cloud : `src/app/api/sync/*`, `src/app/api/cloud/*`
-- Aides aux outils CLI : `src/app/api/cli-tools/*`
-- Filtre IP : `src/app/api/settings/ip-filter` (GET/PUT)
-- Budget de réflexion : `src/app/api/settings/thinking-budget` (GET/PUT)
-- Invite système : `src/app/api/settings/system-prompt` (GET/PUT)
-- Sessions : `src/app/api/sessions` (GET)
-- Limites de débit : `src/app/api/rate-limits` (GET)
-- Résilience : `src/app/api/resilience` (GET/PATCH) — profils de fournisseur, disjoncteur, état limite de débit
-- Réinitialisation de la résilience : `src/app/api/resilience/reset` (POST) — réinitialisation des disjoncteurs + temps de recharge
-- Statistiques du cache : `src/app/api/cache/stats` (GET/DELETE)
-- Disponibilité du modèle : `src/app/api/models/availability` (GET/POST)
-- Télémétrie : `src/app/api/telemetry/summary` (GET)
-- Budget : `src/app/api/usage/budget` (GET/POST)
-- Chaînes de secours : `src/app/api/fallback/chains` (GET/POST/DELETE)
-- Audit de conformité : `src/app/api/compliance/audit-log` (GET)
-- Évaluations : `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET)
-- Politiques : `src/app/api/policies` (GET/POST)## 2) SSE + Translation Core
+- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*`
+- Providers/connections: `src/app/api/providers*`
+- Provider nodes: `src/app/api/provider-nodes*`
+- Custom models: `src/app/api/provider-models` (GET/POST/DELETE)
+- Model catalog: `src/app/api/models/route.ts` (GET)
+- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
+- OAuth: `src/app/api/oauth/*`
+- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing`
+- Usage: `src/app/api/usage/*`
+- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*`
+- CLI tooling helpers: `src/app/api/cli-tools/*`
+- IP filter: `src/app/api/settings/ip-filter` (GET/PUT)
+- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT)
+- System prompt: `src/app/api/settings/system-prompt` (GET/PUT)
+- Sessions: `src/app/api/sessions` (GET)
+- Rate limits: `src/app/api/rate-limits` (GET)
+- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state
+- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns
+- Cache stats: `src/app/api/cache/stats` (GET/DELETE)
+- Model availability: `src/app/api/models/availability` (GET/POST)
+- Telemetry: `src/app/api/telemetry/summary` (GET)
+- Budget: `src/app/api/usage/budget` (GET/POST)
+- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE)
+- Compliance audit: `src/app/api/compliance/audit-log` (GET)
+- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET)
+- Policies: `src/app/api/policies` (GET/POST)
-Principaux modules de flux :
+## 2) SSE + Translation Core
-- Entrée : `src/sse/handlers/chat.ts`
-- Orchestration de base : `open-sse/handlers/chatCore.ts`
-- Adaptateurs d'exécution du fournisseur : `open-sse/executors/*`
-- Détection de format/configuration du fournisseur : `open-sse/services/provider.ts`
-- Analyse/résolution du modèle : `src/sse/services/model.ts`, `open-sse/services/model.ts`
-- Logique de repli du compte : `open-sse/services/accountFallback.ts`
-- Registre de traduction : `open-sse/translator/index.ts`
-- Transformations de flux : `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
-- Extraction/normalisation d'utilisation : `open-sse/utils/usageTracking.ts`
-- Analyseur de balises Think : `open-sse/utils/thinkTagParser.ts`
-- Gestionnaire d'intégration : `open-sse/handlers/embeddings.ts`
-- Registre du fournisseur d'intégration : `open-sse/config/embeddingRegistry.ts`
-- Gestionnaire de génération d'images : `open-sse/handlers/imageGeneration.ts`
-- Registre du fournisseur d'images : `open-sse/config/imageRegistry.ts`
-- Désinfection des réponses : `open-sse/handlers/responseSanitizer.ts`
-- Normalisation des rôles : `open-sse/services/roleNormalizer.ts`
+Main flow modules:
-Services (logique métier) :
+- Entry: `src/sse/handlers/chat.ts`
+- Core orchestration: `open-sse/handlers/chatCore.ts`
+- Provider execution adapters: `open-sse/executors/*`
+- Format detection/provider config: `open-sse/services/provider.ts`
+- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts`
+- Account fallback logic: `open-sse/services/accountFallback.ts`
+- Translation registry: `open-sse/translator/index.ts`
+- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
+- Usage extraction/normalization: `open-sse/utils/usageTracking.ts`
+- Think tag parser: `open-sse/utils/thinkTagParser.ts`
+- Embedding handler: `open-sse/handlers/embeddings.ts`
+- Embedding provider registry: `open-sse/config/embeddingRegistry.ts`
+- Image generation handler: `open-sse/handlers/imageGeneration.ts`
+- Image provider registry: `open-sse/config/imageRegistry.ts`
+- Response sanitization: `open-sse/handlers/responseSanitizer.ts`
+- Role normalization: `open-sse/services/roleNormalizer.ts`
-- Sélection/scoration des comptes : `open-sse/services/accountSelector.ts`
-- Gestion du cycle de vie du contexte : `open-sse/services/contextManager.ts`
-- Application du filtre IP : `open-sse/services/ipFilter.ts`
-- Suivi de session : `open-sse/services/sessionManager.ts`
-- Demande de déduplication : `open-sse/services/signatureCache.ts`
-- Injection d'invite système : `open-sse/services/systemPrompt.ts`
-- Penser la gestion budgétaire : `open-sse/services/thinkingBudget.ts`
-- Routage du modèle générique : `open-sse/services/wildcardRouter.ts`
-- Gestion des limites de débit : `open-sse/services/rateLimitManager.ts`
-- Disjoncteur : `open-sse/services/circuitBreaker.ts`
+Services (business logic):
-Modules de couche de domaine :
+- Account selection/scoring: `open-sse/services/accountSelector.ts`
+- Context lifecycle management: `open-sse/services/contextManager.ts`
+- IP filter enforcement: `open-sse/services/ipFilter.ts`
+- Session tracking: `open-sse/services/sessionManager.ts`
+- Request deduplication: `open-sse/services/signatureCache.ts`
+- System prompt injection: `open-sse/services/systemPrompt.ts`
+- Thinking budget management: `open-sse/services/thinkingBudget.ts`
+- Wildcard model routing: `open-sse/services/wildcardRouter.ts`
+- Rate limit management: `open-sse/services/rateLimitManager.ts`
+- Circuit breaker: `open-sse/services/circuitBreaker.ts`
+- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy
+- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions
-- Disponibilité du modèle : `src/lib/domain/modelAvailability.ts`
-- Règles de coûts/budgets : `src/lib/domain/costRules.ts`
-- Politique de repli : `src/lib/domain/fallbackPolicy.ts`
-- Résolveur combo : `src/lib/domain/comboResolver.ts`
-- Politique de verrouillage : `src/lib/domain/lockoutPolicy.ts`
-- Moteur de politique : `src/domain/policyEngine.ts` — verrouillage centralisé → budget → évaluation de secours
-- Catalogue de codes d'erreur : `src/lib/domain/errorCodes.ts`
-- ID de demande : `src/lib/domain/requestId.ts`
-- Délai d'expiration de la récupération : `src/lib/domain/fetchTimeout.ts`
-- Demande de télémétrie : `src/lib/domain/requestTelemetry.ts`
-- Conformité/audit : `src/lib/domain/compliance/index.ts`
-- Exécuteur d'évaluation : `src/lib/domain/evalRunner.ts`
-- Persistance de l'état du domaine : `src/lib/db/domainState.ts` — SQLite CRUD pour les chaînes de secours, les budgets, l'historique des coûts, l'état de verrouillage, les disjoncteurs
+Domain layer modules:
-Modules du fournisseur OAuth (12 fichiers individuels sous `src/lib/oauth/providers/`) :
+- Model availability: `src/lib/domain/modelAvailability.ts`
+- Cost rules/budgets: `src/lib/domain/costRules.ts`
+- Fallback policy: `src/lib/domain/fallbackPolicy.ts`
+- Combo resolver: `src/lib/domain/comboResolver.ts`
+- Lockout policy: `src/lib/domain/lockoutPolicy.ts`
+- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation
+- Error codes catalog: `src/lib/domain/errorCodes.ts`
+- Request ID: `src/lib/domain/requestId.ts`
+- Fetch timeout: `src/lib/domain/fetchTimeout.ts`
+- Request telemetry: `src/lib/domain/requestTelemetry.ts`
+- Compliance/audit: `src/lib/domain/compliance/index.ts`
+- Eval runner: `src/lib/domain/evalRunner.ts`
+- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers
-- Index du registre : `src/lib/oauth/providers/index.ts`
-- Fournisseurs individuels : `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
-- Thin wrapper : `src/lib/oauth/providers.ts` — réexportations à partir de modules individuels## 3) Persistence Layer
+OAuth provider modules (12 individual files under `src/lib/oauth/providers/`):
-Base de données d'état primaire (SQLite) :
+- Registry index: `src/lib/oauth/providers/index.ts`
+- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
+- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules
-- Infrastructure de base : `src/lib/db/core.ts` (better-sqlite3, migrations, WAL)
-- Façade de réexportation : `src/lib/localDb.ts` (fine couche de compatibilité pour les appelants)
-- fichier : `${DATA_DIR}/storage.sqlite` (ou `$XDG_CONFIG_HOME/omniroute/storage.sqlite` lorsqu'il est défini, sinon `~/.omniroute/storage.sqlite`)
-- entités (tables + espaces de noms KV) : ProviderConnections, ProviderNodes, modelAliases, combos, apiKeys, settings, pricing,**customModels**,**proxyConfig**,**ipFilter**,**thinkingBudget**,**systemPrompt**
+## 3) Persistence Layer
-Persistance d'utilisation :
+Primary state DB (SQLite):
-- façade : `src/lib/usageDb.ts` (modules décomposés dans `src/lib/usage/*`)
-- Tables SQLite dans `storage.sqlite` : `usage_history`, `call_logs`, `proxy_logs`
-- des artefacts de fichiers facultatifs restent pour la compatibilité/débogage (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `
/logs/...`)
-- Les anciens fichiers JSON sont migrés vers SQLite par les migrations de démarrage lorsqu'ils sont présents
+- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL)
+- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers)
+- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`)
+- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt**
-Base de données d'état du domaine (SQLite) :
+Usage persistence:
-- `src/lib/db/domainState.ts` — Opérations CRUD pour l'état du domaine
-- Tableaux (créés dans `src/lib/db/core.ts`) : `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
-- Modèle de cache en écriture : les cartes en mémoire font autorité au moment de l'exécution ; les mutations sont écrites de manière synchrone dans SQLite ; l'état est restauré à partir de la base de données lors d'un démarrage à froid## 4) Auth + Security Surfaces
+- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`)
+- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs`
+- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`)
+- legacy JSON files are migrated to SQLite by startup migrations when present
-- Authentification des cookies du tableau de bord : `src/proxy.ts`, `src/app/api/auth/login/route.ts`
-- Génération/vérification de clé API : `src/shared/utils/apiKey.ts`
-- Les secrets du fournisseur ont persisté dans les entrées `providerConnections`
-- Prise en charge du proxy sortant via `open-sse/utils/proxyFetch.ts` (vars env) et `open-sse/utils/networkProxy.ts` (configurable par fournisseur ou global)## 5) Cloud Sync
+Domain State DB (SQLite):
-- Initialisation du planificateur : `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
-- Tâche périodique : `src/shared/services/cloudSyncScheduler.ts`
-- Tâche périodique : `src/shared/services/modelSyncScheduler.ts`
-- Route de contrôle : `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`)
+- `src/lib/db/domainState.ts` — CRUD operations for domain state
+- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
+- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start
+
+## 4) Auth + Security Surfaces
+
+- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts`
+- API key generation/verification: `src/shared/utils/apiKey.ts`
+- Provider secrets persisted in `providerConnections` entries
+- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global)
+
+## 5) Cloud Sync
+
+- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
+- Periodic task: `src/shared/services/cloudSyncScheduler.ts`
+- Periodic task: `src/shared/services/modelSyncScheduler.ts`
+- Control route: `src/app/api/sync/cloud/route.ts`
+
+## Request Lifecycle (`/v1/chat/completions`)
```mermaid
sequenceDiagram
@@ -338,7 +363,9 @@ flowchart TD
Q -- No --> R[Return all unavailable]
```
-Les décisions de secours sont pilotées par « open-sse/services/accountFallback.ts » à l'aide de codes d'état et d'heuristiques de messages d'erreur. Le routage combiné ajoute une protection supplémentaire : les 400 à l'échelle du fournisseur, tels que les échecs de bloc de contenu en amont et de validation de rôle, sont traités comme des échecs locaux du modèle afin que les cibles combinées ultérieures puissent toujours s'exécuter.## OAuth Onboarding and Token Refresh Lifecycle
+Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run.
+
+## OAuth Onboarding and Token Refresh Lifecycle
```mermaid
sequenceDiagram
@@ -368,7 +395,9 @@ sequenceDiagram
Test-->>UI: validation result
```
-L'actualisation pendant le trafic en direct est exécutée dans `open-sse/handlers/chatCore.ts` via l'exécuteur `refreshCredentials()`.## Cloud Sync Lifecycle (Enable / Sync / Disable)
+Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`.
+
+## Cloud Sync Lifecycle (Enable / Sync / Disable)
```mermaid
sequenceDiagram
@@ -400,7 +429,9 @@ sequenceDiagram
Sync-->>UI: disabled
```
-La synchronisation périodique est déclenchée par « CloudSyncScheduler » lorsque le cloud est activé.## Data Model and Storage Map
+Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled.
+
+## Data Model and Storage Map
```mermaid
erDiagram
@@ -501,12 +532,14 @@ erDiagram
}
```
-Fichiers de stockage physique :
+Physical storage files:
-- Base de données d'exécution principale : `${DATA_DIR}/storage.sqlite`
-- lignes de journal de requête : `${DATA_DIR}/log.txt` (artefact compat/debug)
-- archives de charge utile d'appel structurées : `${DATA_DIR}/call_logs/`
-- sessions facultatives de débogage de traduction/demande : `/logs/...`## Deployment Topology
+- primary runtime DB: `${DATA_DIR}/storage.sqlite`
+- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact)
+- structured call payload archives: `${DATA_DIR}/call_logs/`
+- optional translator/request debug sessions: `/logs/...`
+
+## Deployment Topology
```mermaid
flowchart LR
@@ -541,205 +574,249 @@ flowchart LR
### Route and API Modules
-- `src/app/api/v1/*`, `src/app/api/v1beta/*` : API de compatibilité
-- `src/app/api/v1/providers/[provider]/*` : routes dédiées par fournisseur (chat, intégrations, images)
-- `src/app/api/providers*` : fournisseur CRUD, validation, tests
-- `src/app/api/provider-nodes*` : gestion des nœuds compatibles personnalisés
-- `src/app/api/provider-models` : gestion de modèles personnalisés (CRUD)
-- `src/app/api/models/route.ts` : API de catalogue de modèles (alias + modèles personnalisés)
-- `src/app/api/oauth/*` : flux OAuth/device-code
-- `src/app/api/keys*` : cycle de vie de la clé API locale
-- `src/app/api/models/alias` : gestion des alias
-- `src/app/api/combos*` : gestion des combos de repli
-- `src/app/api/pricing` : remplacements de prix pour le calcul des coûts
-- `src/app/api/settings/proxy` : configuration du proxy (GET/PUT/DELETE)
-- `src/app/api/settings/proxy/test` : test de connectivité proxy sortant (POST)
-- `src/app/api/usage/*` : API d'utilisation et de logs
-- `src/app/api/sync/*` + `src/app/api/cloud/*` : synchronisation cloud et assistants orientés cloud
-- `src/app/api/cli-tools/*` : rédacteurs/vérificateurs de configuration CLI locaux
-- `src/app/api/settings/ip-filter` : liste autorisée/liste de blocage IP (GET/PUT)
-- `src/app/api/settings/thinking-budget` : configuration du budget du jeton de réflexion (GET/PUT)
-- `src/app/api/settings/system-prompt` : invite système globale (GET/PUT)
-- `src/app/api/sessions` : liste des sessions actives (GET)
-- `src/app/api/rate-limits` : statut de limite de débit par compte (GET)### Routing and Execution Core
+- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs
+- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images)
+- `src/app/api/providers*`: provider CRUD, validation, testing
+- `src/app/api/provider-nodes*`: custom compatible node management
+- `src/app/api/provider-models`: custom model management (CRUD)
+- `src/app/api/models/route.ts`: model catalog API (aliases + custom models)
+- `src/app/api/oauth/*`: OAuth/device-code flows
+- `src/app/api/keys*`: local API key lifecycle
+- `src/app/api/models/alias`: alias management
+- `src/app/api/combos*`: fallback combo management
+- `src/app/api/pricing`: pricing overrides for cost calculation
+- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE)
+- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST)
+- `src/app/api/usage/*`: usage and logs APIs
+- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers
+- `src/app/api/cli-tools/*`: local CLI config writers/checkers
+- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT)
+- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT)
+- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT)
+- `src/app/api/sessions`: active session listing (GET)
+- `src/app/api/rate-limits`: per-account rate limit status (GET)
-- `src/sse/handlers/chat.ts` : analyse des requêtes, gestion des combos, boucle de sélection de compte
-- `open-sse/handlers/chatCore.ts` : traduction, envoi de l'exécuteur, gestion des nouvelles tentatives/actualisations, configuration du flux
-- `open-sse/executors/*` : comportement du réseau et du format spécifique au fournisseur### Translation Registry and Format Converters
+### Routing and Execution Core
-- `open-sse/translator/index.ts` : registre et orchestration du traducteur
-- Demander des traducteurs : `open-sse/translator/request/*`
-- Traducteurs de réponses : `open-sse/translator/response/*`
-- Constantes de format : `open-sse/translator/formats.ts`### Persistence
+- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop
+- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup
+- `open-sse/executors/*`: provider-specific network and format behavior
-- `src/lib/db/*` : configuration/état persistant et persistance du domaine sur SQLite
-- `src/lib/localDb.ts` : réexportation de compatibilité pour les modules DB
-- `src/lib/usageDb.ts` : façade historique d'utilisation/journaux d'appels au-dessus des tables SQLite## Provider Executor Coverage (Strategy Pattern)
+### Translation Registry and Format Converters
-Chaque fournisseur dispose d'un exécuteur spécialisé étendant `BaseExecutor` (dans `open-sse/executors/base.ts`), qui fournit la construction d'URL, la construction d'en-tête, les nouvelles tentatives avec intervalle exponentiel, les hooks d'actualisation des informations d'identification et la méthode d'orchestration `execute()`.
+- `open-sse/translator/index.ts`: translator registry and orchestration
+- Request translators: `open-sse/translator/request/*`
+- Response translators: `open-sse/translator/response/*`
+- Format constants: `open-sse/translator/formats.ts`
-| Exécuteur testamentaire | Fournisseur(s) | Manutention spéciale |
-| ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------------------------------------------------- |
-| `Exécuteur par défaut` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Configuration dynamique d'URL/d'en-tête par fournisseur |
-| `AntigravityExecutor` | Google Antigravité | ID de projet/session personnalisés, analyse réessayée après |
-| `CodexExecutor` | Codex OpenAI | Injecte des instructions système, force un effort de raisonnement |
-| `CurseurExécuteur` | Curseur IDE | Protocole ConnectRPC, encodage Protobuf, signature de demande via somme de contrôle |
-| `GithubExecutor` | Copilote GitHub | Actualisation du jeton Copilot, en-têtes imitant VSCode |
-| `KiroExécuteur` | AWS CodeWhisperer/Kiro | Format binaire AWS EventStream → conversion SSE |
-| `GeminiCLIEExecutor` | CLI Gémeaux | Cycle d'actualisation du jeton Google OAuth |
+### Persistence
-Tous les autres fournisseurs (y compris les nœuds compatibles personnalisés) utilisent « DefaultExecutor ».## Provider Compatibility Matrix
+- `src/lib/db/*`: persistent config/state and domain persistence on SQLite
+- `src/lib/localDb.ts`: compatibility re-export for DB modules
+- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables
-| Fournisseur | Formater | Authentification | Flux | Hors flux | Actualisation des jetons | API d'utilisation |
-| --------------------- | ----------------- | ------------------------------- | ---------------- | --------- | ------------------------ | ---------------------------- | ------------------------------ |
-| Claude | Claude | Clé API/OAuth | ✅ | ✅ | ✅ | ⚠️ Administrateur uniquement |
-| Gémeaux | Gémeaux | Clé API/OAuth | ✅ | ✅ | ✅ | ⚠️Console Cloud |
-| CLI Gémeaux | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️Console Cloud |
-| Antigravité | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ API de quota complet |
-| OpenAI | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ |
-| Codex | réponses ouvertes | OAuth | ✅ forcé | ❌ | ✅ | ✅ Limites de taux |
-| Copilote GitHub | ouvert | OAuth + Jeton Copilot | ✅ | ✅ | ✅ | ✅ Instantanés de quotas |
-| Curseur | curseur | Somme de contrôle personnalisée | ✅ | ✅ | ❌ | ❌ |
-| Kiro | Kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits |
-| Qwen | ouvert | OAuth | ✅ | ✅ | ✅ | ⚠️ Par demande |
-| Qoder | ouvert | OAuth (de base) | ✅ | ✅ | ✅ | ⚠️ Par demande |
-| OuvrirRouter | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ |
-| GLM/Kimi/MiniMax | Claude | Clé API | ✅ | ✅ | ❌ | ❌ |
-| Recherche profonde | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ |
-| Groq | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ |
-| xAI (Grok) | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ |
-| Mistral | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ |
-| Perplexité | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ |
-| Ensemble IA | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ |
-| IA de feux d'artifice | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ |
-| Cérébraux | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ |
-| Cohérer | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ |
-| NIM NVIDIA | ouvert | Clé API | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage |
+## Provider Executor Coverage (Strategy Pattern)
-Les formats sources détectés incluent :
+Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method.
-- `openaï`
+| Executor | Provider(s) | Special Handling |
+| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- |
+| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider |
+| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing |
+| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort |
+| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum |
+| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers |
+| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion |
+| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle |
+
+All other providers (including custom compatible nodes) use the `DefaultExecutor`.
+
+## Provider Compatibility Matrix
+
+| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API |
+| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ |
+| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only |
+| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API |
+| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits |
+| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots |
+| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ |
+| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits |
+| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request |
+| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request |
+| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ |
+| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+
+## Format Translation Coverage
+
+Detected source formats include:
+
+- `openai`
- `openai-responses`
-- 'Claude'
-- `Gémeaux`
+- `claude`
+- `gemini`
-Les formats cibles incluent :
+Target formats include:
-- Discussion/Réponses OpenAI
- -Claude
-- Enveloppe Gemini/Gemini-CLI/Antigravité
- -Kiro
-- Curseur
+- OpenAI chat/Responses
+- Claude
+- Gemini/Gemini-CLI/Antigravity envelope
+- Kiro
+- Cursor
-Les traductions utilisent**OpenAI comme format hub**— toutes les conversions passent par OpenAI comme intermédiaire :```
+Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate:
+
+```
Source Format → OpenAI (hub) → Target Format
+```
-````
+Translations are selected dynamically based on source payload shape and provider target format.
-Les traductions sont sélectionnées dynamiquement en fonction de la forme de la charge utile source et du format cible du fournisseur.
+Additional processing layers in the translation pipeline:
-Couches de traitement supplémentaires dans le pipeline de traduction :
+- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance
+- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE)
+- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field
+- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema`
--**Désinfection des réponses**— Supprime les champs non standard des réponses au format OpenAI (à la fois en streaming et hors streaming) pour garantir une stricte conformité au SDK.
--**Normalisation des rôles**— Convertit « développeur » → « système » pour les cibles non OpenAI ; fusionne `system` → `user` pour les modèles qui rejettent le rôle système (GLM, ERNIE)
--**Think tag extraction**— Analyse les blocs `...` du contenu dans le champ `reasoning_content`
--**Sortie structurée**— Convertit OpenAI `response_format.json_schema` en `responseMimeType` + `responseSchema` de Gemini## Supported API Endpoints
+## Supported API Endpoints
-| Point de terminaison | Formater | Gestionnaire |
+| Endpoint | Format | Handler |
| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- |
-| `POST /v1/chat/complétions` | Chat OpenAI | `src/sse/handlers/chat.ts` |
-| `POST /v1/messages` | Messages de Claude | Même gestionnaire (détecté automatiquement) |
-| `POST /v1/réponses` | Réponses OpenAI | `open-sse/handlers/responsesHandler.ts` |
-| `POST /v1/embeddings` | Intégrations OpenAI | `open-sse/handlers/embeddings.ts` |
-| `GET /v1/embeddings` | Liste des modèles | Itinéraire API |
-| `POST /v1/images/générations` | Images OpenAI | `open-sse/handlers/imageGeneration.ts` |
-| `GET /v1/images/générations` | Liste des modèles | Itinéraire API |
-| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dédié par fournisseur avec validation du modèle |
-| `POST /v1/providers/{provider}/embeddings` | Intégrations OpenAI | Dédié par fournisseur avec validation du modèle |
-| `POST /v1/providers/{provider}/images/générations` | Images OpenAI | Dédié par fournisseur avec validation du modèle |
-| `POST /v1/messages/count_tokens` | Compte de jetons Claude | Itinéraire API |
-| `GET /v1/models` | Liste des modèles OpenAI | Route API (chat + intégration + image + modèles personnalisés) |
-| `GET /api/models/catalogue` | Catalogue | Tous les modèles regroupés par fournisseur + type |
-| `POST /v1beta/models/*:streamGenerateContent` | Natif des Gémeaux | Itinéraire API |
-| `GET/PUT/DELETE /api/settings/proxy` | Configuration du proxy | Configuration du proxy réseau |
-| `POST /api/settings/proxy/test` | Connectivité proxy | Point de terminaison du test d’intégrité/de connectivité du proxy |
-| `GET/POST/DELETE /api/provider-models` | Modèles de fournisseurs | Métadonnées du modèle de fournisseur soutenant les modèles disponibles personnalisés et gérés |## Bypass Handler
+| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` |
+| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) |
+| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` |
+| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` |
+| `GET /v1/embeddings` | Model listing | API route |
+| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` |
+| `GET /v1/images/generations` | Model listing | API route |
+| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation |
+| `POST /v1/messages/count_tokens` | Claude Token Count | API route |
+| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) |
+| `GET /api/models/catalog` | Catalog | All models grouped by provider + type |
+| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route |
+| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration |
+| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint |
+| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models |
-Le gestionnaire de contournement (`open-sse/utils/bypassHandler.ts`) intercepte les requêtes « jetables » connues de Claude CLI — pings d'échauffement, extractions de titres et nombre de jetons — et renvoie une**fausse réponse**sans consommer de jetons du fournisseur en amont. Ceci est déclenché uniquement lorsque `User-Agent` contient `claude-cli`.## Request Logger Pipeline
+## Bypass Handler
-L'enregistreur de requêtes (`open-sse/utils/requestLogger.ts`) fournit un pipeline de journalisation de débogage en 7 étapes, désactivé par défaut, activé via `ENABLE_REQUEST_LOGS=true` :```
+The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`.
+
+## Request Logger Pipeline
+
+The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`:
+
+```
1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json
→ 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt
-````
+```
-Les fichiers sont écrits dans `/logs//` pour chaque session de requête.## Failure Modes and Resilience
+Files are written to `/logs//` for each request session.
+
+## Failure Modes and Resilience
## 1) Account/Provider Availability
-- Temps de recharge du compte du fournisseur en cas d'erreurs transitoires/taux/auth.
-- repli du compte avant l'échec de la demande
-- repli du modèle combiné lorsque le chemin modèle/fournisseur actuel est épuisé## 2) Token Expiry
+- provider account cooldown on transient/rate/auth errors
+- account fallback before failing request
+- combo model fallback when current model/provider path is exhausted
-- pré-vérification et actualisation avec nouvelle tentative pour les fournisseurs actualisables
-- Nouvelle tentative 401/403 après tentative d'actualisation dans le chemin principal## 3) Stream Safety
+## 2) Token Expiry
-- contrôleur de flux prenant en charge la déconnexion
-- flux de traduction avec vidage de fin de flux et gestion `[DONE]`
-- repli de l'estimation de l'utilisation lorsque les métadonnées d'utilisation du fournisseur sont manquantes## 4) Cloud Sync Degradation
+- pre-check and refresh with retry for refreshable providers
+- 401/403 retry after refresh attempt in core path
-- des erreurs de synchronisation apparaissent mais l'exécution locale continue
-- le planificateur a une logique capable de réessayer, mais l'exécution périodique appelle actuellement une synchronisation à tentative unique par défaut## 5) Data Integrity
+## 3) Stream Safety
-- Migrations de schéma SQLite et hooks de mise à niveau automatique au démarrage
-- chemin de compatibilité de migration JSON → SQLite hérité## Observability and Operational Signals
+- disconnect-aware stream controller
+- translation stream with end-of-stream flush and `[DONE]` handling
+- usage estimation fallback when provider usage metadata is missing
-Sources de visibilité d'exécution :
+## 4) Cloud Sync Degradation
-- les journaux de la console de `src/sse/utils/logger.ts`
-- agrégats d'utilisation par requête dans SQLite (`usage_history`, `call_logs`, `proxy_logs`)
-- captures de charge utile détaillées en quatre étapes dans SQLite (`request_detail_logs`) lorsque `settings.detailed_logs_enabled=true`
-- journal textuel de l'état de la demande dans `log.txt` (facultatif/compat)
-- Journaux facultatifs de requêtes/traductions approfondies sous `logs/` lorsque `ENABLE_REQUEST_LOGS=true`
-- points de terminaison d'utilisation du tableau de bord (`/api/usage/*`) pour la consommation de l'interface utilisateur
+- sync errors are surfaced but local runtime continues
+- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default
-La capture détaillée de la charge utile des requêtes stocke jusqu'à quatre étapes de charge utile JSON par appel routé :
+## 5) Data Integrity
-- demande brute reçue du client
-- requête traduite effectivement envoyée en amont
-- réponse du fournisseur reconstruite en JSON ; les réponses diffusées en continu sont compactées dans le résumé final ainsi que les métadonnées du flux
-- réponse finale du client renvoyée par OmniRoute ; les réponses diffusées en continu sont stockées dans le même formulaire récapitulatif compact## Security-Sensitive Boundaries
+- SQLite schema migrations and auto-upgrade hooks at startup
+- legacy JSON → SQLite migration compatibility path
-- Le secret JWT (`JWT_SECRET`) sécurise la vérification/signature des cookies de session du tableau de bord
-- L'amorçage du mot de passe initial (`INITIAL_PASSWORD`) doit être explicitement configuré pour le provisionnement de première exécution
-- Le secret de la clé API HMAC (`API_KEY_SECRET`) sécurise le format de clé API locale générée
-- Les secrets du fournisseur (clés/jetons API) sont conservés dans la base de données locale et doivent être protégés au niveau du système de fichiers
-- Les points de terminaison de synchronisation dans le cloud s'appuient sur l'authentification par clé API + la sémantique de l'identifiant de la machine## Environment and Runtime Matrix
+## Observability and Operational Signals
-Variables d'environnement activement utilisées par le code :
+Runtime visibility sources:
-- Application/authentification : `JWT_SECRET`, `INITIAL_PASSWORD`
-- Stockage : `DATA_DIR`
-- Comportement du nœud compatible : `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
-- Remplacement facultatif de la base de stockage (Linux/macOS lorsque `DATA_DIR` n'est pas défini) : `XDG_CONFIG_HOME`
-- Hachage de sécurité : `API_KEY_SECRET`, `MACHINE_ID_SALT`
-- Journalisation : `ENABLE_REQUEST_LOGS`
-- URL de synchronisation/cloud : `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
-- Proxy sortant : `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` et variantes minuscules
-- Indicateurs de fonctionnalités SOCKS5 : `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
-- Aides de plate-forme/d'exécution (pas de configuration spécifique à l'application) : `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`## Known Architectural Notes
+- console logs from `src/sse/utils/logger.ts`
+- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`)
+- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true`
+- textual request status log in `log.txt` (optional/compat)
+- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true`
+- dashboard usage endpoints (`/api/usage/*`) for UI consumption
-1. `usageDb` et `localDb` partagent la même politique de répertoire de base (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) avec la migration des fichiers hérités.
-2. `/api/v1/route.ts` délègue au même constructeur de catalogue unifié utilisé par `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) pour éviter la dérive sémantique.
-3. L'enregistreur de requêtes écrit les en-têtes/corps complets lorsqu'il est activé ; traiter le répertoire des journaux comme sensible.
-4. Le comportement du cloud dépend de l'exactitude de « NEXT_PUBLIC_BASE_URL » et de l'accessibilité du point de terminaison du cloud.
-5. Le répertoire `open-sse/` est publié sous le nom `@omniroute/open-sse`**npm workspace package**. Le code source l'importe via `@omniroute/open-sse/...` (résolu par Next.js `transpilePackages`). Les chemins de fichiers dans ce document utilisent toujours le nom de répertoire « open-sse/ » pour des raisons de cohérence.
-6. Les graphiques du tableau de bord utilisent**Recharts**(basé sur SVG) pour des visualisations analytiques accessibles et interactives (graphiques à barres d'utilisation du modèle, tableaux de répartition des fournisseurs avec taux de réussite).
-7. Les tests E2E utilisent**Playwright**(`tests/e2e/`), exécutés via `npm run test:e2e`. Les tests unitaires utilisent**l'exécuteur de test Node.js**(`tests/unit/`), exécutés via `npm run test:unit`. Le code source sous `src/` est**TypeScript**(`.ts`/`.tsx`) ; l'espace de travail `open-sse/` reste JavaScript (`.js`).
-8. La page Paramètres est organisée en 5 onglets : Sécurité, Routage (6 stratégies globales : remplissage en premier, round-robin, p2c, aléatoire, moins utilisé, coût optimisé), Résilience (limites de débit modifiables, disjoncteur, politiques), IA (budget de réflexion, invite système, cache d'invite), Avancé (proxy).## Operational Verification Checklist
+Detailed request payload capture stores up to four JSON payload stages per routed call:
-- Construire à partir des sources : `npm run build`
-- Construire l'image Docker : `docker build -t omniroute .`
-- Démarrez le service et vérifiez :
+- raw request received from the client
+- translated request actually sent upstream
+- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata
+- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form
+
+## Security-Sensitive Boundaries
+
+- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing
+- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning
+- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format
+- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level
+- Cloud sync endpoints rely on API key auth + machine id semantics
+
+## Environment and Runtime Matrix
+
+Environment variables actively used by code:
+
+- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD`
+- Storage: `DATA_DIR`
+- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
+- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME`
+- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT`
+- Logging: `ENABLE_REQUEST_LOGS`
+- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
+- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants
+- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
+- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`
+
+## Known Architectural Notes
+
+1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration.
+2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift.
+3. Request logger writes full headers/body when enabled; treat log directory as sensitive.
+4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability.
+5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency.
+6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates).
+7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`).
+8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy).
+9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed.
+10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22.
+11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible.
+
+## Operational Verification Checklist
+
+- Build from source: `npm run build`
+- Build Docker image: `docker build -t omniroute .`
+- Start service and verify:
- `GET /api/settings`
- `GET /api/v1/models`
-- L'URL de base cible CLI doit être `http://:20128/v1` lorsque `PORT=20128`
+- CLI target base URL should be `http://:20128/v1` when `PORT=20128`
diff --git a/docs/i18n/fr/docs/FEATURES.md b/docs/i18n/fr/docs/FEATURES.md
index 114a35308c..c6b62030d1 100644
--- a/docs/i18n/fr/docs/FEATURES.md
+++ b/docs/i18n/fr/docs/FEATURES.md
@@ -4,102 +4,168 @@
---
-Guide visuel de chaque section du tableau de bord OmniRoute.---
+
+
+Visual guide to every section of the OmniRoute dashboard.
+
+---
## 🔌 Providers
-Gérez les connexions des fournisseurs d'IA : fournisseurs OAuth (Claude Code, Codex, Gemini CLI), fournisseurs de clés API (Groq, DeepSeek, OpenRouter) et fournisseurs gratuits (Qoder, Qwen, Kiro). Les comptes Kiro incluent le suivi du solde créditeur : crédits restants, allocation totale et date de renouvellement visibles dans Tableau de bord → Utilisation.
+Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage.
+
+
---
## 🎨 Combos
-Créez des combinaisons de routage de modèles avec 6 stratégies : prioritaire, pondérée, à tour de rôle, aléatoire, la moins utilisée et optimisée en termes de coûts. Chaque combo enchaîne plusieurs modèles avec un repli automatique et comprend des modèles rapides et des contrôles de préparation.
+Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks.
+
+
---
## 📊 Analytics
-Analyses d'utilisation complètes avec consommation de jetons, estimations de coûts, cartes thermiques d'activité, graphiques de distribution hebdomadaire et répartitions par fournisseur.
+Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns.
+
+
---
## 🏥 System Health
-Surveillance en temps réel : disponibilité, mémoire, version, centiles de latence (p50/p95/p99), statistiques du cache et états des disjoncteurs du fournisseur.
+Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states.
+
+
---
## 🔧 Translator Playground
-Quatre modes de débogage des traductions d'API :**Playground**(convertisseur de format),**Chat Tester**(requêtes en direct),**Test Bench**(tests par lots) et**Live Monitor**(flux en temps réel).
+Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream).
+
+
---
## 🎮 Model Playground _(v2.0.9+)_
-Testez n’importe quel modèle directement depuis le tableau de bord. Sélectionnez le fournisseur, le modèle et le point de terminaison, rédigez des invites avec Monaco Editor, diffusez les réponses en temps réel, abandonnez en cours de route et affichez les métriques de synchronisation.---
+Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics.
+
+---
## 🎨 Themes _(v2.0.5+)_
-Thèmes de couleurs personnalisables pour l'ensemble du tableau de bord. Choisissez parmi 7 couleurs prédéfinies (corail, bleu, rouge, vert, violet, orange, cyan) ou créez un thème personnalisé en choisissant n'importe quelle couleur hexadécimale. Prend en charge les modes clair, sombre et système.---
+Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode.
+
+---
## ⚙️ Settings
-Panneau de paramètres complet avec onglets :
+Comprehensive settings panel with tabs:
--**Général**— Stockage système, gestion des sauvegardes (base de données d'exportation/importation) -**Apparence**— Sélecteur de thème (sombre/clair/système), préréglages de thèmes de couleurs et couleurs personnalisées, visibilité du journal de santé, contrôles de visibilité des éléments de la barre latérale -**Sécurité**— Protection des points de terminaison de l'API, blocage des fournisseurs personnalisés, filtrage IP, informations de session -**Routage**— Alias de modèle, dégradation des tâches en arrière-plan -**Résilience**— Persistance des limites de débit, réglage du disjoncteur, désactivation automatique des comptes interdits, surveillance de l'expiration des fournisseurs -**Avancé**— Remplacements de configuration, piste d'audit de configuration, mode de dégradation de repli
+- **General** — System storage, backup management (export/import database)
+- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls
+- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info
+- **Routing** — Model aliases, background task degradation
+- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration
+- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode
+
+
---
## 🔧 CLI Tools
-Configuration en un clic pour les outils de codage d'IA : Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor et Factory Droid. Comprend l'application/la réinitialisation automatisée de la configuration, les profils de connexion et le mappage de modèle.
+One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping.
+
+
---
## 🤖 CLI Agents _(v2.0.11+)_
-Tableau de bord pour découvrir et gérer les agents CLI. Affiche une grille de 14 agents intégrés (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) avec :
+Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with:
--**Statut de l'installation**— Installé/Introuvable avec détection de version -**Badges de protocole**— stdio, HTTP, etc. -**Agents personnalisés**— Enregistrez n'importe quel outil CLI via un formulaire (nom, binaire, commande de version, arguments de spawn) -**CLI Fingerprint Matching**— Bascule par fournisseur pour faire correspondre les signatures de requête CLI natives, réduisant ainsi le risque d'interdiction tout en préservant l'adresse IP du proxy.---
+- **Installation status** — Installed / Not Found with version detection
+- **Protocol badges** — stdio, HTTP, etc.
+- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args)
+- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP
+
+---
+
+## 🔗 Context Relay _(v3.5.5+)_
+
+A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context.
+
+Configurable via combo-level or global settings:
+- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%)
+- **Max Messages For Summary** — How much recent history to condense
+- **Summary Model** — Optional override model for generating the handoff summary
+
+Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md).
+
+---
+
+## 🛡️ Proxy Hardening _(v3.5.5+)_
+
+Comprehensive proxy configuration enforcement across the entire request pipeline:
+
+- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments
+- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings
+- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22
+- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS
+
+---
## 🖼️ Media _(v2.0.3+)_
-Générez des images, des vidéos et de la musique à partir du tableau de bord. Prend en charge OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open et MusicGen.---
+Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen.
+
+---
## 📝 Request Logs
-Journalisation des demandes en temps réel avec filtrage par fournisseur, modèle, compte et clé API. Affiche les codes d'état, l'utilisation des jetons, la latence et les détails de la réponse.
+Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details.
+
+
---
## 🌐 API Endpoint
-Votre point de terminaison d'API unifié avec répartition des capacités : achèvements de chat, API de réponses, intégrations, génération d'images, reclassement, transcription audio, synthèse vocale, modérations et clés API enregistrées. Intégration de Cloudflare Quick Tunnel et prise en charge du proxy cloud pour l'accès à distance.
+Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access.
+
+
---
## 🔑 API Key Management
-Créez, définissez et révoquez des clés API. Chaque clé peut être limitée à des modèles/fournisseurs spécifiques avec un accès complet ou des autorisations en lecture seule. Gestion visuelle des clés avec suivi de l'utilisation.---
+Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking.
+
+---
## 📋 Audit Log
-Suivi des actions administratives avec filtrage par type d'action, acteur, cible, adresse IP et horodatage. Historique complet des événements de sécurité.---
+Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history.
+
+---
## 🖥️ Desktop Application
-Application de bureau Native Electron pour Windows, macOS et Linux. Exécutez OmniRoute en tant qu'application autonome avec intégration dans la barre d'état système, prise en charge hors ligne, mise à jour automatique et installation en un clic.
+Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install.
-Principales caractéristiques :
+Key features:
-- Sondage de préparation du serveur (pas d'écran vide au démarrage à froid)
-- Barre d'état système avec gestion des ports
-- Politique de sécurité du contenu
-- Verrouillage à instance unique
-- Mise à jour automatique au redémarrage
-- Interface utilisateur conditionnelle à la plate-forme (feux de signalisation macOS, barre de titre par défaut Windows/Linux)
-- Emballage de build Hardened Electron — les « node_modules » liés symboliquement dans le bundle autonome sont détectés et rejetés avant l'empaquetage, évitant ainsi la dépendance d'exécution sur la machine de build (v2.5.5+)
+- Server readiness polling (no blank screen on cold start)
+- System tray with port management
+- Content Security Policy
+- Single-instance lock
+- Auto-update on restart
+- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar)
+- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+)
-📖 Voir [`electron/README.md`](../electron/README.md) pour une documentation complète.
+📖 See [`electron/README.md`](../electron/README.md) for full documentation.
diff --git a/docs/i18n/fr/docs/TROUBLESHOOTING.md b/docs/i18n/fr/docs/TROUBLESHOOTING.md
index ef38899445..219c35c693 100644
--- a/docs/i18n/fr/docs/TROUBLESHOOTING.md
+++ b/docs/i18n/fr/docs/TROUBLESHOOTING.md
@@ -4,68 +4,142 @@
---
-Problèmes courants et solutions pour OmniRoute.---
+
+
+Common problems and solutions for OmniRoute.
+
+---
## Quick Fixes
-| Problème | Solutions |
-| ---------------------------------------------- | -------------------------------------------------------------------------------------- | --- |
-| La première connexion ne fonctionne pas | Définissez `INITIAL_PASSWORD` dans `.env` (pas de valeur par défaut codée en dur) |
-| Le tableau de bord s'ouvre sur le mauvais port | Définissez `PORT=20128` et `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
-| Aucun journal de requête sous `logs/` | Définir `ENABLE_REQUEST_LOGS=true` |
-| EACCES : autorisation refusée | Définissez `DATA_DIR=/path/to/writable/dir` pour remplacer `~/.omniroute` |
-| La stratégie de routage ne sauvegarde pas | Mise à jour vers v1.4.11+ (correctif du schéma Zod pour la persistance des paramètres) | --- |
+| Problem | Solution |
+| ----------------------------- | ------------------------------------------------------------------ |
+| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) |
+| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
+| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` |
+| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` |
+| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) |
+| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below |
+| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below |
+
+---
+
+## Node.js Compatibility
+
+
+
+### Login page crashes or shows "Module self-registration" error
+
+**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database.
+
+**Symptoms:**
+- Login page shows a blank screen or a server error
+- Console shows `Error: Module did not self-register` or similar native binding errors
+- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected
+
+**Fix:**
+
+1. Install Node.js 22 LTS (recommended):
+ ```bash
+ nvm install 22
+ nvm use 22
+ ```
+2. Verify your version: `node --version` should show `v22.x.x`
+3. Reinstall OmniRoute: `npm install -g omniroute`
+4. Restart: `omniroute`
+
+> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**.
+
+---
+
+## Proxy Issues
+
+
+
+### Provider validation shows "fetch failed"
+
+**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing.
+
+**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically.
+
+### Token health check fails with "fetch failed"
+
+**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection.
+
+**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+.
+
+### SOCKS5 proxy returns "invalid onRequestStart method"
+
+**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation.
+
+**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+.
+
+---
## Provider Issues
### "Language model did not provide messages"
-**Cause :**Quota de fournisseur épuisé.
+**Cause:** Provider quota exhausted.
-**Correction :**
+**Fix:**
-1. Vérifiez le suivi des quotas du tableau de bord
-2. Utilisez un combo avec des niveaux de secours
-3. Passez au niveau moins cher/gratuit### Rate Limiting
+1. Check dashboard quota tracker
+2. Use a combo with fallback tiers
+3. Switch to cheaper/free tier
-**Cause :**Quota d'abonnement épuisé.
+### Rate Limiting
-**Correction :**
+**Cause:** Subscription quota exhausted.
-- Ajouter une solution de secours : `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-- Utilisez GLM/MiniMax comme sauvegarde bon marché### OAuth Token Expired
+**Fix:**
-OmniRoute actualise automatiquement les jetons. Si les problèmes persistent :
+- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Use GLM/MiniMax as cheap backup
-1. Tableau de bord → Fournisseur → Reconnecter
-2. Supprimez et rajoutez la connexion du fournisseur---
+### OAuth Token Expired
+
+OmniRoute auto-refreshes tokens. If issues persist:
+
+1. Dashboard → Provider → Reconnect
+2. Delete and re-add the provider connection
+
+---
## Cloud Issues
### Cloud Sync Errors
-1. Vérifiez que `BASE_URL` pointe vers votre instance en cours d'exécution (par exemple, `http://localhost:20128`)
-2. Vérifiez que « CLOUD_URL » pointe vers votre point de terminaison cloud (par exemple, « https://omniroute.dev »)
-3. Gardez les valeurs `NEXT_PUBLIC_*` alignées avec les valeurs côté serveur### Cloud `stream=false` Returns 500
+1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`)
+2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`)
+3. Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**Symptôme :**`Jeton inattendu 'd'...` sur le point de terminaison cloud pour les appels sans streaming.
+### Cloud `stream=false` Returns 500
-**Cause :**Upstream renvoie la charge utile SSE alors que le client attend du JSON.
+**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls.
-**Solution de contournement :**Utilisez « stream=true » pour les appels directs vers le cloud. Le runtime local inclut le repli SSE → JSON.### Cloud Says Connected but "Invalid API key"
+**Cause:** Upstream returns SSE payload while client expects JSON.
-1. Créez une nouvelle clé à partir du tableau de bord local (`/api/keys`)
-2. Exécutez la synchronisation cloud : Activer le cloud → Synchroniser maintenant
-3. Les clés anciennes/non synchronisées peuvent toujours renvoyer « 401 » sur le cloud---
+**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback.
+
+### Cloud Says Connected but "Invalid API key"
+
+1. Create a fresh key from local dashboard (`/api/keys`)
+2. Run cloud sync: Enable Cloud → Sync Now
+3. Old/non-synced keys can still return `401` on cloud
+
+---
## Docker Issues
### CLI Tool Shows Not Installed
-1. Vérifiez les champs d'exécution : `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
-2. Pour le mode portable : utilisez la cible d'image `runner-cli` (CLI fournies)
-3. Pour le mode de montage de l'hôte : définissez `CLI_EXTRA_PATHS` et montez le répertoire bin de l'hôte en lecture seule
-4. Si `installed=true` et `runnable=false` : le binaire a été trouvé mais le contrôle de santé a échoué### Quick Runtime Validation
+1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
+2. For portable mode: use image target `runner-cli` (bundled CLIs)
+3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only
+4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck
+
+### Quick Runtime Validation
```bash
curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
@@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,
### High Costs
-1. Vérifiez les statistiques d'utilisation dans le tableau de bord → Utilisation
-2. Basculez le modèle principal vers GLM/MiniMax
-3. Utilisez l'offre gratuite (Gemini CLI, Qoder) pour les tâches non critiques
-4. Définissez les budgets de coûts par clé API : Tableau de bord → Clés API → Budget---
+1. Check usage stats in Dashboard → Usage
+2. Switch primary model to GLM/MiniMax
+3. Use free tier (Gemini CLI, Qoder) for non-critical tasks
+4. Set cost budgets per API key: Dashboard → API Keys → Budget
+
+---
## Debugging
### Enable Request Logs
-Définissez `ENABLE_REQUEST_LOGS=true` dans votre fichier `.env`. Les journaux apparaissent dans le répertoire `logs/`.### Check Provider Health
+Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory.
+
+### Check Provider Health
```bash
# Health dashboard
@@ -100,101 +178,135 @@ curl http://localhost:20128/api/monitoring/health
### Runtime Storage
-- État principal : `${DATA_DIR}/storage.sqlite` (fournisseurs, combos, alias, clés, paramètres)
-- Utilisation : tables SQLite dans `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + facultatif `${DATA_DIR}/log.txt` et `${DATA_DIR}/call_logs/`
-- Journaux de requête : `/logs/...` (quand `ENABLE_REQUEST_LOGS=true`)---
+- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings)
+- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/`
+- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`)
+
+---
## Circuit Breaker Issues
### Provider stuck in OPEN state
-Lorsque le disjoncteur d'un fournisseur est OUVERT, les demandes sont bloquées jusqu'à l'expiration du temps de recharge.
+When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires.
-**Correction :**
+**Fix:**
-1. Accédez à**Tableau de bord → Paramètres → Résilience**
-2. Vérifiez la carte de disjoncteur du fournisseur concerné
-3. Cliquez sur**Réinitialiser tout**pour effacer tous les disjoncteurs ou attendez l'expiration du temps de recharge.
-4. Vérifiez que le fournisseur est réellement disponible avant de réinitialiser### Provider keeps tripping the circuit breaker
+1. Go to **Dashboard → Settings → Resilience**
+2. Check the circuit breaker card for the affected provider
+3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire
+4. Verify the provider is actually available before resetting
-Si un fournisseur entre à plusieurs reprises dans l’état OPEN :
+### Provider keeps tripping the circuit breaker
-1. Vérifiez**Tableau de bord → Santé → Santé du fournisseur**pour connaître le modèle d'échec.
-2. Accédez à**Paramètres → Résilience → Profils de fournisseur**et augmentez le seuil d'échec.
-3. Vérifiez si le fournisseur a modifié les limites de l'API ou nécessite une ré-authentification
-4. Examinez la télémétrie de latence : une latence élevée peut provoquer des échecs liés au délai d'attente.---
+If a provider repeatedly enters OPEN state:
+
+1. Check **Dashboard → Health → Provider Health** for the failure pattern
+2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold
+3. Check if the provider has changed API limits or requires re-authentication
+4. Review latency telemetry — high latency may cause timeout-based failures
+
+---
## Audio Transcription Issues
### "Unsupported model" error
-- Assurez-vous d'utiliser le préfixe correct : `deepgram/nova-3` ou `assemblyai/best`
-- Vérifiez que le fournisseur est connecté dans**Tableau de bord → Fournisseurs**### Transcription returns empty or fails
+- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best`
+- Verify the provider is connected in **Dashboard → Providers**
-- Vérifiez les formats audio pris en charge : `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm`
-- Vérifiez que la taille du fichier est dans les limites du fournisseur (généralement < 25 Mo)
-- Vérifier la validité de la clé API du fournisseur dans la carte du fournisseur---
+### Transcription returns empty or fails
+
+- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm`
+- Verify file size is within provider limits (typically < 25MB)
+- Check provider API key validity in the provider card
+
+---
## Translator Debugging
-Utilisez**Tableau de bord → Traducteur**pour déboguer les problèmes de traduction de format :
+Use **Dashboard → Translator** to debug format translation issues:
-| Mode | Quand utiliser |
-| ---------------------- | -------------------------------------------------------------------------------------------------------------------- | ------------------------ |
-| **Aire de jeux** | Comparez les formats d'entrée/sortie côte à côte : collez une requête qui a échoué pour voir comment elle se traduit |
-| **Testeur de chat** | Envoyez des messages en direct et inspectez la charge utile complète de la demande/réponse, y compris les en-têtes |
-| **Banc d'essai** | Exécutez des tests par lots sur les combinaisons de formats pour identifier les traductions défectueuses |
-| **Moniteur en direct** | Observez le flux de requêtes en temps réel pour détecter les problèmes de traduction intermittents | ### Common format issues |
+| Mode | When to Use |
+| ---------------- | -------------------------------------------------------------------------------------------- |
+| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates |
+| **Chat Tester** | Send live messages and inspect the full request/response payload including headers |
+| **Test Bench** | Run batch tests across format combinations to find which translations are broken |
+| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues |
--**Les balises de réflexion n'apparaissent pas**— Vérifiez si le fournisseur cible prend en charge la réflexion et le paramètre de budget de réflexion -**Abandon des appels d'outils**— Certaines traductions de format peuvent supprimer des champs non pris en charge ; vérifier en mode Playground -**Invite système manquante**— Claude et Gemini gèrent les invites système différemment ; vérifier le résultat de la traduction -**Le SDK renvoie une chaîne brute au lieu d'un objet**— Corrigé dans la version 1.1.0 : le désinfectant de réponse supprime désormais les champs non standard (`x_groq`, `usage_breakdown`, etc.) qui provoquent des échecs de validation OpenAI SDK Pydantic -**GLM/ERNIE rejette le rôle `système`**— Corrigé dans la version 1.1.0 : le normalisateur de rôle fusionne automatiquement les messages système dans les messages utilisateur pour les modèles incompatibles -**Rôle de « développeur » non reconnu**— Corrigé dans la version 1.1.0 : automatiquement converti en « système » pour les fournisseurs non OpenAI -**`json_schema` ne fonctionne pas avec Gemini**— Corrigé dans la v1.1.0 : `response_format` est maintenant converti en `responseMimeType` + `responseSchema` de Gemini---
+### Common format issues
+
+- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting
+- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode
+- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output
+- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures
+- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models
+- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers
+- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema`
+
+---
## Resilience Settings
### Auto rate-limit not triggering
-- La limite de débit automatique s'applique uniquement aux fournisseurs de clés API (pas à OAuth/abonnement)
-- Vérifiez que**Paramètres → Résilience → Profils de fournisseur**a activé la limite de débit automatique.
-- Vérifiez si le fournisseur renvoie les codes d'état « 429 » ou les en-têtes « Retry-After »### Tuning exponential backoff
+- Auto rate-limit only applies to API key providers (not OAuth/subscription)
+- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled
+- Check if the provider returns `429` status codes or `Retry-After` headers
-Les profils de fournisseur prennent en charge ces paramètres :
+### Tuning exponential backoff
--**Délai de base**— Temps d'attente initial après le premier échec (par défaut : 1 s) -**Délai maximum**— Limite maximale du temps d'attente (par défaut : 30 s) -**Multiplicateur**— De combien augmenter le délai par échec consécutif (par défaut : 2x)### Anti-thundering herd
+Provider profiles support these settings:
-Lorsque de nombreuses requêtes simultanées atteignent un fournisseur à débit limité, OmniRoute utilise mutex + limitation de débit automatique pour sérialiser les requêtes et éviter les échecs en cascade. Ceci est automatique pour les fournisseurs de clés API.---
+- **Base delay** — Initial wait time after first failure (default: 1s)
+- **Max delay** — Maximum wait time cap (default: 30s)
+- **Multiplier** — How much to increase delay per consecutive failure (default: 2x)
+
+### Anti-thundering herd
+
+When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers.
+
+---
## Optional RAG / LLM failure taxonomy (16 problems)
-Certains utilisateurs d'OmniRoute placent la passerelle devant les RAG ou les piles d'agents. Dans ces configurations, il est courant de voir un schéma étrange : OmniRoute semble sain (fournisseurs activés, profils de routage corrects, aucune alerte de limite de débit) mais la réponse finale est toujours fausse.
+Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong.
-En pratique, ces incidents proviennent généralement du pipeline RAG en aval, et non de la passerelle elle-même.
+In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself.
-Si vous souhaitez un vocabulaire partagé pour décrire ces échecs, vous pouvez utiliser le WFGY ProblemMap, une ressource textuelle externe sous licence MIT qui définit seize modèles d'échecs RAG/LLM récurrents. À un niveau élevé, il couvre :
+If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers:
-- dérive de récupération et limites de contexte brisées
-- index vides ou obsolètes et magasins de vecteurs
-- intégration versus inadéquation sémantique
-- problèmes d'assemblage rapide et de fenêtre contextuelle
-- effondrement de la logique et réponses trop confiantes
-- échecs de la longue chaîne et de la coordination des agents
-- mémoire multi-agents et dérive des rôles
-- problèmes de déploiement et d'ordre d'amorçage
+- retrieval drift and broken context boundaries
+- empty or stale indexes and vector stores
+- embedding versus semantic mismatch
+- prompt assembly and context window issues
+- logic collapse and overconfident answers
+- long chain and agent coordination failures
+- multi agent memory and role drift
+- deployment and bootstrap ordering problems
-L'idée est simple :
+The idea is simple:
-1. Lorsque vous enquêtez sur une mauvaise réponse, capturez :
- - tâche et demande de l'utilisateur
- - combo d'itinéraire ou de fournisseur dans OmniRoute
- - tout contexte RAG utilisé en aval (documents récupérés, appels d'outils, etc.)
-2. Cartographiez l'incident avec un ou deux numéros WFGY ProblemMap (« No.1 » … « No.16 »).
-3. Stockez le numéro dans votre propre tableau de bord, runbook ou suivi des incidents à côté des journaux OmniRoute.
-4. Utilisez la page WFGY correspondante pour décider si vous devez modifier votre pile RAG, votre récupérateur ou votre stratégie de routage.
+1. When you investigate a bad response, capture:
+ - user task and request
+ - route or provider combo in OmniRoute
+ - any RAG context used downstream (retrieved documents, tool calls, etc)
+2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`).
+3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs.
+4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy.
-Texte intégral et recettes concrètes en direct ici (licence MIT, texte uniquement) :
+Full text and concrete recipes live here (MIT license, text only):
-[WFGY ProblemMap README](https://github.com/onestadao/WFGY/blob/main/ProblemMap/README.md)
+[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md)
-Vous pouvez ignorer cette section si vous n'exécutez pas de RAG ou de pipelines d'agent derrière OmniRoute.---
+You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute.
+
+---
## Still Stuck?
--**Problèmes GitHub** : [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**Architecture** : Voir [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) pour les détails internes -**Référence API** : voir [`docs/API_REFERENCE.md`](API_REFERENCE.md) pour tous les points de terminaison -**Tableau de bord de santé** : consultez**Tableau de bord → Santé**pour connaître l'état du système en temps réel -**Traducteur** : utilisez**Tableau de bord → Traducteur**pour déboguer les problèmes de format
+- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details
+- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints
+- **Health Dashboard**: Check **Dashboard → Health** for real-time system status
+- **Translator**: Use **Dashboard → Translator** to debug format issues
diff --git a/docs/i18n/fr/llm.txt b/docs/i18n/fr/llm.txt
new file mode 100644
index 0000000000..d6223886bf
--- /dev/null
+++ b/docs/i18n/fr/llm.txt
@@ -0,0 +1,476 @@
+# OmniRoute (Français)
+
+🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt)
+
+---
+
+
+> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app.
+
+## Aperçu
+
+OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free.
+
+**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost.
+
+**Current version:** 3.5.5
+
+## Tech Stack
+
+- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`)
+- **Framework:** Next.js 16 (App Router) with TypeScript 5.9
+- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations)
+- **State management:** Zustand (client), SQLite (server persistence)
+- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons
+- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth
+- **Schemas:** Zod v4 for all API / MCP input validation
+- **Background jobs:** Custom token health check scheduler, 24h model auto-sync
+- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses
+- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine
+- **i18n:** next-intl with 30 languages
+- **Desktop:** Electron (cross-platform: Windows, macOS, Linux)
+- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`)
+
+## Project Structure
+
+```
+/
+├── src/ # Main application source
+│ ├── app/ # Next.js App Router pages and API routes
+│ │ ├── (dashboard)/ # Dashboard UI pages
+│ │ │ └── dashboard/
+│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents)
+│ │ │ ├── analytics/ # Usage analytics and charts
+│ │ │ ├── api-manager/ # API key management
+│ │ │ ├── audit/ # Audit logs
+│ │ │ ├── auto-combo/ # Auto-combo engine dashboard
+│ │ │ ├── cache/ # Cache dashboard (semantic cache stats)
+│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.)
+│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates)
+│ │ │ ├── costs/ # Cost tracking per provider/model
+│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs
+│ │ │ ├── health/ # System health (uptime, circuit breakers, latency)
+│ │ │ ├── limits/ # Rate limits dashboard
+│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed)
+│ │ │ ├── media/ # Image/video/music generation + transcription
+│ │ │ ├── memory/ # Memory system dashboard
+│ │ │ ├── onboarding/ # Onboarding wizard
+│ │ │ ├── playground/ # Model playground (Monaco editor, streaming)
+│ │ │ ├── providers/ # Provider management (OAuth + API key + free)
+│ │ │ ├── search-tools/ # Search tools configuration
+│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced)
+│ │ │ ├── skills/ # Skills system dashboard
+│ │ │ ├── translator/ # Format translator + debug tools
+│ │ │ └── usage/ # Usage history
+│ │ ├── api/ # REST API endpoints (51 route directories)
+│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings,
+│ │ │ │ # images, audio, videos, music, moderations, rerank, search,
+│ │ │ │ # responses, messages, registered-keys, quotas, accounts)
+│ │ │ ├── v1beta/ # Gemini-compatible API
+│ │ │ ├── a2a/ # A2A agent management API
+│ │ │ ├── acp/ # ACP agent management API
+│ │ │ ├── oauth/ # OAuth flows per provider
+│ │ │ ├── providers/ # Provider CRUD and batch testing
+│ │ │ ├── models/ # Dashboard model listing and aliases
+│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains)
+│ │ │ ├── memory/ # Memory system API
+│ │ │ ├── skills/ # Skills system API
+│ │ │ ├── evals/ # Eval runner API
+│ │ │ ├── mcp/ # MCP HTTP transport API
+│ │ │ ├── search/ # Search provider API
+│ │ │ ├── webhooks/ # Webhook management
+│ │ │ ├── tunnels/ # Cloudflare tunnel management
+│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.)
+│ │ ├── landing/ # Landing page
+│ │ ├── login/ # Login page
+│ │ ├── forgot-password/ # Password recovery
+│ │ ├── status/ # Status page
+│ │ └── docs/ # In-app documentation
+│ ├── domain/ # Domain types and policy engine
+│ │ ├── policyEngine.ts # Central policy engine
+│ │ ├── comboResolver.ts # Combo resolution logic
+│ │ ├── costRules.ts # Cost calculation rules
+│ │ ├── degradation.ts # Graceful degradation
+│ │ ├── fallbackPolicy.ts # Fallback behavior
+│ │ ├── lockoutPolicy.ts # Account lockout logic
+│ │ ├── modelAvailability.ts # Model availability checks
+│ │ ├── providerExpiration.ts # Provider credential expiration
+│ │ ├── quotaCache.ts # Quota caching layer
+│ │ ├── configAudit.ts # Configuration auditing
+│ │ └── responses.ts # Domain response types
+│ ├── i18n/ # Internationalization
+│ │ └── messages/ # 30 language JSON files
+│ ├── lib/ # Core libraries
+│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server
+│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting)
+│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup
+│ │ │ └── streaming.ts # SSE streaming for A2A
+│ │ ├── acp/ # Agent Communication Protocol registry and manager
+│ │ ├── compliance/ # Compliance policy engine
+│ │ ├── db/ # SQLite database layer (21 modules + migrations)
+│ │ │ ├── core.ts # Database initialization, connection, schema
+│ │ │ ├── providers.ts # Provider connection CRUD
+│ │ │ ├── models.ts # Model catalog management
+│ │ │ ├── combos.ts # Combo configuration
+│ │ │ ├── apiKeys.ts # API key management
+│ │ │ ├── settings.ts # Settings persistence
+│ │ │ ├── backup.ts # Database backup/restore
+│ │ │ ├── proxies.ts # Proxy registry
+│ │ │ ├── prompts.ts # Prompt templates
+│ │ │ ├── webhooks.ts # Webhook subscriptions
+│ │ │ ├── detailedLogs.ts # Detailed request logging
+│ │ │ ├── domainState.ts # Domain state persistence
+│ │ │ ├── registeredKeys.ts # Registered API keys with quotas
+│ │ │ ├── quotaSnapshots.ts # Quota snapshot history
+│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings
+│ │ │ ├── cliToolState.ts # CLI tool state tracking
+│ │ │ ├── encryption.ts # Data encryption
+│ │ │ ├── readCache.ts # Read-through cache layer
+│ │ │ ├── secrets.ts # Secrets management
+│ │ │ ├── stateReset.ts # State reset utilities
+│ │ │ ├── migrationRunner.ts # Schema migration runner
+│ │ │ └── migrations/ # 16 SQL migration files
+│ │ ├── evals/ # Eval runner and scheduler
+│ │ ├── memory/ # Persistent conversational memory
+│ │ │ ├── extraction.ts # Memory extraction from conversations
+│ │ │ ├── injection.ts # Memory injection into context
+│ │ │ ├── retrieval.ts # Memory retrieval/search
+│ │ │ ├── store.ts # Memory persistence layer
+│ │ │ └── summarization.ts # Memory summarization
+│ │ ├── oauth/ # OAuth providers, services, and utilities
+│ │ │ ├── constants/ # Default OAuth credentials (overridable via env)
+│ │ │ ├── providers/ # Provider-specific OAuth configs
+│ │ │ ├── services/ # Provider-specific token exchange logic
+│ │ │ └── utils/ # PKCE, callback server, token helpers
+│ │ ├── plugins/ # Plugin system
+│ │ ├── skills/ # Extensible skill framework
+│ │ │ ├── registry.ts # Skill registration
+│ │ │ ├── executor.ts # Skill execution engine
+│ │ │ ├── sandbox.ts # Skill sandbox environment
+│ │ │ ├── builtin/ # Built-in skills
+│ │ │ ├── interception.ts # Skill request interception
+│ │ │ └── injection.ts # Skill context injection
+│ │ ├── usage/ # Usage tracking system
+│ │ │ ├── callLogs.ts # Call log persistence
+│ │ │ ├── costCalculator.ts # Cost calculation engine
+│ │ │ └── usageHistory.ts # Usage history queries
+│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers
+│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management
+│ │ ├── pricingSync.ts # LiteLLM pricing data sync
+│ │ ├── semanticCache.ts # Semantic caching layer
+│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler
+│ │ ├── webhookDispatcher.ts # Webhook event dispatcher
+│ │ └── localDb.ts # Unified re-export layer for all DB modules
+│ ├── middleware/ # Request middleware
+│ │ └── promptInjectionGuard.ts # Prompt injection detection
+│ ├── mitm/ # MITM proxy capability
+│ │ ├── cert/ # Certificate management
+│ │ ├── dns/ # DNS handling
+│ │ ├── targets/ # Target routing
+│ │ └── manager.ts # MITM proxy manager
+│ ├── shared/ # Shared utilities, components, and constants
+│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.)
+│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes
+│ │ ├── contracts/ # Shared API contracts
+│ │ ├── hooks/ # React hooks
+│ │ ├── middleware/ # Shared middleware utilities
+│ │ ├── schemas/ # Shared Zod schemas
+│ │ ├── services/ # Shared services
+│ │ ├── types/ # Shared TypeScript types
+│ │ ├── validation/ # Zod schemas (settings, providers, routes)
+│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID)
+│ ├── sse/ # SSE proxy pipeline
+│ │ ├── services/ # Auth resolution, format translation, response handling
+│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency
+│ ├── store/ # Zustand client-side stores (theme, providers, etc.)
+│ └── types/ # TypeScript type definitions
+├── open-sse/ # Standalone SSE server (npm workspace)
+│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video,
+│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models)
+│ ├── executors/ # Provider-specific request executors (14 executors)
+│ │ ├── base.ts # Base executor with shared logic
+│ │ ├── default.ts # Default OpenAI-compatible executor
+│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum)
+│ │ ├── codex.ts # OpenAI Codex CLI
+│ │ ├── antigravity.ts # Antigravity IDE
+│ │ ├── github.ts # GitHub Copilot
+│ │ ├── gemini-cli.ts # Gemini CLI
+│ │ ├── kiro.ts # Kiro AI
+│ │ ├── qoder.ts # Qoder AI
+│ │ ├── vertex.ts # Vertex AI (Service Account JSON)
+│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI
+│ │ ├── opencode.ts # OpenCode Zen/Go
+│ │ ├── pollinations.ts # Pollinations AI
+│ │ └── puter.ts # Puter AI
+│ ├── handlers/ # Request handlers per API type (11 handlers)
+│ │ ├── chatCore.ts # Main chat completions handler
+│ │ ├── responsesHandler.ts # OpenAI Responses API handler
+│ │ ├── embeddings.ts # Embedding generation
+│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.)
+│ │ ├── videoGeneration.ts # Video generation
+│ │ ├── musicGeneration.ts # Music generation
+│ │ ├── audioSpeech.ts # Text-to-speech
+│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI)
+│ │ ├── moderations.ts # Content moderation
+│ │ ├── rerank.ts # Reranking API
+│ │ └── search.ts # Web search API
+│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP)
+│ │ ├── server.ts # MCP server core (tool registration, scope enforcement)
+│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools)
+│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a)
+│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes)
+│ │ ├── audit.ts # Tool call audit logging
+│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat
+│ │ └── httpTransport.ts # HTTP transport handler
+│ ├── services/ # 36+ service modules
+│ │ ├── combo.ts # Core routing engine
+│ │ ├── usage.ts # Usage tracking
+│ │ ├── tokenRefresh.ts # OAuth token refresh
+│ │ ├── rateLimitManager.ts # Rate limit management
+│ │ ├── accountFallback.ts # Multi-account fallback
+│ │ ├── sessionManager.ts # Session management
+│ │ ├── wildcardRouter.ts # Wildcard model routing
+│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration)
+│ │ ├── intentClassifier.ts # Request intent classification
+│ │ ├── taskAwareRouter.ts # Task-aware routing
+│ │ ├── thinkingBudget.ts # Thinking budget management
+│ │ ├── contextManager.ts # Context window management
+│ │ ├── modelDeprecation.ts # Model deprecation handling
+│ │ ├── modelFamilyFallback.ts # Intra-family model fallback
+│ │ ├── emergencyFallback.ts # Emergency fallback
+│ │ ├── workflowFSM.ts # Workflow state machine
+│ │ ├── backgroundTaskDetector.ts # Background task detection
+│ │ ├── ipFilter.ts # IP-based access control
+│ │ ├── signatureCache.ts # CLI signature caching
+│ │ ├── volumeDetector.ts # Request volume detection
+│ │ ├── contextHandoff.ts # Context relay handoff generation and injection
+│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay
+│ │ └── ... # Additional services (14 more modules)
+│ ├── transformer/ # Responses API transformer
+│ │ └── responsesTransformer.ts
+│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek)
+│ │ ├── request/ # Request translators per provider
+│ │ ├── response/ # Response translators per provider
+│ │ ├── helpers/ # Translation helpers
+│ │ └── image/ # Image format translation
+│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.)
+├── electron/ # Electron desktop app (cross-platform)
+│ ├── main.js # Electron main process
+│ ├── preload.js # Preload script (IPC bridge)
+│ └── assets/ # App icons and assets
+├── tests/ # Test suites
+│ ├── unit/ # 122 unit test files
+│ ├── integration/ # Integration tests
+│ ├── e2e/ # Playwright E2E tests
+│ ├── security/ # Security tests
+│ ├── translator/ # Translator-specific tests
+│ └── load/ # Load tests
+├── docs/ # Documentation
+│ ├── i18n/ # 30-language translated docs
+│ ├── ARCHITECTURE.md # Full architecture documentation
+│ ├── API_REFERENCE.md # API reference
+│ ├── USER_GUIDE.md # User guide
+│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview
+│ ├── CLI-TOOLS.md # CLI tools integration guide
+│ ├── A2A-SERVER.md # A2A agent protocol documentation
+│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring)
+│ ├── MCP-SERVER.md # MCP server (25 tools)
+│ ├── TROUBLESHOOTING.md # Troubleshooting guide
+│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide
+│ ├── openapi.yaml # OpenAPI specification
+│ └── screenshots/ # Dashboard screenshots
+├── bin/ # CLI entry points (omniroute, reset-password)
+├── scripts/ # Build and utility scripts
+└── .env.example # Environment variable template
+```
+
+## Key Features (v3.5.5)
+
+### Core Proxy
+- **60+ AI providers** with automatic format translation
+- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible)
+- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay
+- **4-tier fallback**: Subscription → API Key → Cheap → Free
+- **Context Relay strategy**: Session handoff summaries on account rotation for continuity
+- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown
+- **Semantic caching** with cache hit/miss headers
+- **Idempotency** with configurable dedup window
+- **Circuit breaker** per provider with configurable thresholds
+- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback
+- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers
+- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement
+- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization
+- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills
+- **Prompt Injection Guard**: Middleware-level prompt injection detection
+- **MITM Proxy**: Certificate management, DNS handling, and target routing
+- **Cloudflare Tunnels**: Managed tunnel creation for remote access
+- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches)
+
+### Sécurité
+- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs)
+- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()`
+- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()`
+- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection
+- **CLI Fingerprint Matching** — Per-provider request signature matching
+- **Prompt injection guard** — Request middleware detection
+- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`)
+- **PII sanitizer** — Sensitive data scrubbing in logs
+
+### Dashboard Pages (23 sections)
+- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons
+- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies
+- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics
+- **Analytics** — Token consumption, cost, heatmaps, distributions
+- **Health** — Uptime, memory, latency percentiles, circuit breakers
+- **Logs** — Request, Proxy, Audit, Console (tabbed)
+- **Audit** — Audit trail and compliance logging
+- **Costs** — Cost tracking per provider/model
+- **Limits** — Rate limit monitoring
+- **Cache** — Semantic cache statistics and management
+- **CLI Tools** — One-click configuration for 10+ AI CLI tools
+- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration
+- **Playground** — Test any model with Monaco editor, streaming responses
+- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files)
+- **Search Tools** — Search provider configuration and testing
+- **Memory** — Memory system management and visualization
+- **Skills** — Skills framework management and execution
+- **Translator** — Format debugging: playground, chat tester, test bench, live monitor
+- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced
+- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed)
+- **Onboarding** — Setup wizard for new users
+- **Usage** — Usage history and analytics
+- **API Manager** — API key management with scoped permissions
+
+### Protocol Support
+- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations`
+- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens`
+- **OpenAI Responses** — `/v1/responses`
+- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}`
+- **Ollama** — `/v1/api/chat`, `/api/tags`
+- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily)
+- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP)
+- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills)
+- **ACP** — Agent Communication Protocol registry and manager
+
+### MCP Server (25 Tools)
+| Category | Tools |
+|-----------|-------|
+| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` |
+| Memory (3) | `memory_search`, `memory_add`, `memory_clear` |
+| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` |
+
+**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience`
+
+### Provider Categories
+
+**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI
+
+**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline
+
+**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan
+
+**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs
+
+### Internationalization
+- 30 languages for UI (all dashboard pages)
+- 30 translated documentation sets in docs/i18n/
+- Language switcher in documentation
+
+## Key Architectural Decisions
+
+1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints.
+
+2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently.
+
+3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation.
+
+4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity.
+
+5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client.
+
+6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes.
+
+7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`).
+
+8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages.
+
+9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations.
+
+10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`.
+
+11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation.
+
+12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline.
+
+13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast.
+
+## Main Flows
+
+### Proxy Request Flow
+1. Client sends OpenAI-format request to `/v1/chat/completions`
+2. API key validation
+3. Model resolution: direct model or combo lookup
+4. For combos: iterate through models with selected strategy
+5. Auth resolution: get credentials for the target provider
+6. Format translation: OpenAI → provider native format
+7. CLI fingerprint matching (if enabled for provider)
+8. Upstream request with circuit breaker and rate limiting
+9. Response translation: provider → OpenAI format
+10. omniModel tag sanitization (strip internal tags)
+11. SSE streaming back to client
+12. Memory extraction (if memory system enabled)
+13. Usage logging and cost calculation
+
+### OAuth Flow
+1. Dashboard initiates `/api/oauth/[provider]/authorize`
+2. User completes OAuth login in browser
+3. Callback hits `/api/oauth/[provider]/exchange`
+4. Tokens stored as a provider connection in SQLite
+5. Background job refreshes tokens before expiry
+
+## Important Notes for LLMs
+
+1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only).
+
+2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`).
+
+3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services.
+
+4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`.
+
+5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module.
+
+6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`).
+
+7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes.
+
+8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB.
+
+9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown.
+
+10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130).
+
+11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux).
+
+12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint.
+
+13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API.
+
+14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API.
+
+15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time.
+
+16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`.
+
+17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22.
+
+18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false.
+
+## Links
+
+- Repository: https://github.com/diegosouzapw/OmniRoute
+- Website: https://omniroute.online
+- npm: https://www.npmjs.com/package/omniroute
+- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute
+- Documentation: See `/docs/` directory
diff --git a/docs/i18n/he/README.md b/docs/i18n/he/README.md
index e272608e37..3c6c207268 100644
--- a/docs/i18n/he/README.md
+++ b/docs/i18n/he/README.md
@@ -4,6 +4,7 @@
---
+
### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback.
_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._
@@ -247,7 +248,7 @@ Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Eve
- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention
- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI
- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next
-- **Custom Combos** — Customizable fallback chains with 9 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random)
+- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**)
- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard
@@ -1308,7 +1309,17 @@ Then in `/dashboard/media` → **Transcription** tab: upload any audio or video
## 💡 Key Features
-OmniRoute v2.0 is built as an operational platform, not just a relay proxy.
+OmniRoute v3.5 is built as an operational platform, not just a relay proxy.
+
+### 🆕 New — v3.5.5 Highlights (Apr 2026)
+
+| Feature | What It Does |
+| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation |
+| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs |
+| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions |
+| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection |
+| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts |
### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
@@ -1360,7 +1371,8 @@ OmniRoute v2.0 is built as an operational platform, not just a relay proxy.
| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions |
| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection |
| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry |
-| 🎨 **Custom Combos** | 9 balancing strategies + fallback chain control |
+| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control |
+| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session |
| 🌐 **Wildcard Router** | `provider/*` dynamic routing |
| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits |
| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety |
@@ -2187,9 +2199,10 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux
| ---------------------------------------------- | --------------------------------------------------- |
| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment |
| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples |
-| [MCP Server](open-sse/mcp-server/README.md) | 16 MCP tools, IDE configs, Python/TS/Go clients |
+| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients |
| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt |
| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing |
+| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation |
| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions |
| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals |
| [Contributing](CONTRIBUTING.md) | Development setup and guidelines |
diff --git a/docs/i18n/he/docs/ARCHITECTURE.md b/docs/i18n/he/docs/ARCHITECTURE.md
index c774f5933d..101c8aca35 100644
--- a/docs/i18n/he/docs/ARCHITECTURE.md
+++ b/docs/i18n/he/docs/ARCHITECTURE.md
@@ -4,6 +4,8 @@
---
+
+
_Last updated: 2026-03-28_
## Executive Summary
@@ -36,6 +38,7 @@ Core capabilities:
- Anti-thundering herd protection with mutex locking
- Signature-based request deduplication cache
- Domain layer: model availability, cost rules, fallback policy, lockout policy
+- Context Relay: session handoff summaries for account rotation continuity
- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers)
- Policy engine for centralized request evaluation (lockout → budget → fallback)
- Request telemetry with p50/p95/p99 latency aggregation
@@ -222,6 +225,8 @@ Services (business logic):
- Wildcard model routing: `open-sse/services/wildcardRouter.ts`
- Rate limit management: `open-sse/services/rateLimitManager.ts`
- Circuit breaker: `open-sse/services/circuitBreaker.ts`
+- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy
+- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions
Domain layer modules:
@@ -802,7 +807,10 @@ Environment variables actively used by code:
5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency.
6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates).
7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`).
-8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies), AI (thinking budget, system prompt, prompt cache), Advanced (proxy).
+8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy).
+9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed.
+10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22.
+11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible.
## Operational Verification Checklist
diff --git a/docs/i18n/he/docs/FEATURES.md b/docs/i18n/he/docs/FEATURES.md
index 596e04ad05..2efa645cae 100644
--- a/docs/i18n/he/docs/FEATURES.md
+++ b/docs/i18n/he/docs/FEATURES.md
@@ -4,6 +4,8 @@
---
+
+
Visual guide to every section of the OmniRoute dashboard.
---
@@ -18,7 +20,7 @@ Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI)
## 🎨 Combos
-Create model routing combos with 6 strategies: priority, weighted, round-robin, random, least-used, and cost-optimized. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks.
+Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks.

@@ -68,7 +70,7 @@ Comprehensive settings panel with tabs:
- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls
- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info
- **Routing** — Model aliases, background task degradation
-- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring
+- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration
- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode

@@ -94,6 +96,30 @@ Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in a
---
+## 🔗 Context Relay _(v3.5.5+)_
+
+A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context.
+
+Configurable via combo-level or global settings:
+- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%)
+- **Max Messages For Summary** — How much recent history to condense
+- **Summary Model** — Optional override model for generating the handoff summary
+
+Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md).
+
+---
+
+## 🛡️ Proxy Hardening _(v3.5.5+)_
+
+Comprehensive proxy configuration enforcement across the entire request pipeline:
+
+- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments
+- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings
+- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22
+- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS
+
+---
+
## 🖼️ Media _(v2.0.3+)_
Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen.
diff --git a/docs/i18n/he/docs/TROUBLESHOOTING.md b/docs/i18n/he/docs/TROUBLESHOOTING.md
index d49dd89c9f..f1f130ce04 100644
--- a/docs/i18n/he/docs/TROUBLESHOOTING.md
+++ b/docs/i18n/he/docs/TROUBLESHOOTING.md
@@ -4,6 +4,8 @@
---
+
+
Common problems and solutions for OmniRoute.
---
@@ -17,6 +19,60 @@ Common problems and solutions for OmniRoute.
| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` |
| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` |
| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) |
+| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below |
+| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below |
+
+---
+
+## Node.js Compatibility
+
+
+
+### Login page crashes or shows "Module self-registration" error
+
+**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database.
+
+**Symptoms:**
+- Login page shows a blank screen or a server error
+- Console shows `Error: Module did not self-register` or similar native binding errors
+- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected
+
+**Fix:**
+
+1. Install Node.js 22 LTS (recommended):
+ ```bash
+ nvm install 22
+ nvm use 22
+ ```
+2. Verify your version: `node --version` should show `v22.x.x`
+3. Reinstall OmniRoute: `npm install -g omniroute`
+4. Restart: `omniroute`
+
+> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**.
+
+---
+
+## Proxy Issues
+
+
+
+### Provider validation shows "fetch failed"
+
+**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing.
+
+**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically.
+
+### Token health check fails with "fetch failed"
+
+**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection.
+
+**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+.
+
+### SOCKS5 proxy returns "invalid onRequestStart method"
+
+**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation.
+
+**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+.
---
diff --git a/docs/i18n/he/llm.txt b/docs/i18n/he/llm.txt
new file mode 100644
index 0000000000..1eec894ae5
--- /dev/null
+++ b/docs/i18n/he/llm.txt
@@ -0,0 +1,476 @@
+# OmniRoute (עברית)
+
+🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt)
+
+---
+
+
+> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app.
+
+## סקירה כללית
+
+OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free.
+
+**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost.
+
+**Current version:** 3.5.5
+
+## Tech Stack
+
+- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`)
+- **Framework:** Next.js 16 (App Router) with TypeScript 5.9
+- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations)
+- **State management:** Zustand (client), SQLite (server persistence)
+- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons
+- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth
+- **Schemas:** Zod v4 for all API / MCP input validation
+- **Background jobs:** Custom token health check scheduler, 24h model auto-sync
+- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses
+- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine
+- **i18n:** next-intl with 30 languages
+- **Desktop:** Electron (cross-platform: Windows, macOS, Linux)
+- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`)
+
+## Project Structure
+
+```
+/
+├── src/ # Main application source
+│ ├── app/ # Next.js App Router pages and API routes
+│ │ ├── (dashboard)/ # Dashboard UI pages
+│ │ │ └── dashboard/
+│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents)
+│ │ │ ├── analytics/ # Usage analytics and charts
+│ │ │ ├── api-manager/ # API key management
+│ │ │ ├── audit/ # Audit logs
+│ │ │ ├── auto-combo/ # Auto-combo engine dashboard
+│ │ │ ├── cache/ # Cache dashboard (semantic cache stats)
+│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.)
+│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates)
+│ │ │ ├── costs/ # Cost tracking per provider/model
+│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs
+│ │ │ ├── health/ # System health (uptime, circuit breakers, latency)
+│ │ │ ├── limits/ # Rate limits dashboard
+│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed)
+│ │ │ ├── media/ # Image/video/music generation + transcription
+│ │ │ ├── memory/ # Memory system dashboard
+│ │ │ ├── onboarding/ # Onboarding wizard
+│ │ │ ├── playground/ # Model playground (Monaco editor, streaming)
+│ │ │ ├── providers/ # Provider management (OAuth + API key + free)
+│ │ │ ├── search-tools/ # Search tools configuration
+│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced)
+│ │ │ ├── skills/ # Skills system dashboard
+│ │ │ ├── translator/ # Format translator + debug tools
+│ │ │ └── usage/ # Usage history
+│ │ ├── api/ # REST API endpoints (51 route directories)
+│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings,
+│ │ │ │ # images, audio, videos, music, moderations, rerank, search,
+│ │ │ │ # responses, messages, registered-keys, quotas, accounts)
+│ │ │ ├── v1beta/ # Gemini-compatible API
+│ │ │ ├── a2a/ # A2A agent management API
+│ │ │ ├── acp/ # ACP agent management API
+│ │ │ ├── oauth/ # OAuth flows per provider
+│ │ │ ├── providers/ # Provider CRUD and batch testing
+│ │ │ ├── models/ # Dashboard model listing and aliases
+│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains)
+│ │ │ ├── memory/ # Memory system API
+│ │ │ ├── skills/ # Skills system API
+│ │ │ ├── evals/ # Eval runner API
+│ │ │ ├── mcp/ # MCP HTTP transport API
+│ │ │ ├── search/ # Search provider API
+│ │ │ ├── webhooks/ # Webhook management
+│ │ │ ├── tunnels/ # Cloudflare tunnel management
+│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.)
+│ │ ├── landing/ # Landing page
+│ │ ├── login/ # Login page
+│ │ ├── forgot-password/ # Password recovery
+│ │ ├── status/ # Status page
+│ │ └── docs/ # In-app documentation
+│ ├── domain/ # Domain types and policy engine
+│ │ ├── policyEngine.ts # Central policy engine
+│ │ ├── comboResolver.ts # Combo resolution logic
+│ │ ├── costRules.ts # Cost calculation rules
+│ │ ├── degradation.ts # Graceful degradation
+│ │ ├── fallbackPolicy.ts # Fallback behavior
+│ │ ├── lockoutPolicy.ts # Account lockout logic
+│ │ ├── modelAvailability.ts # Model availability checks
+│ │ ├── providerExpiration.ts # Provider credential expiration
+│ │ ├── quotaCache.ts # Quota caching layer
+│ │ ├── configAudit.ts # Configuration auditing
+│ │ └── responses.ts # Domain response types
+│ ├── i18n/ # Internationalization
+│ │ └── messages/ # 30 language JSON files
+│ ├── lib/ # Core libraries
+│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server
+│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting)
+│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup
+│ │ │ └── streaming.ts # SSE streaming for A2A
+│ │ ├── acp/ # Agent Communication Protocol registry and manager
+│ │ ├── compliance/ # Compliance policy engine
+│ │ ├── db/ # SQLite database layer (21 modules + migrations)
+│ │ │ ├── core.ts # Database initialization, connection, schema
+│ │ │ ├── providers.ts # Provider connection CRUD
+│ │ │ ├── models.ts # Model catalog management
+│ │ │ ├── combos.ts # Combo configuration
+│ │ │ ├── apiKeys.ts # API key management
+│ │ │ ├── settings.ts # Settings persistence
+│ │ │ ├── backup.ts # Database backup/restore
+│ │ │ ├── proxies.ts # Proxy registry
+│ │ │ ├── prompts.ts # Prompt templates
+│ │ │ ├── webhooks.ts # Webhook subscriptions
+│ │ │ ├── detailedLogs.ts # Detailed request logging
+│ │ │ ├── domainState.ts # Domain state persistence
+│ │ │ ├── registeredKeys.ts # Registered API keys with quotas
+│ │ │ ├── quotaSnapshots.ts # Quota snapshot history
+│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings
+│ │ │ ├── cliToolState.ts # CLI tool state tracking
+│ │ │ ├── encryption.ts # Data encryption
+│ │ │ ├── readCache.ts # Read-through cache layer
+│ │ │ ├── secrets.ts # Secrets management
+│ │ │ ├── stateReset.ts # State reset utilities
+│ │ │ ├── migrationRunner.ts # Schema migration runner
+│ │ │ └── migrations/ # 16 SQL migration files
+│ │ ├── evals/ # Eval runner and scheduler
+│ │ ├── memory/ # Persistent conversational memory
+│ │ │ ├── extraction.ts # Memory extraction from conversations
+│ │ │ ├── injection.ts # Memory injection into context
+│ │ │ ├── retrieval.ts # Memory retrieval/search
+│ │ │ ├── store.ts # Memory persistence layer
+│ │ │ └── summarization.ts # Memory summarization
+│ │ ├── oauth/ # OAuth providers, services, and utilities
+│ │ │ ├── constants/ # Default OAuth credentials (overridable via env)
+│ │ │ ├── providers/ # Provider-specific OAuth configs
+│ │ │ ├── services/ # Provider-specific token exchange logic
+│ │ │ └── utils/ # PKCE, callback server, token helpers
+│ │ ├── plugins/ # Plugin system
+│ │ ├── skills/ # Extensible skill framework
+│ │ │ ├── registry.ts # Skill registration
+│ │ │ ├── executor.ts # Skill execution engine
+│ │ │ ├── sandbox.ts # Skill sandbox environment
+│ │ │ ├── builtin/ # Built-in skills
+│ │ │ ├── interception.ts # Skill request interception
+│ │ │ └── injection.ts # Skill context injection
+│ │ ├── usage/ # Usage tracking system
+│ │ │ ├── callLogs.ts # Call log persistence
+│ │ │ ├── costCalculator.ts # Cost calculation engine
+│ │ │ └── usageHistory.ts # Usage history queries
+│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers
+│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management
+│ │ ├── pricingSync.ts # LiteLLM pricing data sync
+│ │ ├── semanticCache.ts # Semantic caching layer
+│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler
+│ │ ├── webhookDispatcher.ts # Webhook event dispatcher
+│ │ └── localDb.ts # Unified re-export layer for all DB modules
+│ ├── middleware/ # Request middleware
+│ │ └── promptInjectionGuard.ts # Prompt injection detection
+│ ├── mitm/ # MITM proxy capability
+│ │ ├── cert/ # Certificate management
+│ │ ├── dns/ # DNS handling
+│ │ ├── targets/ # Target routing
+│ │ └── manager.ts # MITM proxy manager
+│ ├── shared/ # Shared utilities, components, and constants
+│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.)
+│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes
+│ │ ├── contracts/ # Shared API contracts
+│ │ ├── hooks/ # React hooks
+│ │ ├── middleware/ # Shared middleware utilities
+│ │ ├── schemas/ # Shared Zod schemas
+│ │ ├── services/ # Shared services
+│ │ ├── types/ # Shared TypeScript types
+│ │ ├── validation/ # Zod schemas (settings, providers, routes)
+│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID)
+│ ├── sse/ # SSE proxy pipeline
+│ │ ├── services/ # Auth resolution, format translation, response handling
+│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency
+│ ├── store/ # Zustand client-side stores (theme, providers, etc.)
+│ └── types/ # TypeScript type definitions
+├── open-sse/ # Standalone SSE server (npm workspace)
+│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video,
+│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models)
+│ ├── executors/ # Provider-specific request executors (14 executors)
+│ │ ├── base.ts # Base executor with shared logic
+│ │ ├── default.ts # Default OpenAI-compatible executor
+│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum)
+│ │ ├── codex.ts # OpenAI Codex CLI
+│ │ ├── antigravity.ts # Antigravity IDE
+│ │ ├── github.ts # GitHub Copilot
+│ │ ├── gemini-cli.ts # Gemini CLI
+│ │ ├── kiro.ts # Kiro AI
+│ │ ├── qoder.ts # Qoder AI
+│ │ ├── vertex.ts # Vertex AI (Service Account JSON)
+│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI
+│ │ ├── opencode.ts # OpenCode Zen/Go
+│ │ ├── pollinations.ts # Pollinations AI
+│ │ └── puter.ts # Puter AI
+│ ├── handlers/ # Request handlers per API type (11 handlers)
+│ │ ├── chatCore.ts # Main chat completions handler
+│ │ ├── responsesHandler.ts # OpenAI Responses API handler
+│ │ ├── embeddings.ts # Embedding generation
+│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.)
+│ │ ├── videoGeneration.ts # Video generation
+│ │ ├── musicGeneration.ts # Music generation
+│ │ ├── audioSpeech.ts # Text-to-speech
+│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI)
+│ │ ├── moderations.ts # Content moderation
+│ │ ├── rerank.ts # Reranking API
+│ │ └── search.ts # Web search API
+│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP)
+│ │ ├── server.ts # MCP server core (tool registration, scope enforcement)
+│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools)
+│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a)
+│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes)
+│ │ ├── audit.ts # Tool call audit logging
+│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat
+│ │ └── httpTransport.ts # HTTP transport handler
+│ ├── services/ # 36+ service modules
+│ │ ├── combo.ts # Core routing engine
+│ │ ├── usage.ts # Usage tracking
+│ │ ├── tokenRefresh.ts # OAuth token refresh
+│ │ ├── rateLimitManager.ts # Rate limit management
+│ │ ├── accountFallback.ts # Multi-account fallback
+│ │ ├── sessionManager.ts # Session management
+│ │ ├── wildcardRouter.ts # Wildcard model routing
+│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration)
+│ │ ├── intentClassifier.ts # Request intent classification
+│ │ ├── taskAwareRouter.ts # Task-aware routing
+│ │ ├── thinkingBudget.ts # Thinking budget management
+│ │ ├── contextManager.ts # Context window management
+│ │ ├── modelDeprecation.ts # Model deprecation handling
+│ │ ├── modelFamilyFallback.ts # Intra-family model fallback
+│ │ ├── emergencyFallback.ts # Emergency fallback
+│ │ ├── workflowFSM.ts # Workflow state machine
+│ │ ├── backgroundTaskDetector.ts # Background task detection
+│ │ ├── ipFilter.ts # IP-based access control
+│ │ ├── signatureCache.ts # CLI signature caching
+│ │ ├── volumeDetector.ts # Request volume detection
+│ │ ├── contextHandoff.ts # Context relay handoff generation and injection
+│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay
+│ │ └── ... # Additional services (14 more modules)
+│ ├── transformer/ # Responses API transformer
+│ │ └── responsesTransformer.ts
+│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek)
+│ │ ├── request/ # Request translators per provider
+│ │ ├── response/ # Response translators per provider
+│ │ ├── helpers/ # Translation helpers
+│ │ └── image/ # Image format translation
+│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.)
+├── electron/ # Electron desktop app (cross-platform)
+│ ├── main.js # Electron main process
+│ ├── preload.js # Preload script (IPC bridge)
+│ └── assets/ # App icons and assets
+├── tests/ # Test suites
+│ ├── unit/ # 122 unit test files
+│ ├── integration/ # Integration tests
+│ ├── e2e/ # Playwright E2E tests
+│ ├── security/ # Security tests
+│ ├── translator/ # Translator-specific tests
+│ └── load/ # Load tests
+├── docs/ # Documentation
+│ ├── i18n/ # 30-language translated docs
+│ ├── ARCHITECTURE.md # Full architecture documentation
+│ ├── API_REFERENCE.md # API reference
+│ ├── USER_GUIDE.md # User guide
+│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview
+│ ├── CLI-TOOLS.md # CLI tools integration guide
+│ ├── A2A-SERVER.md # A2A agent protocol documentation
+│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring)
+│ ├── MCP-SERVER.md # MCP server (25 tools)
+│ ├── TROUBLESHOOTING.md # Troubleshooting guide
+│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide
+│ ├── openapi.yaml # OpenAPI specification
+│ └── screenshots/ # Dashboard screenshots
+├── bin/ # CLI entry points (omniroute, reset-password)
+├── scripts/ # Build and utility scripts
+└── .env.example # Environment variable template
+```
+
+## Key Features (v3.5.5)
+
+### Core Proxy
+- **60+ AI providers** with automatic format translation
+- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible)
+- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay
+- **4-tier fallback**: Subscription → API Key → Cheap → Free
+- **Context Relay strategy**: Session handoff summaries on account rotation for continuity
+- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown
+- **Semantic caching** with cache hit/miss headers
+- **Idempotency** with configurable dedup window
+- **Circuit breaker** per provider with configurable thresholds
+- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback
+- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers
+- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement
+- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization
+- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills
+- **Prompt Injection Guard**: Middleware-level prompt injection detection
+- **MITM Proxy**: Certificate management, DNS handling, and target routing
+- **Cloudflare Tunnels**: Managed tunnel creation for remote access
+- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches)
+
+### אבטחה
+- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs)
+- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()`
+- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()`
+- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection
+- **CLI Fingerprint Matching** — Per-provider request signature matching
+- **Prompt injection guard** — Request middleware detection
+- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`)
+- **PII sanitizer** — Sensitive data scrubbing in logs
+
+### Dashboard Pages (23 sections)
+- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons
+- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies
+- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics
+- **Analytics** — Token consumption, cost, heatmaps, distributions
+- **Health** — Uptime, memory, latency percentiles, circuit breakers
+- **Logs** — Request, Proxy, Audit, Console (tabbed)
+- **Audit** — Audit trail and compliance logging
+- **Costs** — Cost tracking per provider/model
+- **Limits** — Rate limit monitoring
+- **Cache** — Semantic cache statistics and management
+- **CLI Tools** — One-click configuration for 10+ AI CLI tools
+- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration
+- **Playground** — Test any model with Monaco editor, streaming responses
+- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files)
+- **Search Tools** — Search provider configuration and testing
+- **Memory** — Memory system management and visualization
+- **Skills** — Skills framework management and execution
+- **Translator** — Format debugging: playground, chat tester, test bench, live monitor
+- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced
+- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed)
+- **Onboarding** — Setup wizard for new users
+- **Usage** — Usage history and analytics
+- **API Manager** — API key management with scoped permissions
+
+### Protocol Support
+- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations`
+- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens`
+- **OpenAI Responses** — `/v1/responses`
+- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}`
+- **Ollama** — `/v1/api/chat`, `/api/tags`
+- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily)
+- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP)
+- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills)
+- **ACP** — Agent Communication Protocol registry and manager
+
+### MCP Server (25 Tools)
+| Category | Tools |
+|-----------|-------|
+| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` |
+| Memory (3) | `memory_search`, `memory_add`, `memory_clear` |
+| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` |
+
+**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience`
+
+### Provider Categories
+
+**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI
+
+**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline
+
+**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan
+
+**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs
+
+### Internationalization
+- 30 languages for UI (all dashboard pages)
+- 30 translated documentation sets in docs/i18n/
+- Language switcher in documentation
+
+## Key Architectural Decisions
+
+1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints.
+
+2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently.
+
+3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation.
+
+4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity.
+
+5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client.
+
+6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes.
+
+7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`).
+
+8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages.
+
+9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations.
+
+10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`.
+
+11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation.
+
+12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline.
+
+13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast.
+
+## Main Flows
+
+### Proxy Request Flow
+1. Client sends OpenAI-format request to `/v1/chat/completions`
+2. API key validation
+3. Model resolution: direct model or combo lookup
+4. For combos: iterate through models with selected strategy
+5. Auth resolution: get credentials for the target provider
+6. Format translation: OpenAI → provider native format
+7. CLI fingerprint matching (if enabled for provider)
+8. Upstream request with circuit breaker and rate limiting
+9. Response translation: provider → OpenAI format
+10. omniModel tag sanitization (strip internal tags)
+11. SSE streaming back to client
+12. Memory extraction (if memory system enabled)
+13. Usage logging and cost calculation
+
+### OAuth Flow
+1. Dashboard initiates `/api/oauth/[provider]/authorize`
+2. User completes OAuth login in browser
+3. Callback hits `/api/oauth/[provider]/exchange`
+4. Tokens stored as a provider connection in SQLite
+5. Background job refreshes tokens before expiry
+
+## Important Notes for LLMs
+
+1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only).
+
+2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`).
+
+3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services.
+
+4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`.
+
+5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module.
+
+6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`).
+
+7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes.
+
+8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB.
+
+9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown.
+
+10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130).
+
+11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux).
+
+12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint.
+
+13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API.
+
+14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API.
+
+15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time.
+
+16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`.
+
+17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22.
+
+18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false.
+
+## Links
+
+- Repository: https://github.com/diegosouzapw/OmniRoute
+- Website: https://omniroute.online
+- npm: https://www.npmjs.com/package/omniroute
+- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute
+- Documentation: See `/docs/` directory
diff --git a/docs/i18n/hi/README.md b/docs/i18n/hi/README.md
index 36549e45b4..bd6ee203d5 100644
--- a/docs/i18n/hi/README.md
+++ b/docs/i18n/hi/README.md
@@ -4,11 +4,14 @@
---
+
### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback.
-_आपका सार्वभौमिक एपीआई प्रॉक्सी - एक समापन बिंदु, 60+ प्रदाता, शून्य डाउनटाइम। अब**एमसीपी सर्वर (25 टूल्स)**,**ए2ए प्रोटोकॉल**,**मेमोरी/स्किल सिस्टम**और**इलेक्ट्रॉन डेस्कटॉप ऐप**के साथ।_
+_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._
-**चैट समापन • एंबेडिंग • छवि निर्माण • वीडियो • संगीत • ऑडियो • पुनःरैंकिंग •**वेब खोज**• एमसीपी सर्वर • ए2ए प्रोटोकॉल • 100% टाइपस्क्रिप्ट**---
+**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript**
+
+---
@@ -39,9 +42,13 @@ _आपका सार्वभौमिक एपीआई प्रॉक्
[](https://omniroute.online)
[](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-[🌐 वेबसाइट](https://omniroute.online) • [🚀 त्वरित प्रारंभ](#-त्वरित-प्रारंभ) • [💡 विशेषताएं](#-कुंजी-विशेषताएं) • [📖 दस्तावेज़](#-दस्तावेज़ीकरण) • [💰 मूल्य निर्धारण](#-मूल्य निर्धारण-एक नजर में) • [💬 व्हाट्सएप](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-🌐**इसमें उपलब्ध:**🇺🇸 [अंग्रेजी](README.md) | 🇧🇷 [पुर्तगाली (ब्राजील)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [फ़्रांसीसी](docs/i18n/fr/README.md) | 🇮🇹 [इतालवी](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [जर्मन](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [डांस्क](docs/i18n/da/README.md) | 🇫🇮 [सुओमी](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [मग्यार](docs/i18n/hu/README.md) | 🇮🇩 [बहासा इंडोनेशिया](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [बहासा मेलायु](docs/i18n/ms/README.md) | 🇳🇱 [नीदरलैंड्स](docs/i18n/nl/README.md) | 🇳🇴 [नॉर्स्क](docs/i18n/no/README.md) | 🇵🇹 [पुर्तगाली (पुर्तगाल)](docs/i18n/pt/README.md) | 🇷🇴 [रोमानिया](docs/i18n/ro/README.md) | 🇵🇱 [पोल्स्की](docs/i18n/pl/README.md) | 🇸🇰[स्लोवेनसीना](docs/i18n/sk/README.md) | 🇸🇪 [स्वेन्स्का](docs/i18n/sv/README.md) | 🇵🇭 [फ़िलिपिनो](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)---
+
+
+🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)
+
+---
## 🖼️ Main Dashboard
@@ -54,552 +61,628 @@ _आपका सार्वभौमिक एपीआई प्रॉक्
## 📸 Dashboard Preview
-<सारांश>डैशबोर्ड स्क्रीनशॉट देखने के लिए क्लिक करेंसारांश>
+Click to see dashboard screenshots
-| पेज | स्क्रीनशॉट |
-| ---------------- | ------------------------------------------------------------ | ---------- |
-| **प्रदाता** |  |
-| **कॉम्बोस** |  |
-| **एनालिटिक्स** |  |
-| **स्वास्थ्य** |  |
-| **अनुवादक** |  |
-| **सेटिंग्स** |  |
-| **सीएलआई उपकरण** |  |
-| **उपयोग लॉग** |  |
-| **अंतबिंदु** |  | |
+| Page | Screenshot |
+| -------------- | ------------------------------------------------- |
+| **Providers** |  |
+| **Combos** |  |
+| **Analytics** |  |
+| **Health** |  |
+| **Translator** |  |
+| **Settings** |  |
+| **CLI Tools** |  |
+| **Usage Logs** |  |
+| **Endpoints** |  |
+
+
---
### 🤖 Free AI Provider for your favorite coding agents
-_OmniRoute के माध्यम से किसी भी AI-संचालित IDE या CLI टूल को कनेक्ट करें - असीमित कोडिंग के लिए निःशुल्क API गेटवे।_
+_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._
-<तालिका>
-
-
-
-
-ओपनक्लॉ
-
-⭐ 205K
- |
-
-
-
-नैनोबॉट
-
-⭐ 20.9K
- |
-
-
-
-पिकोक्लॉ
-
-⭐ 14.6K
- |
-
-
-
-जीरोक्लॉ
-
-⭐ 9.9K
- |
-
-
-
-आयरनक्लॉ
-
-⭐ 2.1K
- |
-
-
-
-
-
-ओपनकोड
-
-⭐ 106K
- |
-
-
-
-कोडेक्स सीएलआई
-
-⭐ 60.8K
- |
-
-
-
-क्लाउड कोड
-
-⭐ 67.3K
- |
-
-
-
-मिथुन सीएलआई
-
-⭐ 94.7K
- |
-
-
-
-किलो कोड
-
-⭐ 15.5K
- |
-
-तालिका>
+
-📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota---
+📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota
+
+---
## 🤔 Why OmniRoute?
-**पैसा बर्बाद करना और सीमा पार करना बंद करें:**
+**Stop wasting money and hitting limits:**
--
सदस्यता कोटा हर महीने अप्रयुक्त रूप से समाप्त हो जाता है
--
दर सीमा आपको कोडिंग के बीच में रोक देती है
--
महंगे एपीआई ($20-50/माह प्रति प्रदाता)
--
प्रदाताओं के बीच मैन्युअल स्विचिंग
+-
Subscription quota expires unused every month
+-
Rate limits stop you mid-coding
+-
Expensive APIs ($20-50/month per provider)
+-
Manual switching between providers
-**OmniRoute इसका समाधान करता है:**
+**OmniRoute solves this:**
-- ✅**सब्सक्रिप्शन अधिकतम करें**- कोटा ट्रैक करें, रीसेट से पहले हर बिट का उपयोग करें
-- ✅**ऑटो फ़ॉलबैक**- सदस्यता → एपीआई कुंजी → सस्ता → निःशुल्क, शून्य डाउनटाइम
-- ✅**मल्टी-अकाउंट**- प्रति प्रदाता खातों के बीच राउंड-रॉबिन
-- ✅**यूनिवर्सल**- क्लाउड कोड, कोडेक्स, जेमिनी सीएलआई, कर्सर, क्लाइन, ओपनक्लॉ, किसी भी सीएलआई टूल के साथ काम करता है---
+- ✅ **Maximize subscriptions** - Track quota, use every bit before reset
+- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime
+- ✅ **Multi-account** - Round-robin between accounts per provider
+- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool
+
+---
## 📧 Support
-> 💬**हमारे समुदाय में शामिल हों!**[व्हाट्सएप ग्रुप](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) - सहायता प्राप्त करें, टिप्स साझा करें और अपडेट रहें।
+> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated.
--**वेबसाइट**: [omniroute.online](https://omniroute.online) -**गिटहब**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**मुद्दे**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**व्हाट्सएप**: [सामुदायिक समूह](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**योगदान**: [CONTRIBUTING.md](CONTRIBUTING.md) देखें, एक पीआर खोलें, या एक `अच्छा पहला अंक` चुनें -**मूल परियोजना**: [डेकोलुआ द्वारा 9राउटर](https://github.com/decolua/9router)### 🐛 Reporting a Bug?
+- **Website**: [omniroute.online](https://omniroute.online)
+- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute)
+- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue`
+- **Original Project**: [9router by decolua](https://github.com/decolua/9router)
-कोई समस्या खोलते समय, कृपया सिस्टम-जानकारी कमांड चलाएँ और जेनरेट की गई फ़ाइल संलग्न करें:```bash
+### 🐛 Reporting a Bug?
+
+When opening an issue, please run the system-info command and attach the generated file:
+
+```bash
npm run system-info
-
```
-यह आपके Node.js संस्करण, ओमनीरूट संस्करण, ओएस विवरण, स्थापित सीएलआई उपकरण (क्यूडर, जेमिनी, क्लाउड, कोडेक्स, एंटीग्रेविटी, ड्रॉइड, आदि), डॉकर/पीएम2 स्थिति और सिस्टम पैकेज के साथ एक `system-info.txt` उत्पन्न करता है - वह सब कुछ जो हमें आपकी समस्या को शीघ्रता से पुन: उत्पन्न करने के लिए चाहिए। फ़ाइल को सीधे अपने GitHub मुद्दे से संलग्न करें।---
+This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue.
+
+---
## 🔄 How It Works
```
-
┌─────────────┐
-│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
-│ Tool │
+│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
+│ Tool │
└──────┬──────┘
-│ http://localhost:20128/v1
-↓
+ │ http://localhost:20128/v1
+ ↓
┌─────────────────────────────────────────┐
-│ OmniRoute (Smart Router) │
-│ • Format translation (OpenAI ↔ Claude) │
-│ • Quota tracking + Embeddings + Images │
-│ • Auto token refresh │
+│ OmniRoute (Smart Router) │
+│ • Format translation (OpenAI ↔ Claude) │
+│ • Quota tracking + Embeddings + Images │
+│ • Auto token refresh │
└──────┬──────────────────────────────────┘
-│
-├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
-│ ↓ quota exhausted
-├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
-│ ↓ budget limit
-├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
-│ ↓ budget limit
-└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
+ │
+ ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
+ │ ↓ quota exhausted
+ ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
+ │ ↓ budget limit
+ ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
+ │ ↓ budget limit
+ └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
Result: Never stop coding, minimal cost
-
-````
+```
---
## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases
->**एआई टूल का उपयोग करने वाला प्रत्येक डेवलपर प्रतिदिन इन समस्याओं का सामना करता है।**ओम्नीरूट को उन सभी को हल करने के लिए बनाया गया था - लागत वृद्धि से लेकर क्षेत्रीय ब्लॉक तक, टूटे हुए ओएथ प्रवाह से लेकर प्रोटोकॉल संचालन और एंटरप्राइज़ अवलोकन तक।
+> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability.
-<विवरण>
-<सारांश>💸 1. "मैं एक महंगी सदस्यता के लिए भुगतान करता हूं लेकिन फिर भी सीमा से बाधित होता हूं"सारांश>
+
+💸 1. "I pay for an expensive subscription but still get interrupted by limits"
-डेवलपर्स क्लाउड प्रो, कोडेक्स प्रो, या गिटहब कोपायलट के लिए $20-200/माह का भुगतान करते हैं। यहां तक कि भुगतान करने पर भी, कोटा की एक सीमा होती है - 5 घंटे का उपयोग, साप्ताहिक सीमा, या प्रति मिनट की दर सीमा। मध्य-कोडिंग सत्र में, प्रदाता प्रत्युत्तर देना बंद कर देता है और डेवलपर प्रवाह और उत्पादकता खो देता है।
+Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity.
-**ओम्नीरूट इसे कैसे हल करता है:**
+**How OmniRoute solves it:**
--**स्मार्ट 4-टियर फ़ॉलबैक**- यदि सदस्यता कोटा समाप्त हो जाता है, तो स्वचालित रूप से एपीआई कुंजी पर रीडायरेक्ट हो जाता है → सस्ता → शून्य मैन्युअल हस्तक्षेप के साथ मुफ़्त
--**प्रदाता ट्रैकिंग सीमाएं**- कैश्ड कोटा स्नैपशॉट सर्वर-साइड शेड्यूल पर रीफ्रेश होता है (डिफ़ॉल्ट `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) यूआई में मैन्युअल रीफ्रेश के साथ उपलब्ध है
--**मल्टी-अकाउंट सपोर्ट**- ऑटो राउंड-रॉबिन के साथ प्रति प्रदाता एकाधिक खाते - जब एक खत्म हो जाता है, तो अगले पर स्विच हो जाता है
--**कस्टम कॉम्बो**- 9 संतुलन रणनीतियों (प्राथमिकता, भारित, भरण-प्रथम, राउंड-रॉबिन, पी2सी, यादृच्छिक, कम से कम उपयोग, लागत-अनुकूलित, सख्त-यादृच्छिक) के साथ अनुकूलन योग्य फ़ॉलबैक चेन
--**कोडेक्स बिजनेस कोटा**- बिजनेस/टीम कार्यक्षेत्र कोटा की निगरानी सीधे डैशबोर्ड में
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention
+- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI
+- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next
+- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**)
+- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard
-<विवरण>
-<सारांश>🔌 2. "मुझे कई प्रदाताओं का उपयोग करने की आवश्यकता है लेकिन प्रत्येक के पास एक अलग एपीआई है"सारांश>
+
-ओपनएआई एक प्रारूप का उपयोग करता है, क्लाउड (एंथ्रोपिक) दूसरे का उपयोग करता है, जेमिनी एक और का उपयोग करता है। यदि कोई डेवलपर विभिन्न प्रदाताओं के मॉडल का परीक्षण करना चाहता है या उनके बीच फ़ॉलबैक करना चाहता है, तो उन्हें एसडीके को फिर से कॉन्फ़िगर करना होगा, एंडपॉइंट बदलना होगा, असंगत प्रारूपों से निपटना होगा। कस्टम प्रदाताओं (फ्रेंडएलआई, एनआईएम) के पास गैर-मानक मॉडल एंडपॉइंट हैं।
+
+🔌 2. "I need to use multiple providers but each has a different API"
-**ओम्नीरूट इसे कैसे हल करता है:**
+OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints.
--**एकीकृत समापन बिंदु**- एक एकल `http://localhost:20128/v1` सभी 60+ प्रदाताओं के लिए प्रॉक्सी के रूप में कार्य करता है
--**प्रारूप अनुवाद**- स्वचालित और पारदर्शी: ओपनएआई ↔ क्लाउड ↔ जेमिनी ↔ प्रतिक्रिया एपीआई
--**प्रतिक्रिया स्वच्छता**- गैर-मानक फ़ील्ड (`x_groq`, `usage_breakdown`, `service_tier`) को हटा दें जो OpenAI SDK v1.83+ को तोड़ता है
--**भूमिका सामान्यीकरण**- गैर-ओपनएआई प्रदाताओं के लिए `डेवलपर` → `सिस्टम` को रूपांतरित करता है; GLM/ERNIE के लिए `सिस्टम` → `उपयोगकर्ता`
--**थिंक टैग एक्सट्रैक्शन**- डीपसीक आर1 जैसे मॉडलों से `<थिंक>` ब्लॉक को मानकीकृत `रीज़निंग_कंटेंट` में निकाला जाता है
--**मिथुन के लिए संरचित आउटपुट**- `json_schema` → `responseMimeType`/`responseSchema` स्वचालित रूपांतरण
--**`स्ट्रीम` डिफ़ॉल्ट रूप से `झूठा`**होता है - ओपनएआई स्पेक के साथ संरेखित होता है, पायथन/रस्ट/गो एसडीके में अप्रत्याशित एसएसई से बचता है
+**How OmniRoute solves it:**
-<विवरण>
-<सारांश>🌐 3. "मेरा AI प्रदाता मेरे क्षेत्र/देश को ब्लॉक कर देता है"सारांश>
+- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers
+- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
+- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+
+- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE
+- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content`
+- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion
+- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs
-OpenAI/Codex जैसे प्रदाता कुछ भौगोलिक क्षेत्रों से पहुंच को रोकते हैं। उपयोगकर्ताओं को OAuth और API कनेक्शन के दौरान `unsupported_country_region_territory` जैसी त्रुटियां मिलती हैं। यह विकासशील देशों के डेवलपर्स के लिए विशेष रूप से निराशाजनक है।
+
-**ओम्नीरूट इसे कैसे हल करता है:**
+
+🌐 3. "My AI provider blocks my region/country"
--**3-स्तरीय प्रॉक्सी कॉन्फ़िगरेशन**- 3 स्तरों पर कॉन्फ़िगर करने योग्य प्रॉक्सी: वैश्विक (सभी ट्रैफ़िक), प्रति-प्रदाता (केवल एक प्रदाता), और प्रति-कनेक्शन/कुंजी
--**रंग-कोडित प्रॉक्सी बैज**- दृश्य संकेतक: 🟢 वैश्विक प्रॉक्सी, 🟡 प्रदाता प्रॉक्सी, 🔵 कनेक्शन प्रॉक्सी, हमेशा आईपी दिखाता है
--**प्रॉक्सी के माध्यम से OAuth टोकन एक्सचेंज**- OAuth प्रवाह भी प्रॉक्सी के माध्यम से जाता है, `unsupported_country_region_territory` को हल करता है
--**प्रॉक्सी के माध्यम से कनेक्शन परीक्षण**- कनेक्शन परीक्षण कॉन्फ़िगर प्रॉक्सी का उपयोग करते हैं (अब कोई प्रत्यक्ष बाईपास नहीं)
--**SOCKS5 समर्थन**- आउटबाउंड रूटिंग के लिए पूर्ण SOCKS5 प्रॉक्सी समर्थन
--**टीएलएस फ़िंगरप्रिंट स्पूफिंग**- बॉट डिटेक्शन को बायपास करने के लिए `wreq-js` के माध्यम से ब्राउज़र जैसा टीएलएस फ़िंगरप्रिंट
--**🔏 सीएलआई फ़िंगरप्रिंट मिलान**- मूल सीएलआई बाइनरी हस्ताक्षरों से मिलान करने के लिए हेडर और बॉडी फ़ील्ड को पुन: व्यवस्थित करता है, जिससे खाता फ़्लैगिंग जोखिम काफी कम हो जाता है। प्रॉक्सी आईपी संरक्षित है - आपको एक साथ स्टील्थ**और**आईपी मास्किंग दोनों मिलते हैं
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries.
-<विवरण>
-<सारांश>🆓 4. "मैं कोडिंग के लिए AI का उपयोग करना चाहता हूं लेकिन मेरे पास पैसे नहीं हैं"सारांश>
+**How OmniRoute solves it:**
-हर कोई AI सदस्यता के लिए $20-200/माह का भुगतान नहीं कर सकता। छात्रों, उभरते देशों के डेवलपर्स, शौकीनों और फ्रीलांसरों को शून्य लागत पर गुणवत्ता वाले मॉडल तक पहुंच की आवश्यकता है।
+- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key
+- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP
+- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory`
+- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass)
+- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection
+- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously
-**ओम्नीरूट इसे कैसे हल करता है:**
+
--**Free Tier Providers Built-in**— Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, विज़न-मॉडल), किरो (क्लाउड + एडब्ल्यूएस बिल्डर आईडी मुफ़्त), जेमिनी सीएलआई (180K टोकन/माह मुफ़्त)
--**ओलामा क्लाउड**- निःशुल्क "लाइट उपयोग" स्तर के साथ `api.ollama.com` पर क्लाउड-होस्टेड ओलामा मॉडल; `ollamacloud/` उपसर्ग का उपयोग करें
--**केवल-नि:शुल्क कॉम्बो**- चेन `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/माह शून्य डाउनटाइम के साथ
--**एनवीडिया एनआईएम फ्री एक्सेस**- ~40 आरपीएम डेव-बिल्ड.एनवीडिया.कॉम पर 70+ मॉडलों तक हमेशा के लिए मुफ्त एक्सेस (क्रेडिट से शुद्ध दर सीमा तक संक्रमण)
--**लागत अनुकूलित रणनीति**- रूटिंग रणनीति जो स्वचालित रूप से सबसे सस्ते उपलब्ध प्रदाता को चुनती है
+
+🆓 4. "I want to use AI for coding but I have no money"
-<विवरण>
-<सारांश>🔒 5. "मुझे अपने AI गेटवे को अनधिकृत पहुंच से बचाने की आवश्यकता है"सारांश>
+Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost.
-नेटवर्क (LAN, VPS, Docker) में AI गेटवे को उजागर करते समय, पते वाला कोई भी व्यक्ति डेवलपर के टोकन/कोटा का उपभोग कर सकता है। सुरक्षा के बिना, एपीआई दुरुपयोग, त्वरित इंजेक्शन और दुरुपयोग के प्रति संवेदनशील हैं।
+**How OmniRoute solves it:**
-**ओम्नीरूट इसे कैसे हल करता है:**
+- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free)
+- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix
+- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime
+- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits)
+- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider
--**एपीआई कुंजी प्रबंधन**- एक समर्पित `/डैशबोर्ड/एपीआई-मैनेजर` पेज के साथ प्रति प्रदाता जेनरेशन, रोटेशन और स्कोपिंग
--**मॉडल-स्तरीय अनुमतियाँ**- सभी को अनुमति दें/प्रतिबंधित टॉगल के साथ एपीआई कुंजियों को विशिष्ट मॉडल (`ओपनाई/*`, वाइल्डकार्ड पैटर्न) तक सीमित करें
--**एपीआई एंडपॉइंट सुरक्षा**- `/v1/मॉडल` के लिए एक कुंजी की आवश्यकता है और लिस्टिंग से विशिष्ट प्रदाताओं को ब्लॉक करें
--**ऑथ गार्ड + सीएसआरएफ सुरक्षा**- सभी डैशबोर्ड रूट `withAuth` मिडलवेयर + सीएसआरएफ टोकन से सुरक्षित हैं
--**रेट लिमिटर**- कॉन्फ़िगर करने योग्य विंडो के साथ प्रति-आईपी दर सीमित करना
--**आईपी फ़िल्टरिंग**- अभिगम नियंत्रण के लिए अनुमति सूची/अवरुद्ध सूची
--**प्रॉम्प्ट इंजेक्शन गार्ड**- दुर्भावनापूर्ण प्रॉम्प्ट पैटर्न के विरुद्ध स्वच्छता
--**एईएस-256-जीसीएम एन्क्रिप्शन**- क्रेडेंशियल आराम से एन्क्रिप्ट किए गए
+
-<विवरण>
-<सारांश>🛑 6. "मेरा प्रदाता बंद हो गया और मैंने अपना कोडिंग प्रवाह खो दिया"सारांश>
+
+🔒 5. "I need to protect my AI gateway from unauthorized access"
-एआई प्रदाता अस्थिर हो सकते हैं, 5xx त्रुटियाँ लौटा सकते हैं, या अस्थायी दर सीमा तक पहुँच सकते हैं। यदि कोई डेवलपर किसी एकल प्रदाता पर निर्भर करता है, तो वे बाधित हो जाते हैं। सर्किट ब्रेकर के बिना, बार-बार पुनः प्रयास करने से एप्लिकेशन क्रैश हो सकता है।
+When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse.
-**ओम्नीरूट इसे कैसे हल करता है:**
+**How OmniRoute solves it:**
--**प्रति-मॉडल सर्किट ब्रेकर**- कॉन्फ़िगर करने योग्य थ्रेसहोल्ड और कूलडाउन (बंद/खुला/आधा-खुला) के साथ ऑटो-खुला/बंद, कैस्केडिंग ब्लॉक से बचने के लिए प्रति-मॉडल स्कोप्ड
--**एक्सपोनेंशियल बैकऑफ़**- प्रगतिशील पुनः प्रयास में देरी
--**एंटी-थंडरिंग हर्ड**- म्यूटेक्स + समवर्ती रिट्री तूफानों के खिलाफ सेमाफोर सुरक्षा
--**कॉम्बो फ़ॉलबैक चेन**- यदि प्राथमिक प्रदाता विफल हो जाता है, तो बिना किसी हस्तक्षेप के स्वचालित रूप से चेन से गिर जाता है
--**कॉम्बो सर्किट ब्रेकर**- कॉम्बो श्रृंखला के भीतर विफल प्रदाताओं को स्वचालित रूप से अक्षम करता है
--**स्वास्थ्य डैशबोर्ड**- अपटाइम मॉनिटरिंग, सर्किट ब्रेकर स्थिति, लॉकआउट, कैश आँकड़े, p50/p95/p99 विलंबता
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page
+- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle
+- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing
+- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens
+- **Rate Limiter** — Per-IP rate limiting with configurable windows
+- **IP Filtering** — Allowlist/blocklist for access control
+- **Prompt Injection Guard** — Sanitization against malicious prompt patterns
+- **AES-256-GCM Encryption** — Credentials encrypted at rest
-<विवरण>
-<सारांश>🔧 7. "प्रत्येक AI उपकरण को कॉन्फ़िगर करना कठिन और दोहराव वाला है"सारांश>
+
-डेवलपर्स कर्सर, क्लाउड कोड, कोडेक्स सीएलआई, ओपनक्लाव, जेमिनी सीएलआई, किलो कोड का उपयोग करते हैं... प्रत्येक टूल को एक अलग कॉन्फ़िगरेशन (एपीआई एंडपॉइंट, कुंजी, मॉडल) की आवश्यकता होती है। प्रदाताओं या मॉडलों को स्विच करते समय पुन: कॉन्फ़िगर करना समय की बर्बादी है।
+
+🛑 6. "My provider went down and I lost my coding flow"
-**ओम्नीरूट इसे कैसे हल करता है:**
+AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application.
--**सीएलआई टूल्स डैशबोर्ड**- क्लाउड कोड, कोडेक्स सीएलआई, ओपनक्लाव, किलो कोड, एंटीग्रेविटी, क्लाइन के लिए एक-क्लिक सेटअप वाला समर्पित पेज
--**गिटहब कोपायलट कॉन्फिग जेनरेटर**- बल्क मॉडल चयन के साथ वीएस कोड के लिए `चैटलैंग्वेजमॉडल.जेसन` जेनरेट करता है
--**ऑनबोर्डिंग विज़ार्ड**- पहली बार उपयोगकर्ताओं के लिए निर्देशित 4-चरणीय सेटअप
--**एक समापन बिंदु, सभी मॉडल**- `http://localhost:20128/v1` को एक बार कॉन्फ़िगर करें, 60+ प्रदाताओं तक पहुंचें
+**How OmniRoute solves it:**
-<विवरण>
-<सारांश>🔑 8. "एकाधिक प्रदाताओं से OAuth टोकन प्रबंधित करना नरक है"सारांश>
+- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks
+- **Exponential Backoff** — Progressive retry delays
+- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms
+- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention
+- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain
+- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency
-क्लाउड कोड, कोडेक्स, जेमिनी सीएलआई, कोपायलट - सभी समाप्त होने वाले टोकन के साथ OAuth 2.0 का उपयोग करते हैं। डेवलपर्स को लगातार पुन: प्रमाणित करने, `client_secret is missing`, `redirect_uri_mismatch` और दूरस्थ सर्वर पर विफलताओं से निपटने की आवश्यकता होती है। LAN/VPS पर OAuth विशेष रूप से समस्याग्रस्त है।
+
-**ओम्नीरूट इसे कैसे हल करता है:**
+
+🔧 7. "Configuring each AI tool is tedious and repetitive"
--**ऑटो टोकन रिफ्रेश**- OAuth टोकन समाप्ति से पहले पृष्ठभूमि में रिफ्रेश होते हैं
--**OAuth 2.0 (PKCE) बिल्ट-इन**- क्लाउड कोड, कोडेक्स, जेमिनी सीएलआई, कोपायलट, किरो, क्वेन, कोडर के लिए स्वचालित प्रवाह
--**मल्टी-अकाउंट OAuth**- JWT/ID टोकन निष्कर्षण के माध्यम से प्रति प्रदाता एकाधिक खाते
--**OAuth LAN/रिमोट फिक्स**- `redirect_uri` के लिए निजी आईपी पहचान + रिमोट सर्वर के लिए मैनुअल यूआरएल मोड
--**Nginx के पीछे OAuth**- रिवर्स प्रॉक्सी संगतता के लिए `window.location.origin` का उपयोग करता है
--**दूरस्थ OAuth मार्गदर्शिका**— VPS/Docker पर Google क्लाउड क्रेडेंशियल के लिए चरण-दर-चरण मार्गदर्शिका
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time.
-<विवरण>
-<सारांश>📊 9. "मुझे नहीं पता कि मैं कितना और कहां खर्च कर रहा हूं"सारांश>
+**How OmniRoute solves it:**
-डेवलपर्स कई भुगतान प्रदाताओं का उपयोग करते हैं लेकिन खर्च के बारे में कोई एकीकृत दृष्टिकोण नहीं रखते हैं। प्रत्येक प्रदाता का अपना बिलिंग डैशबोर्ड होता है, लेकिन कोई समेकित दृश्य नहीं होता है। अप्रत्याशित लागतें बढ़ सकती हैं।
+- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
+- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection
+- **Onboarding Wizard** — Guided 4-step setup for first-time users
+- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers
-**ओम्नीरूट इसे कैसे हल करता है:**
+
--**लागत विश्लेषण डैशबोर्ड**— प्रति प्रदाता प्रति टोकन लागत ट्रैकिंग और बजट प्रबंधन
--**प्रति स्तर बजट सीमा**- प्रति स्तर खर्च की अधिकतम सीमा जो स्वचालित फ़ॉलबैक को ट्रिगर करती है
--**प्रति-मॉडल मूल्य निर्धारण कॉन्फ़िगरेशन**- प्रति मॉडल कॉन्फ़िगर करने योग्य कीमतें
--**प्रति एपीआई कुंजी उपयोग सांख्यिकी**- अनुरोध गणना और प्रति कुंजी अंतिम बार उपयोग किया गया टाइमस्टैम्प
--**एनालिटिक्स डैशबोर्ड**- स्टेट कार्ड, मॉडल उपयोग चार्ट, सफलता दर और विलंबता के साथ प्रदाता तालिका
+
+🔑 8. "Managing OAuth tokens from multiple providers is hell"
-<विवरण>
-<सारांश>🐛 10. "मैं एआई कॉल में त्रुटियों और समस्याओं का निदान नहीं कर सकता"सारांश>
+Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic.
-जब कोई कॉल विफल हो जाती है, तो देव को पता नहीं चलता कि यह दर सीमा, समाप्त टोकन, गलत प्रारूप या प्रदाता त्रुटि थी। विभिन्न टर्मिनलों पर खंडित लॉग। अवलोकन के बिना, डिबगिंग परीक्षण-और-त्रुटि है।
+**How OmniRoute solves it:**
-**ओम्नीरूट इसे कैसे हल करता है:**
+- **Auto Token Refresh** — OAuth tokens refresh in background before expiration
+- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
+- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction
+- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers
+- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility
+- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker
--**एकीकृत लॉग डैशबोर्ड**- 4 टैब: अनुरोध लॉग, प्रॉक्सी लॉग, ऑडिट लॉग, कंसोल
--**कंसोल लॉग व्यूअर**- रंग-कोडित स्तरों, ऑटो-स्क्रॉल, खोज, फ़िल्टर के साथ वास्तविक समय टर्मिनल-शैली व्यूअर
--**SQLite प्रॉक्सी लॉग्स**- लगातार लॉग जो सर्वर पुनरारंभ होने से बचे रहते हैं
--**अनुवादक खेल का मैदान**- 4 डिबगिंग मोड: खेल का मैदान (प्रारूप अनुवाद), चैट टेस्टर (राउंड-ट्रिप), टेस्ट बेंच (बैच), लाइव मॉनिटर (वास्तविक समय)
--**अनुरोध टेलीमेट्री**- p50/p95/p99 विलंबता + X-अनुरोध-आईडी ट्रेसिंग
--**रोटेशन के साथ फ़ाइल-आधारित लॉगिंग**- ऐप लॉग आकार, अवधारण दिनों और संग्रह गणना के अनुसार घूमते हैं; कॉल लॉग कलाकृतियाँ अवधारण दिनों और फ़ाइल गणना के अनुसार घूमती हैं
--**सिस्टम जानकारी रिपोर्ट**- `npm run system-info` आपके पूर्ण वातावरण (नोड संस्करण, ओमनीरूट संस्करण, ओएस, सीएलआई उपकरण, डॉकर/पीएम2 स्थिति) के साथ `system-info.txt` उत्पन्न करता है। त्वरित ट्राइएज के लिए समस्याओं की रिपोर्ट करते समय इसे संलग्न करें।
+
-<विवरण>
-<सारांश>🏗️ 11. "प्रवेश द्वार की तैनाती और रखरखाव जटिल है"सारांश>
+
+📊 9. "I don't know how much I'm spending or where"
-विभिन्न वातावरणों (स्थानीय, वीपीएस, डॉकर, क्लाउड) में एआई प्रॉक्सी को स्थापित करना, कॉन्फ़िगर करना और बनाए रखना श्रम-गहन है। हार्डकोडेड पथ, निर्देशिकाओं पर `EACCES`, पोर्ट विरोध और क्रॉस-प्लेटफ़ॉर्म बिल्ड जैसी समस्याएं घर्षण बढ़ाती हैं।
+Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up.
-**ओम्नीरूट इसे कैसे हल करता है:**
+**How OmniRoute solves it:**
--**एनपीएम ग्लोबल इंस्टाल**- `एनपीएम इंस्टाल -जी ऑम्निरूटे && ऑम्निरूटे` - हो गया
--**डॉकर मल्टी-प्लेटफ़ॉर्म**- AMD64 + ARM64 नेटिव (Apple सिलिकॉन, AWS ग्रेविटॉन, रास्पबेरी पाई)
--**डॉकर कंपोज प्रोफाइल**- `बेस` (कोई सीएलआई उपकरण नहीं) और `सीएलआई` (क्लाउड कोड, कोडेक्स, ओपनक्लाव के साथ)
--**इलेक्ट्रॉन डेस्कटॉप ऐप**- सिस्टम ट्रे, ऑटो-स्टार्ट, ऑफ़लाइन मोड के साथ विंडोज/मैकओएस/लिनक्स के लिए मूल ऐप
--**स्प्लिट-पोर्ट मोड**- उन्नत परिदृश्यों के लिए अलग-अलग पोर्ट पर एपीआई और डैशबोर्ड (रिवर्स प्रॉक्सी, कंटेनर नेटवर्किंग)
--**क्लाउड सिंक**- क्लाउडफ्लेयर वर्कर्स के माध्यम से सभी डिवाइसों में कॉन्फिग सिंक्रोनाइजेशन
--**डीबी बैकअप**- बाह्य रूप से प्रबंधित बैकअप के लिए `DISABLE_SQLITE_AUTO_BACKUP` के साथ सभी सेटिंग्स का स्वचालित बैकअप, पुनर्स्थापना, निर्यात और आयात
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider
+- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback
+- **Per-Model Pricing Configuration** — Configurable prices per model
+- **Usage Statistics Per API Key** — Request count and last-used timestamp per key
+- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency
-<विवरण>
-<सारांश>🌍 12. "इंटरफ़ेस केवल अंग्रेजी है और मेरी टीम अंग्रेजी नहीं बोलती है"सारांश>
+
-गैर-अंग्रेजी भाषी देशों, विशेष रूप से लैटिन अमेरिका, एशिया और यूरोप में टीमें, केवल अंग्रेजी इंटरफेस के साथ संघर्ष करती हैं। भाषा बाधाएँ अपनाने को कम करती हैं और कॉन्फ़िगरेशन त्रुटियों को बढ़ाती हैं।
+
+🐛 10. "I can't diagnose errors and problems in AI calls"
-**ओम्नीरूट इसे कैसे हल करता है:**
+When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error.
--**डैशबोर्ड i18n - 30 भाषाएँ**- अरबी, बल्गेरियाई, डेनिश, जर्मन, स्पेनिश, फिनिश, फ्रेंच, हिब्रू, हिंदी, हंगेरियन, इंडोनेशियाई, इतालवी, जापानी, कोरियाई, मलय, डच, नॉर्वेजियन, पोलिश, पुर्तगाली (पीटी/बीआर), रोमानियाई, रूसी, स्लोवाक, स्वीडिश, थाई, यूक्रेनी, वियतनामी, चीनी, फिलिपिनो, अंग्रेजी सहित सभी 500+ कुंजियाँ अनुवादित
--**आरटीएल समर्थन**- अरबी और हिब्रू के लिए दाएं से बाएं समर्थन
--**बहु-भाषा रीडमी**- 30 पूर्ण दस्तावेज़ीकरण अनुवाद
--**भाषा चयनकर्ता**- वास्तविक समय स्विचिंग के लिए हेडर में ग्लोब आइकन
+**How OmniRoute solves it:**
-<विवरण>
-<सारांश>🔄 13. "मुझे चैट से अधिक की आवश्यकता है - मुझे एम्बेडिंग, चित्र, ऑडियो की आवश्यकता है"सारांश>
+- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console
+- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter
+- **SQLite Proxy Logs** — Persistent logs that survive server restarts
+- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time)
+- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing
+- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count
+- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage.
-एआई का मतलब सिर्फ चैट पूरा करना नहीं है। डेवलपर्स को छवियां उत्पन्न करने, ऑडियो ट्रांसक्राइब करने, आरएजी के लिए एम्बेडिंग बनाने, दस्तावेज़ों को फिर से रैंक करने और सामग्री को मॉडरेट करने की आवश्यकता होती है। प्रत्येक एपीआई का एक अलग समापन बिंदु और प्रारूप होता है।
+
-**ओम्नीरूट इसे कैसे हल करता है:**
+
+🏗️ 11. "Deploying and maintaining the gateway is complex"
--**एंबेडिंग**- `/v1/एंबेडिंग` 6 प्रदाताओं और 9+ मॉडल के साथ
--**छवि निर्माण**- 10 प्रदाताओं और 20+ मॉडलों के साथ `/v1/छवियां/पीढ़ी` (ओपनएआई, एक्सएआई, टुगेदर, फायरवर्क्स, नेबियस, हाइपरबोलिक, नैनोबनाना, एंटीग्रेविटी, एसडी वेबयूआई, कॉम्फीयूआई)
--**टेक्स्ट-टू-वीडियो**- `/v1/वीडियो/पीढ़ी` - कॉम्फीयूआई (एनिमेटडिफ, एसवीडी) और एसडी वेबयूआई
--**टेक्स्ट-टू-म्यूजिक**- `/v1/म्यूजिक/पीढ़ी` - कॉम्फीयूआई (स्थिर ऑडियो ओपन, म्यूजिकजेन)
--**ऑडियो ट्रांसक्रिप्शन**- `/v1/ऑडियो/ट्रांसक्रिप्शन` - व्हिस्पर + एनवीडिया एनआईएम, हगिंगफेस, क्वेन3
--**टेक्स्ट-टू-स्पीच**- `/v1/ऑडियो/स्पीच` - इलेवनलैब्स, एनवीडिया एनआईएम, हगिंगफेस, कोक्वी, टोरटोइज़, क्वेन3,**इनवर्ल्ड**,**कार्टेसिया**,**प्लेएचटी**, + मौजूदा प्रदाता
--**मॉडरेशन**- `/v1/मॉडरेशन` - सामग्री सुरक्षा जांच
--**रीरैंकिंग**— `/v1/rerank` — दस्तावेज़ प्रासंगिकता रीरैंकिंग
--**प्रतिक्रिया एपीआई**- कोडेक्स के लिए पूर्ण `/v1/प्रतिक्रिया` समर्थन
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction.
-<विवरण>
-<सारांश>🧪 14. "मेरे पास सभी मॉडलों की गुणवत्ता का परीक्षण और तुलना करने का कोई तरीका नहीं है"सारांश>
+**How OmniRoute solves it:**
-डेवलपर्स जानना चाहते हैं कि उनके उपयोग के मामले में कौन सा मॉडल सबसे अच्छा है - कोड, अनुवाद, तर्क - लेकिन मैन्युअल रूप से तुलना करना धीमा है। कोई एकीकृत eval उपकरण मौजूद नहीं है।
+- **npm global install** — `npm install -g omniroute && omniroute` — done
+- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi)
+- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw)
+- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode
+- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking)
+- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers
+- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups
-**ओम्नीरूट इसे कैसे हल करता है:**
+
--**एलएलएम मूल्यांकन**- अभिवादन, गणित, भूगोल, कोड जनरेशन, JSON अनुपालन, अनुवाद, मार्कडाउन, सुरक्षा इनकार को कवर करने वाले 10 प्री-लोडेड मामलों के साथ गोल्डन सेट परीक्षण
--**4 मिलान रणनीतियाँ**- `सटीक`, `शामिल`, `रेगेक्स`, `कस्टम` (जेएस फ़ंक्शन)
--**अनुवादक खेल का मैदान परीक्षण बेंच**- एकाधिक इनपुट और अपेक्षित आउटपुट, क्रॉस-प्रदाता तुलना के साथ बैच परीक्षण
--**चैट परीक्षक**- दृश्य प्रतिक्रिया प्रतिपादन के साथ पूर्ण राउंड-ट्रिप
--**लाइव मॉनिटर**- प्रॉक्सी के माध्यम से बहने वाले सभी अनुरोधों की वास्तविक समय स्ट्रीम
+
+🌍 12. "The interface is English-only and my team doesn't speak English"
-<विवरण>
-<सारांश>📈 15. "मुझे प्रदर्शन खोए बिना स्केल करने की आवश्यकता है"सारांश>
+Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors.
-जैसे-जैसे अनुरोध की मात्रा बढ़ती है, कैशिंग के बिना वही प्रश्न डुप्लिकेट लागत उत्पन्न करते हैं। निष्क्रियता के बिना, डुप्लिकेट अपशिष्ट प्रसंस्करण का अनुरोध करता है। प्रति-प्रदाता दर सीमा का सम्मान किया जाना चाहिए।
+**How OmniRoute solves it:**
-**ओम्नीरूट इसे कैसे हल करता है:**
+- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English
+- **RTL Support** — Right-to-left support for Arabic and Hebrew
+- **Multi-Language READMEs** — 30 complete documentation translations
+- **Language Selector** — Globe icon in header for real-time switching
--**सिमेंटिक कैश**- दो-स्तरीय कैश (हस्ताक्षर + सिमेंटिक) लागत और विलंबता को कम करता है
--**अनुरोध Idempotency**- समान अनुरोधों के लिए 5s डिडुप्लीकेशन विंडो
--**दर सीमा का पता लगाना**- प्रति-प्रदाता आरपीएम, न्यूनतम अंतर, और अधिकतम समवर्ती ट्रैकिंग
--**संपादन योग्य दर सीमाएँ**— सेटिंग्स में कॉन्फ़िगर करने योग्य डिफ़ॉल्ट → दृढ़ता के साथ लचीलापन
--**एपीआई कुंजी सत्यापन कैश**- उत्पादन प्रदर्शन के लिए 3-स्तरीय कैश
--**टेलीमेट्री के साथ स्वास्थ्य डैशबोर्ड**— p50/p95/p99 विलंबता, कैश आँकड़े, अपटाइम
+
-<विवरण>
-<सारांश>🤖 16. "मैं विश्व स्तर पर मॉडल व्यवहार को नियंत्रित करना चाहता हूं"सारांश>
+
+🔄 13. "I need more than chat — I need embeddings, images, audio"
-ऐसे डेवलपर जो सभी प्रतिक्रियाएं एक विशिष्ट भाषा में, एक विशिष्ट लहजे में चाहते हैं, या तर्क टोकन को सीमित करना चाहते हैं। प्रत्येक टूल/अनुरोध में इसे कॉन्फ़िगर करना अव्यावहारिक है।
+AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format.
-**ओम्नीरूट इसे कैसे हल करता है:**
+**How OmniRoute solves it:**
--**सिस्टम प्रॉम्प्ट इंजेक्शन**— ग्लोबल प्रॉम्प्ट सभी अनुरोधों पर लागू होता है
--**सोच बजट सत्यापन**- प्रति अनुरोध तर्क टोकन आवंटन नियंत्रण (पासथ्रू, ऑटो, कस्टम, अनुकूली)
--**9 रूटिंग रणनीतियाँ**- वैश्विक रणनीतियाँ जो यह निर्धारित करती हैं कि अनुरोध कैसे वितरित किए जाते हैं
--**वाइल्डकार्ड राउटर**- `प्रदाता/*` पैटर्न किसी भी प्रदाता को गतिशील रूप से रूट करता है
--**कॉम्बो सक्षम/अक्षम टॉगल**— कॉम्बो को सीधे डैशबोर्ड से टॉगल करें
--**प्रदाता टॉगल**— एक क्लिक से प्रदाता के लिए सभी कनेक्शन सक्षम/अक्षम करें
--**अवरुद्ध प्रदाता**- `/v1/मॉडल` सूची से विशिष्ट प्रदाताओं को बाहर निकालें
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models
+- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
+- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI
+- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
+- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
+- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers
+- **Moderations** — `/v1/moderations` — Content safety checks
+- **Reranking** — `/v1/rerank` — Document relevance reranking
+- **Responses API** — Full `/v1/responses` support for Codex
-<विवरण>
-<सारांश>🧰 17. "मुझे प्रथम श्रेणी उत्पाद क्षमताओं के रूप में एमसीपी टूल्स की आवश्यकता है"सारांश>
+
-कई एआई गेटवे एमसीपी को केवल एक छिपे हुए कार्यान्वयन विवरण के रूप में उजागर करते हैं। टीमों को एक दृश्यमान, प्रबंधनीय संचालन परत की आवश्यकता होती है।
+
+🧪 14. "I have no way to test and compare quality across models"
-**ओम्नीरूट इसे कैसे हल करता है:**
+Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist.
-- एमसीपी डैशबोर्ड नेविगेशन और एंडपॉइंट प्रोटोकॉल टैब में दिखाई देता है
-- प्रक्रिया, उपकरण, कार्यक्षेत्र और ऑडिट के साथ समर्पित एमसीपी प्रबंधन पृष्ठ
-- `omniroute --mcp` और क्लाइंट ऑनबोर्डिंग के लिए बिल्ट-इन क्विक-स्टार्ट
+**How OmniRoute solves it:**
-<विवरण>
-<सारांश>🧠 18. "मुझे सिंक + स्ट्रीम कार्य पथों के साथ A2A ऑर्केस्ट्रेशन की आवश्यकता है"सारांश>
+- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal
+- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function)
+- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison
+- **Chat Tester** — Full round-trip with visual response rendering
+- **Live Monitor** — Real-time stream of all requests flowing through the proxy
-एजेंट वर्कफ़्लो को जीवनचक्र नियंत्रण के साथ सीधे उत्तर और लंबे समय तक चलने वाले स्ट्रीम निष्पादन दोनों की आवश्यकता होती है।
+
-**ओम्नीरूट इसे कैसे हल करता है:**
+
+📈 15. "I need to scale without losing performance"
-- A2A JSON-RPC एंडपॉइंट (`POST /a2a`) `मैसेज/सेंड` और `मैसेज/स्ट्रीम` के साथ
-- टर्मिनल राज्य प्रसार के साथ एसएसई स्ट्रीमिंग
-- `कार्य/प्राप्त करें` और `कार्य/रद्द करें` के लिए कार्य जीवनचक्र एपीआई
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected.
-<विवरण>
-<सारांश>🛰️ 19. "मुझे वास्तविक एमसीपी प्रक्रिया स्वास्थ्य की आवश्यकता है, अनुमानित स्थिति की नहीं"सारांश>
+**How OmniRoute solves it:**
-परिचालन टीमों को यह जानने की जरूरत है कि क्या एमसीपी वास्तव में जीवित है, न कि केवल एपीआई पहुंच योग्य है या नहीं।
+- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency
+- **Request Idempotency** — 5s deduplication window for identical requests
+- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking
+- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence
+- **API Key Validation Cache** — 3-tier cache for production performance
+- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime
-**ओम्नीरूट इसे कैसे हल करता है:**
+
-- पीआईडी, टाइमस्टैम्प, ट्रांसपोर्ट, टूल काउंट और स्कोप मोड के साथ रनटाइम हार्टबीट फ़ाइल
-- एमसीपी स्थिति एपीआई दिल की धड़कन + हाल की गतिविधि का संयोजन
-- प्रक्रिया/अपटाइम/दिल की धड़कन ताजगी के लिए यूआई स्टेटस कार्ड
+
+🤖 16. "I want to control model behavior globally"
-<विवरण>
-<सारांश>📋 20. "मुझे ऑडिटेबल एमसीपी टूल निष्पादन की आवश्यकता है"सारांश>
+Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical.
-जब उपकरण कॉन्फ़िगरेशन को बदलते हैं या ऑप्स क्रियाओं को ट्रिगर करते हैं, तो टीमों को फोरेंसिक ट्रैसेबिलिटी की आवश्यकता होती है।
+**How OmniRoute solves it:**
-**ओम्नीरूट इसे कैसे हल करता है:**
+- **System Prompt Injection** — Global prompt applied to all requests
+- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive)
+- **9 Routing Strategies** — Global strategies that determine how requests are distributed
+- **Wildcard Router** — `provider/*` patterns route dynamically to any provider
+- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard
+- **Provider Toggle** — Enable/disable all connections for a provider with one click
+- **Blocked Providers** — Exclude specific providers from `/v1/models` listing
-- MCP टूल कॉल के लिए SQLite समर्थित ऑडिट लॉगिंग
-- टूल, सफलता/असफलता, एपीआई कुंजी और पेजिनेशन द्वारा फ़िल्टर
-- डैशबोर्ड ऑडिट टेबल + स्वचालन के लिए आँकड़े समापन बिंदु
+
-<विवरण>
-<सारांश>🔐 21. "मुझे प्रति एकीकरण के लिए स्कोप्ड एमसीपी अनुमतियों की आवश्यकता है"सारांश>
+
+🧰 17. "I need MCP tools as first-class product capabilities"
-विभिन्न ग्राहकों को टूल श्रेणियों तक कम से कम विशेषाधिकार प्राप्त होना चाहिए।
+Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer.
-**ओम्नीरूट इसे कैसे हल करता है:**
+**How OmniRoute solves it:**
-- नियंत्रित टूल एक्सेस के लिए 10 दानेदार एमसीपी स्कोप
-- एमसीपी प्रबंधन यूआई में दायरा प्रवर्तन और दृश्यता
-- परिचालन टूलींग के लिए सुरक्षित डिफ़ॉल्ट मुद्रा
+- MCP appears in the dashboard navigation and endpoint protocol tab
+- Dedicated MCP management page with process, tools, scopes, and audit
+- Built-in quick-start for `omniroute --mcp` and client onboarding
-<विवरण>
-<सारांश>⚙️ 22. "मुझे पुनः तैनाती के बिना परिचालन नियंत्रण की आवश्यकता है"सारांश>
+
-घटनाओं या लागत आयोजनों के दौरान टीमों को त्वरित रनटाइम परिवर्तन की आवश्यकता होती है।
+
+🧠 18. "I need A2A orchestration with sync + stream task paths"
-**ओम्नीरूट इसे कैसे हल करता है:**
+Agent workflows need both direct replies and long-running streamed execution with lifecycle control.
-- कॉम्बो सक्रियण को सीधे एमसीपी डैशबोर्ड से स्विच करें
-- पूर्व-निर्धारित पॉलिसी पैक से लचीलापन प्रोफ़ाइल लागू करें
-- उसी ऑपरेशन पैनल से सर्किट ब्रेकर स्थिति को रीसेट करें
+**How OmniRoute solves it:**
-<विवरण>
-<सारांश>🔄 23. "मुझे लाइव ए2ए कार्य जीवनचक्र दृश्यता और रद्दीकरण की आवश्यकता है"सारांश>
+- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream`
+- SSE streaming with terminal state propagation
+- Task lifecycle APIs for `tasks/get` and `tasks/cancel`
-जीवनचक्र दृश्यता के बिना, कार्य घटनाओं का परीक्षण करना कठिन हो जाता है।
+
-**ओम्नीरूट इसे कैसे हल करता है:**
+
+🛰️ 19. "I need real MCP process health, not guessed status"
-- पेजिनेशन के साथ राज्य/कौशल द्वारा कार्य सूचीकरण/फ़िल्टरिंग
-- कार्य मेटाडेटा, घटनाओं और कलाकृतियों पर ड्रिल-डाउन
-- पुष्टि के साथ कार्य रद्दीकरण समापन बिंदु और यूआई कार्रवाई
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable.
-<विवरण>
-<सारांश>🌊 24. "मुझे A2A लोड के लिए सक्रिय स्ट्रीम मेट्रिक्स की आवश्यकता है"सारांश>
+**How OmniRoute solves it:**
-स्ट्रीमिंग वर्कफ़्लो के लिए समवर्ती और लाइव कनेक्शन में परिचालन अंतर्दृष्टि की आवश्यकता होती है।
+- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode
+- MCP status API combining heartbeat + recent activity
+- UI status cards for process/uptime/heartbeat freshness
-**ओम्नीरूट इसे कैसे हल करता है:**
+
-- सक्रिय स्ट्रीम काउंटर A2A स्थिति में एकीकृत
-- अंतिम कार्य टाइमस्टैम्प और प्रति-राज्य गणना
-- वास्तविक समय ऑप्स निगरानी के लिए A2A डैशबोर्ड कार्ड
+
+📋 20. "I need auditable MCP tool execution"
-<विवरण>
-<सारांश>🪪 25. "मुझे ग्राहकों के लिए मानक एजेंट खोज की आवश्यकता है"सारांश>
+When tools mutate config or trigger ops actions, teams need forensic traceability.
-बाहरी ग्राहकों और ऑर्केस्ट्रेटर्स को ऑनबोर्डिंग के लिए मशीन-पठनीय मेटाडेटा की आवश्यकता होती है।
+**How OmniRoute solves it:**
-**ओम्नीरूट इसे कैसे हल करता है:**
+- SQLite-backed audit logging for MCP tool calls
+- Filters by tool, success/failure, API key, and pagination
+- Dashboard audit table + stats endpoints for automation
-- एजेंट कार्ड `/.well-known/agent.json` पर प्रदर्शित किया गया
-- प्रबंधन यूआई में दिखाई गई क्षमताएं और कौशल
-- A2A स्थिति API में स्वचालन के लिए खोज मेटाडेटा शामिल है
+
-<विवरण>
+
+🔐 21. "I need scoped MCP permissions per integration"
+
+Different clients should have least-privilege access to tool categories.
+
+**How OmniRoute solves it:**
+
+- 10 granular MCP scopes for controlled tool access
+- Scope enforcement and visibility in MCP management UI
+- Safe default posture for operational tooling
+
+
+
+
+⚙️ 22. "I need operational controls without redeploying"
+
+Teams need quick runtime changes during incidents or cost events.
+
+**How OmniRoute solves it:**
+
+- Switch combo activation directly from MCP dashboard
+- Apply resilience profiles from pre-defined policy packs
+- Reset circuit breaker state from the same operations panel
+
+
+
+
+🔄 23. "I need live A2A task lifecycle visibility and cancellation"
+
+Without lifecycle visibility, task incidents become hard to triage.
+
+**How OmniRoute solves it:**
+
+- Task listing/filtering by state/skill with pagination
+- Drill-down on task metadata, events, and artifacts
+- Task cancellation endpoint and UI action with confirmation
+
+
+
+
+🌊 24. "I need active stream metrics for A2A load"
+
+Streaming workflows require operational insight into concurrency and live connections.
+
+**How OmniRoute solves it:**
+
+- Active stream counters integrated into A2A status
+- Last task timestamp and per-state counts
+- A2A dashboard cards for real-time ops monitoring
+
+
+
+
+🪪 25. "I need standard agent discovery for clients"
+
+External clients and orchestrators need machine-readable metadata for onboarding.
+
+**How OmniRoute solves it:**
+
+- Agent Card exposed at `/.well-known/agent.json`
+- Capabilities and skills shown in management UI
+- A2A status API includes discovery metadata for automation
+
+
+
+
🧭 26. "I need protocol discoverability in the product UX"
-यदि उपयोगकर्ता प्रोटोकॉल सतहों की खोज नहीं कर पाते हैं, तो अपनाने और समर्थन की गुणवत्ता में गिरावट आती है।
+If users cannot discover protocol surfaces, adoption and support quality drop.
-**ओम्नीरूट इसे कैसे हल करता है:**
+**How OmniRoute solves it:**
-- प्रॉक्सी, एमसीपी, ए2ए और एपीआई एंडपॉइंट के लिए टैब के साथ समेकित**एंडपॉइंट**पेज
-- एमसीपी और ए2ए के लिए इनलाइन सेवा स्थिति टॉगल (ऑनलाइन/ऑफ़लाइन)।
-- सिंहावलोकन से लेकर समर्पित प्रबंधन टैब तक के लिंक
+- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints
+- Inline service status toggles (Online/Offline) for MCP and A2A
+- Links from overview to dedicated management tabs
-<विवरण>
-<सारांश>🧪 27. "मुझे वास्तविक ग्राहकों के साथ एंड-टू-एंड प्रोटोकॉल सत्यापन की आवश्यकता है"सारांश>
+
-रिलीज़ से पहले प्रोटोकॉल संगतता को सत्यापित करने के लिए मॉक परीक्षण पर्याप्त नहीं हैं।
+
+🧪 27. "I need end-to-end protocol validation with real clients"
-**ओम्नीरूट इसे कैसे हल करता है:**
+Mock tests are not enough to validate protocol compatibility before release.
-- E2E सुइट जो ऐप को बूट करता है और वास्तविक MCP SDK क्लाइंट ट्रांसपोर्ट का उपयोग करता है
-- A2A क्लाइंट खोज, भेजने, स्ट्रीम करने, प्राप्त करने और प्रवाह को रद्द करने के लिए परीक्षण करता है
-- एमसीपी ऑडिट और ए2ए कार्य एपीआई के खिलाफ दावों की क्रॉस-चेक करें
+**How OmniRoute solves it:**
-<विवरण>
-<सारांश>📡 28. "मुझे सभी इंटरफेस में एकीकृत अवलोकन की आवश्यकता है"सारांश>
+- E2E suite that boots app and uses real MCP SDK client transport
+- A2A client tests for discovery, send, stream, get, and cancel flows
+- Cross-check assertions against MCP audit and A2A tasks APIs
-प्रोटोकॉल द्वारा अवलोकनशीलता को विभाजित करने से ब्लाइंड स्पॉट और लंबा एमटीटीआर बनता है।
+
-**ओम्नीरूट इसे कैसे हल करता है:**
+
+📡 28. "I need unified observability across all interfaces"
-- एक उत्पाद में एकीकृत डैशबोर्ड/लॉग/एनालिटिक्स
-- स्वास्थ्य + ऑडिट + ओपनएआई, एमसीपी और ए2ए परतों में टेलीमेट्री अनुरोध
-- स्थिति और स्वचालन के लिए परिचालन एपीआई
+Splitting observability by protocol creates blind spots and longer MTTR.
-<विवरण>
-<सारांश>💼 29. "मुझे प्रॉक्सी + टूल्स + एजेंट ऑर्केस्ट्रेशन के लिए एक रनटाइम की आवश्यकता है"सारांश>
+**How OmniRoute solves it:**
-कई अलग-अलग सेवाएँ चलाने से परिचालन लागत और विफलता मोड बढ़ जाते हैं।
+- Unified dashboards/logs/analytics in one product
+- Health + audit + request telemetry across OpenAI, MCP, and A2A layers
+- Operational APIs for status and automation
-**ओम्नीरूट इसे कैसे हल करता है:**
+
-- OpenAI-संगत प्रॉक्सी, MCP सर्वर और A2A सर्वर एक स्टैक में
-- साझा प्रमाणीकरण, लचीलापन, डेटा भंडारण और अवलोकन क्षमता
-- सभी संपर्क सतहों पर सुसंगत नीति मॉडल
+
+💼 29. "I need one runtime for proxy + tools + agent orchestration"
-<विवरण>
-<सारांश>🚀 30. "मुझे ग्लू-कोड फैलाव के बिना एजेंटिक वर्कफ़्लो भेजने की आवश्यकता है"सारांश>
+Running many separate services increases operational cost and failure modes.
-कई तदर्थ सेवाओं और स्क्रिप्ट्स को सिलाई करते समय टीमों की गति कम हो जाती है।
+**How OmniRoute solves it:**
-**ओम्नीरूट इसे कैसे हल करता है:**
+- OpenAI-compatible proxy, MCP server, and A2A server in one stack
+- Shared auth, resilience, data store, and observability
+- Consistent policy model across all interaction surfaces
-- ग्राहकों और एजेंटों के लिए एकीकृत समापन बिंदु रणनीति
-- अंतर्निहित प्रोटोकॉल प्रबंधन यूआई और धूम्रपान सत्यापन पथ
-- उत्पादन के लिए तैयार नींव (सुरक्षा, लॉगिंग, लचीलापन, बैकअप)
+
+
+
+🚀 30. "I need to ship agentic workflows without glue-code sprawl"
+
+Teams lose velocity when stitching multiple ad-hoc services and scripts.
+
+**How OmniRoute solves it:**
+
+- Unified endpoint strategy for clients and agents
+- Built-in protocol management UIs and smoke validation paths
+- Production-ready foundations (security, logging, resilience, backup)
+
+
### Example Playbooks (Integrated Use Cases)
-**प्लेबुक ए: सशुल्क सदस्यता + सस्ता बैकअप अधिकतम करें**```txt
+**Playbook A: Maximize paid subscription + cheap backup**
+
+```txt
Combo: "maximize-claude"
1. cc/claude-opus-4-6
2. glm/glm-4.7
@@ -607,21 +690,23 @@ Combo: "maximize-claude"
Monthly cost: $20 + small backup spend
Outcome: higher quality, near-zero interruption
-````
+```
-**प्लेबुक बी: शून्य-लागत कोडिंग स्टैक**```txt
+**Playbook B: Zero-cost coding stack**
+
+```txt
Combo: "free-forever"
-
-1. gc/gemini-3-flash
-2. if/kimi-k2-thinking
-3. qw/qwen3-coder-plus
+ 1. gc/gemini-3-flash
+ 2. if/kimi-k2-thinking
+ 3. qw/qwen3-coder-plus
Monthly cost: $0
Outcome: stable free coding workflow
+```
-````
+**Playbook C: 24/7 always-on fallback chain**
-**प्लेबुक सी: 24/7 हमेशा चालू फ़ॉलबैक श्रृंखला**```txt
+```txt
Combo: "always-on"
1. cc/claude-opus-4-6
2. cx/gpt-5.2-codex
@@ -630,122 +715,134 @@ Combo: "always-on"
5. if/kimi-k2-thinking
Outcome: deep fallback depth for deadline-critical workloads
-````
+```
-**प्लेबुक डी: एजेंट एमसीपी + ए2ए के साथ काम करता है**```txt
+**Playbook D: Agent ops with MCP + A2A**
-1. Start MCP transport (`omniroute --mcp`) for tool-driven operations
-2. Run A2A tasks via `message/send` and `message/stream`
-3. Observe via /dashboard/endpoint (MCP and A2A tabs)
-4. Toggle services via inline status controls
-
-````
+```txt
+1) Start MCP transport (`omniroute --mcp`) for tool-driven operations
+2) Run A2A tasks via `message/send` and `message/stream`
+3) Observe via /dashboard/endpoint (MCP and A2A tabs)
+4) Toggle services via inline status controls
+```
---
## 🆓 Start Free — Zero Configuration Cost
->**$0/माह**पर मिनटों में AI कोडिंग सेटअप करें। इन मुफ़्त खातों को कनेक्ट करें और अंतर्निहित**फ़्री स्टैक**कॉम्बो का उपयोग करें।
+> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo.
-| कदम | कार्रवाई | प्रदाता अनलॉक |
+| Step | Action | Providers Unlocked |
| ---- | -------------------------------------------------- | ------------------------------------------------------------------ |
-| 1 | कनेक्ट**किरो**(AWS बिल्डर आईडी OAuth) | क्लाउड सॉनेट 4.5, हाइकु 4.5 -**असीमित**|
-| 2 | कनेक्ट करें**Qoder**(Google OAuth) | किमी-के2-सोच, क्वेन3-कोडर-प्लस, डीपसीक-आर1... —**असीमित**|
-| 3 | कनेक्ट**क्वेन**(डिवाइस कोड) | qwen3-कोडर-प्लस, qwen3-कोडर-फ़्लैश... —**असीमित**|
-| 4 | कनेक्ट**मिथुन सीएलआई**(Google OAuth) | जेमिनी-3-फ़्लैश, जेमिनी-2.5-प्रो —**180K/महीना मुफ़्त**|
-| 5 | `/डैशबोर्ड/कॉम्बोस` →**फ्री स्टैक ($0)**टेम्पलेट | सभी मुफ़्त प्रदाताओं को स्वचालित रूप से राउंड-रॉबिन करें |
+| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** |
+| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** |
+| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** |
+| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** |
+| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically |
-**किसी भी आईडीई/सीएलआई को यहां इंगित करें:**`http://localhost:20128/v1` · एपीआई कुंजी: `any-string` · हो गया।
+**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done.
->**वैकल्पिक अतिरिक्त कवरेज (निःशुल्क भी):**ग्रोक एपीआई कुंजी (30 आरपीएम मुफ्त), एनवीडिया एनआईएम (40 आरपीएम मुफ्त, 70+ मॉडल), सेरेब्रस (1एम टोकन/दिन), लॉन्गकैट एपीआई कुंजी (50एम टोकन/दिन!), क्लाउडफ्लेयर वर्कर्स एआई (10के न्यूरॉन्स/दिन, 50+ मॉडल)।## त्वरित प्रारंभ
+> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models).
+
+## त्वरित प्रारंभ
### 1) Install and run
```bash
npm install -g omniroute
omniroute
-````
+```
-> **pnpm उपयोगकर्ता:**`better-sqlite3` और `@swc/core` के लिए आवश्यक मूल बिल्ड स्क्रिप्ट को सक्षम करने के लिए इंस्टॉल के बाद `pnpm Approve-builds -g` चलाएँ:
+> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`:
>
-> ```बैश
-> पीएनपीएम इंस्टाल -जी ऑम्नीरूट
-> पीएनपीएम अप्रूव-बिल्ड्स -जी # सभी पैकेजों का चयन करें → स्वीकृत करें
-> सर्वमार्ग
+> ```bash
+> pnpm install -g omniroute
+> pnpm approve-builds -g # Select all packages → approve
+> omniroute
> ```
-डैशबोर्ड `http://localhost:20128` पर खुलता है और एपीआई बेस यूआरएल `http://localhost:20128/v1` है।
+Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`.
-| आदेश | विवरण |
-| ----------------------- | -------------------------------------------------------------------- | ----------- |
-| `सर्वव्यापी` | सर्वर प्रारंभ करें (`पोर्ट=20128`, एपीआई और डैशबोर्ड एक ही पोर्ट पर) |
-| `ओम्नीरूटे--पोर्ट 3000` | कैनोनिकल/एपीआई पोर्ट को 3000 | पर सेट करें |
-| `omniroute --mcp` | MCP सर्वर (stdio ट्रांसपोर्ट) प्रारंभ करें |
-| `omniroute --no-open` | ब्राउज़र को स्वतः न खोलें |
-| `omniroute --help` | सहायता दिखाएँ |
+| Command | Description |
+| ----------------------- | ----------------------------------------------------------- |
+| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) |
+| `omniroute --port 3000` | Set canonical/API port to 3000 |
+| `omniroute --mcp` | Start MCP server (stdio transport) |
+| `omniroute --no-open` | Don't auto-open browser |
+| `omniroute --help` | Show help |
-वैकल्पिक स्प्लिट-पोर्ट मोड:```bash
+Optional split-port mode:
+
+```bash
PORT=20128 DASHBOARD_PORT=20129 omniroute
-
-# API: http://localhost:20128/v1
-
+# API: http://localhost:20128/v1
# Dashboard: http://localhost:20129
-
-````
+```
### Long-Running Streaming Timeouts
-अधिकांश तैनाती के लिए, आपको केवल इसकी आवश्यकता है:
+For most deployments, you only need:
-| परिवर्तनीय | डिफ़ॉल्ट | उद्देश्य |
-| ---------------------- | -------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- |
-| `REQUEST_TIMEOUT_MS` | `600000` | अपस्ट्रीम फ़ेच, छिपे हुए अंडरसी टाइमआउट, टीएलएस फ़िंगरप्रिंट अनुरोध और एपीआई ब्रिज अनुरोध/प्रॉक्सी टाइमआउट के लिए साझा आधार रेखा |
-| `STREAM_IDLE_TIMEOUT_MS` | `REQUEST_TIMEOUT_MS` प्राप्त होता है | ओम्नीरूट द्वारा एसएसई स्ट्रीम को निरस्त करने से पहले स्ट्रीमिंग खंडों के बीच अधिकतम अंतर
+| Variable | Default | Purpose |
+| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- |
+| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts |
+| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream |
-बैकवर्ड संगतता संरक्षित है: मौजूदा `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, और अन्य प्रति-लेयर टाइमआउट संस्करण अभी भी काम करते हैं और साझा बेसलाइन को ओवरराइड करते हैं।
+Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline.
-यदि आपको बेहतर नियंत्रण की आवश्यकता है तो उन्नत ओवरराइड उपलब्ध हैं:| परिवर्तनीय | डिफ़ॉल्ट | उद्देश्य |
-| ------------------------------------------------ | ------------------------------------------------ | ---------------------------------------------------------------------------------- |
-| `FETCH_TIMEOUT_MS` | `REQUEST_TIMEOUT_MS` प्राप्त होता है | मुख्य फ़ेच एबॉर्ट सिग्नल द्वारा उपयोग किया गया कुल अपस्ट्रीम अनुरोध टाइमआउट |
-| `FETCH_HEADERS_TIMEOUT_MS` | `FETCH_TIMEOUT_MS` विरासत में मिला है | अपस्ट्रीम प्रतिक्रिया हेडर प्राप्त करने के लिए निर्धारित समय सीमा |
-| `FETCH_BODY_TIMEOUT_MS` | `FETCH_TIMEOUT_MS` विरासत में मिला है | अपस्ट्रीम बॉडी चंक्स के बीच निर्धारित समय सीमा (`0` इसे अक्षम कर देती है) |
-| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | अंडरसी टीसीपी कनेक्ट टाइमआउट |
-| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | अन्य आइडल कीप-अलाइव सॉकेट टाइमआउट |
-| `TLS_CLIENT_TIMEOUT_MS` | `FETCH_TIMEOUT_MS` विरासत में मिला है | `wreq-js` | के माध्यम से किए गए टीएलएस फिंगरप्रिंट अनुरोधों के लिए टाइमआउट
-| `API_BRIDGE_PROXY_TIMEOUT_MS` | `REQUEST_TIMEOUT_MS` या `30000` प्राप्त होता है | एपीआई पोर्ट से डैशबोर्ड पोर्ट तक `/v1` प्रॉक्सी अग्रेषण के लिए टाइमआउट |
-| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `अधिकतम(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | एपीआई ब्रिज सर्वर पर आने वाले अनुरोध का समय समाप्त |
-| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | एपीआई ब्रिज सर्वर पर इनकमिंग हेडर टाइमआउट |
-| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | एपीआई ब्रिज सर्वर पर कीप-अलाइव टाइमआउट |
-| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | एपीआई ब्रिज सर्वर पर सॉकेट निष्क्रियता टाइमआउट (`0` इसे अक्षम करता है) |
+Advanced overrides are available if you need finer control:
-यदि आप Nginx, Caddy, Cloudflare, या किसी अन्य रिवर्स प्रॉक्सी के पीछे ओमनीरूट चलाते हैं, तो सुनिश्चित करें कि प्रॉक्सी
-टाइमआउट आपके ओमनीरूट स्ट्रीम/फ़ेच टाइमआउट से भी अधिक हैं।### 2) Connect providers and create your API key
+| Variable | Default | Purpose |
+| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- |
+| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal |
+| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers |
+| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) |
+| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout |
+| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout |
+| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` |
+| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port |
+| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server |
+| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server |
+| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server |
+| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) |
-1. डैशबोर्ड → `प्रदाता` खोलें और कम से कम एक प्रदाता (OAuth या API कुंजी) कनेक्ट करें।
-2. डैशबोर्ड → `एंडपॉइंट्स` खोलें और एक एपीआई कुंजी बनाएं।
-3. (वैकल्पिक) डैशबोर्ड → `कॉम्बोस` खोलें और अपनी फ़ॉलबैक श्रृंखला सेट करें।### 3) Point your coding tool to OmniRoute
+If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy
+timeouts are also higher than your OmniRoute stream/fetch timeouts.
+
+### 2) Connect providers and create your API key
+
+1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key).
+2. Open Dashboard → `Endpoints` and create an API key.
+3. (Optional) Open Dashboard → `Combos` and set your fallback chain.
+
+### 3) Point your coding tool to OmniRoute
```txt
Base URL: http://localhost:20128/v1
API Key: [copy from Endpoint page]
Model: if/kimi-k2-thinking (or any provider/model prefix)
-````
+```
-क्लाउड कोड, कोडेक्स सीएलआई, जेमिनी सीएलआई, कर्सर, क्लाइन, ओपनक्लाव, ओपनकोड और ओपनएआई-संगत एसडीके के साथ काम करता है।### 4) Enable and validate protocols (v2.0)
+Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs.
-**एमसीपी (टूल-संचालित संचालन के लिए):**```bash
+### 4) Enable and validate protocols (v2.0)
+
+**MCP (for tool-driven operations):**
+
+```bash
omniroute --mcp
+```
-````
-
-फिर अपने एमसीपी क्लाइंट को `stdio` पर कनेक्ट करें और टूल का परीक्षण करें जैसे:
+Then connect your MCP client over `stdio` and test tools like:
- `omniroute_get_health`
- `omniroute_list_combos`
-**A2A (एजेंट-टू-एजेंट वर्कफ़्लो के लिए):**```bash
+**A2A (for agent-to-agent workflows):**
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
```bash
curl -X POST http://localhost:20128/a2a \
@@ -759,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \
npm run test:protocols:e2e
```
-यह सुइट चल रहे ऐप के विरुद्ध वास्तविक MCP और A2A क्लाइंट प्रवाह को मान्य करता है।### Alternative: run from source
+This suite validates real MCP and A2A client flows against a running app.
+
+### Alternative: run from source
```bash
cp .env.example .env
@@ -767,13 +866,13 @@ npm install
PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev
```
-<विवरण>
-<सारांश>शून्य लिनक्स (`xbps-src` टेम्पलेट)सारांश>
+
+Void Linux (`xbps-src` template)
-Void Linux उपयोगकर्ताओं के लिए, आप `xbps-src` का उपयोग करके एक मूल पैकेज बना सकते हैं। इस ब्लॉक को `srcpkgs/omniroute/template` के रूप में सहेजें:```bash
+For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`:
+```bash
# Template file for 'omniroute'
-
pkgname=omniroute
version=3.4.1
revision=1
@@ -785,7 +884,7 @@ license="MIT"
homepage="https://github.com/diegosouzapw/OmniRoute"
distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz"
checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b
-system_accounts="\_omniroute"
+system_accounts="_omniroute"
omniroute_homedir="/var/lib/omniroute"
export NODE_ENV=production
export npm_config_engine_strict=false
@@ -793,71 +892,70 @@ export npm_config_loglevel=error
export npm_config_fund=false
export npm_config_audit=false
-do_build() { # Determine target CPU arch for node-gyp
-local \_gyp_arch
-case "$XBPS_TARGET_MACHINE" in
-aarch64*) \_gyp_arch=arm64 ;;
-armv7*|armv6*) \_gyp_arch=arm ;;
-i686*) \_gyp_arch=ia32 ;;
-\*) \_gyp_arch=x64 ;;
-esac
+do_build() {
+ # Determine target CPU arch for node-gyp
+ local _gyp_arch
+ case "$XBPS_TARGET_MACHINE" in
+ aarch64*) _gyp_arch=arm64 ;;
+ armv7*|armv6*) _gyp_arch=arm ;;
+ i686*) _gyp_arch=ia32 ;;
+ *) _gyp_arch=x64 ;;
+ esac
- # 1) Install all deps – skip scripts (no network in do_build, native modules
- # compiled separately below; better-sqlite3 is serverExternalPackage so
- # Next.js does not execute it during next build)
- NODE_ENV=development npm ci --ignore-scripts
+ # 1) Install all deps – skip scripts (no network in do_build, native modules
+ # compiled separately below; better-sqlite3 is serverExternalPackage so
+ # Next.js does not execute it during next build)
+ NODE_ENV=development npm ci --ignore-scripts
- # 2) Build the Next.js standalone bundle
- npm run build
+ # 2) Build the Next.js standalone bundle
+ npm run build
- # 3) Copy static assets into standalone
- cp -r .next/static .next/standalone/.next/static
- [ -d public ] && cp -r public .next/standalone/public || true
+ # 3) Copy static assets into standalone
+ cp -r .next/static .next/standalone/.next/static
+ [ -d public ] && cp -r public .next/standalone/public || true
- # 4) Compile better-sqlite3 native binding for the target architecture.
- # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
- # without npm altering them.
- local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
- (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
+ # 4) Compile better-sqlite3 native binding for the target architecture.
+ # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
+ # without npm altering them.
+ local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
+ (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
- # 5) Place the compiled binding into the standalone bundle
- local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
- mkdir -p "$_bs3_release"
- cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
+ # 5) Place the compiled binding into the standalone bundle
+ local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
+ mkdir -p "$_bs3_release"
+ cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
- # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
- # so sharp is not used at runtime; x64 .so files would break aarch64 strip
- rm -rf .next/standalone/node_modules/@img
-
- # 7) Copy pino runtime deps omitted by Next.js static analysis:
- # pino-abstract-transport – required by pino's worker thread
- # split2 – dep of pino-abstract-transport
- # process-warning – dep of pino itself
- for _mod in pino-abstract-transport split2 process-warning; do
- cp -r "node_modules/$_mod" .next/standalone/node_modules/
- done
+ # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
+ # so sharp is not used at runtime; x64 .so files would break aarch64 strip
+ rm -rf .next/standalone/node_modules/@img
+ # 7) Copy pino runtime deps omitted by Next.js static analysis:
+ # pino-abstract-transport – required by pino's worker thread
+ # split2 – dep of pino-abstract-transport
+ # process-warning – dep of pino itself
+ for _mod in pino-abstract-transport split2 process-warning; do
+ cp -r "node_modules/$_mod" .next/standalone/node_modules/
+ done
}
do_check() {
-npm run test:unit
+ npm run test:unit
}
do_install() {
-vmkdir usr/lib/omniroute/.next
+ vmkdir usr/lib/omniroute/.next
- vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
+ vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
- # Prevent removal of empty Next.js app router dirs by the post-install hook
- for _d in \
- .next/standalone/.next/server/app/dashboard \
- .next/standalone/.next/server/app/dashboard/settings \
- .next/standalone/.next/server/app/dashboard/providers; do
- touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
- done
-
- cat > "${WRKDIR}/omniroute" <<'EOF'
+ # Prevent removal of empty Next.js app router dirs by the post-install hook
+ for _d in \
+ .next/standalone/.next/server/app/dashboard \
+ .next/standalone/.next/server/app/dashboard/settings \
+ .next/standalone/.next/server/app/dashboard/providers; do
+ touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
+ done
+ cat > "${WRKDIR}/omniroute" <<'EOF'
#!/bin/sh
export PORT="${PORT:-20128}"
export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}"
@@ -869,10 +967,9 @@ EOF
}
post_install() {
-vlicense LICENSE
+ vlicense LICENSE
}
-
-````
+```
@@ -880,9 +977,11 @@ vlicense LICENSE
## 🐳 Docker
-ओमनीरूट [डॉकर हब](https://hub.docker.com/r/diegosouzapw/omniroute) पर सार्वजनिक डॉकर छवि के रूप में उपलब्ध है।
+OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
-**तेज़ भागना:**```bash
+**Quick run:**
+
+```bash
docker run -d \
--name omniroute \
--restart unless-stopped \
@@ -890,85 +989,96 @@ docker run -d \
-p 20128:20128 \
-v omniroute-data:/app/data \
diegosouzapw/omniroute:latest
-````
+```
-**पर्यावरण फ़ाइल के साथ:**```bash
+**With environment file:**
+```bash
# Copy and edit .env first
-
cp .env.example .env
docker run -d \
- --name omniroute \
- --restart unless-stopped \
- --stop-timeout 40 \
- --env-file .env \
- -p 20128:20128 \
- -v omniroute-data:/app/data \
- diegosouzapw/omniroute:latest
+ --name omniroute \
+ --restart unless-stopped \
+ --stop-timeout 40 \
+ --env-file .env \
+ -p 20128:20128 \
+ -v omniroute-data:/app/data \
+ diegosouzapw/omniroute:latest
+```
-````
+**Using Docker Compose:**
-**डॉकर कंपोज़ का उपयोग करना:**```bash
+```bash
# Base profile (no CLI tools)
docker compose --profile base up -d
# CLI profile (Claude Code, Codex, OpenClaw built-in)
docker compose --profile cli up -d
-````
+```
-डॉकर परिनियोजन के लिए डैशबोर्ड समर्थन में अब `डैशबोर्ड → एंडपॉइंट्स` पर एक-क्लिक**क्लाउडफ्लेयर क्विक टनल**शामिल है। पहला, जरूरत पड़ने पर ही `क्लाउडफ्लेयर` को डाउनलोड करने में सक्षम बनाता है, आपके वर्तमान `/v1` समापन बिंदु पर एक अस्थायी सुरंग शुरू करता है, और उत्पन्न `https://*.trycloudflare.com/v1` यूआरएल को सीधे आपके सामान्य सार्वजनिक यूआरएल के नीचे दिखाता है।
+Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL.
-टिप्पणियाँ:
+Notes:
-- क्विक टनल यूआरएल अस्थायी होते हैं और हर पुनरारंभ के बाद बदल जाते हैं।
-- ओम्निरूट या कंटेनर पुनरारंभ के बाद त्वरित सुरंगें स्वतः बहाल नहीं होती हैं। आवश्यकता पड़ने पर उन्हें डैशबोर्ड से पुनः सक्षम करें।
-- प्रबंधित इंस्टाल वर्तमान में `x64` / `arm64` पर Linux, macOS और Windows का समर्थन करता है।
-- प्रबंधित त्वरित सुरंगें प्रतिबंधित कंटेनर वातावरण में शोर वाले क्विक यूडीपी बफर चेतावनियों से बचने के लिए HTTP/2 परिवहन के लिए डिफ़ॉल्ट हैं। यदि आप एक अलग परिवहन चाहते हैं तो `CLOUDFLARED_PROTOCOL=quic` या `auto` सेट करें।
-- डॉकर छवियां सिस्टम सीए रूट्स को बंडल करती हैं और उन्हें प्रबंधित `क्लाउडफ्लेयर` में भेजती हैं, जो कंटेनर के अंदर सुरंग बूटस्ट्रैप होने पर टीएलएस ट्रस्ट विफलताओं से बचाती है।
-- SQLite वाल मोड में चलता है। `डॉकर स्टॉप` को समाप्त होने की अनुमति दी जानी चाहिए ताकि ओमनीरूट नवीनतम परिवर्तनों को `स्टोरेज.स्क्लाइट` में वापस चेकपॉइंट कर सके।
-- बंडल की गई कंपोज़ फ़ाइलें पहले से ही 40s स्टॉप ग्रेस अवधि निर्धारित करती हैं। यदि आप छवि को सीधे चलाते हैं, तो `--स्टॉप-टाइमआउट 40` (या समान) रखें ताकि मैन्युअल स्टॉप शटडाउन क्लीनअप में कटौती न करें।
-- यदि आप चाहते हैं कि ओमनीरूट डाउनलोड करने के बजाय मौजूदा बाइनरी का उपयोग करे तो `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` सेट करें।
+- Quick Tunnel URLs are temporary and change after every restart.
+- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed.
+- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`.
+- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport.
+- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container.
+- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`.
+- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup.
+- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one.
-**कैडी (HTTPS ऑटो-टीएलएस) के साथ डॉकर कंपोज़ का उपयोग करना:**
+**Using Docker Compose with Caddy (HTTPS Auto-TLS):**
-कैडी के स्वचालित एसएसएल प्रावधान का उपयोग करके ओमनीरूट को सुरक्षित रूप से उजागर किया जा सकता है। सुनिश्चित करें कि आपके डोमेन का DNS A रिकॉर्ड आपके सर्वर के आईपी को इंगित करता है।```yaml
+OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP.
+
+```yaml
services:
-omniroute:
-image: diegosouzapw/omniroute:latest
-container_name: omniroute
-restart: unless-stopped
-volumes: - omniroute-data:/app/data
-environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com
+ omniroute:
+ image: diegosouzapw/omniroute:latest
+ container_name: omniroute
+ restart: unless-stopped
+ volumes:
+ - omniroute-data:/app/data
+ environment:
+ - PORT=20128
+ - NEXT_PUBLIC_BASE_URL=https://your-domain.com
-caddy:
-image: caddy:latest
-container_name: caddy
-restart: unless-stopped
-ports: - "80:80" - "443:443"
-command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
+ caddy:
+ image: caddy:latest
+ container_name: caddy
+ restart: unless-stopped
+ ports:
+ - "80:80"
+ - "443:443"
+ command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
volumes:
-omniroute-data:
+ omniroute-data:
+```
-````
+| Image | Tag | Size | Description |
+| ------------------------ | -------- | ------ | --------------------- |
+| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release |
+| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version |
-| छवि | टैग | आकार | विवरण |
-| ---------------------- | -------- | ------ | ---------------------- |
-| `diegosouzapw/omniroute` | `नवीनतम` | ~250एमबी | नवीनतम स्थिर रिलीज़ |
-| `diegosouzapw/omniroute` | `1.0.3` | ~250एमबी | वर्तमान संस्करण |---
+---
## 🖥️ Desktop App — Offline & Always-On
-> 🆕**नया!**ओमनीरूट अब विंडोज, मैकओएस और लिनक्स के लिए**नेटिव डेस्कटॉप एप्लिकेशन**के रूप में उपलब्ध है।
+> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux.
-ओम्नीरूट को एक स्टैंडअलोन डेस्कटॉप ऐप के रूप में चलाएं - स्थानीय मॉडलों के लिए कोई टर्मिनल, कोई ब्राउज़र, कोई इंटरनेट आवश्यक नहीं है। इलेक्ट्रॉन-आधारित ऐप में शामिल हैं:
+Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes:
-- 🖥️**नेटिव विंडो**- सिस्टम ट्रे एकीकरण के साथ समर्पित ऐप विंडो
-- 🔄**ऑटो-स्टार्ट**— सिस्टम लॉगिन पर ओमनीरूट लॉन्च करें
-- 🔔**मूल सूचनाएं**- कोटा समाप्त होने या प्रदाता समस्याओं के लिए अलर्ट प्राप्त करें
-- ⚡**वन-क्लिक इंस्टॉल**- एनएसआईएस (विंडोज़), डीएमजी (मैकओएस), ऐपइमेज (लिनक्स)
-- 🌐**ऑफ़लाइन मोड**— बंडल सर्वर के साथ पूरी तरह ऑफ़लाइन काम करता है### त्वरित प्रारंभ
+- 🖥️ **Native Window** — Dedicated app window with system tray integration
+- 🔄 **Auto-Start** — Launch OmniRoute on system login
+- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues
+- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux)
+- 🌐 **Offline Mode** — Works fully offline with bundled server
+
+### त्वरित प्रारंभ
```bash
# Development mode
@@ -979,308 +1089,370 @@ npm run electron:build # Current platform
npm run electron:build:win # Windows (.exe)
npm run electron:build:mac # macOS (.dmg) — x64 & arm64
npm run electron:build:linux # Linux (.AppImage)
-````
+```
### System Tray
-न्यूनतम होने पर, ओमनीरूट त्वरित क्रियाओं के साथ आपके सिस्टम ट्रे में रहता है:
+When minimized, OmniRoute lives in your system tray with quick actions:
-- डैशबोर्ड खोलें
-- सर्वर पोर्ट बदलें
-- आवेदन छोड़ें
+- Open dashboard
+- Change server port
+- Quit application
-📖 पूर्ण दस्तावेज: [`electron/README.md`](electron/README.md)---
+📖 Full documentation: [`electron/README.md`](electron/README.md)
+
+---
## 💰 Pricing at a Glance
-| टियर | प्रदाता | लागत | कोटा रीसेट | के लिए सर्वश्रेष्ठ |
-| ----------------- | ---------------------------- | -------------------------------- | ---------------------- | -------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| **💳 सदस्यता** | क्लाउड कोड (प्रो) | $20/माह | 5 घंटे + साप्ताहिक | पहले ही सदस्यता ले ली है |
-| | कोडेक्स (प्लस/प्रो) | $20-200/महीना | 5 घंटे + साप्ताहिक | OpenAI उपयोगकर्ता |
-| | जेमिनी सीएलआई | **मुफ़्त** | 180K/माह + 1K/दिन | सब लोग! |
-| | गिटहब कोपायलट | $10-19/माह | मासिक | GitHub उपयोगकर्ता |
-| **🔑एपीआई कुंजी** | एनवीडिया एनआईएम | **मुफ़्त**(हमेशा के लिए देव) | ~40 आरपीएम | 70+ खुले मॉडल |
-| | सेरेब्रस | **मुफ़्त**(1 मिलियन टोकन/दिन) | 60K टीपीएम / 30 आरपीएम | दुनिया का सबसे तेज़ |
-| | ग्रोक | **मुफ़्त**(30 आरपीएम) | 14.4K आरपीडी | अल्ट्रा-फास्ट लामा/जेम्मा |
-| | डीपसीक V3.2 | $0.27/$1.10 प्रति 1 मिलियन | कोई नहीं | सर्वोत्तम मूल्य/गुणवत्ता तर्क |
-| | xAI ग्रोक-4 फास्ट | **$0.20/$0.50 प्रति 1 मिलियन**🆕 | कोई नहीं | सबसे तेज़ + टूल कॉलिंग, अल्ट्रालो |
-| | xAI ग्रोक-4 (मानक) | $0.20/$1.50 प्रति 1 मिलियन 🆕 | कोई नहीं | xAI से रीज़निंग फ्लैगशिप |
-| | मिस्ट्रल | नि:शुल्क परीक्षण + सशुल्क | दर सीमित | यूरोपीय एआई |
-| | ओपनराउटर | भुगतान-प्रति-उपयोग | कोई नहीं | 100+ मॉडल कुल मिलाकर। |
-| **💰सस्ता** | GLM-5 (Z.AI के माध्यम से) 🆕 | $0.5/1 मिलियन | प्रतिदिन सुबह 10 बजे | 128K आउटपुट, नवीनतम फ्लैगशिप |
-| | जीएलएम-4.7 | $0.6/1 मिलियन | प्रतिदिन सुबह 10 बजे | बजट बैकअप |
-| | मिनीमैक्स एम2.5 🆕 | $0.3/1M इनपुट | 5 घंटे की रोलिंग | तर्क + एजेंटिक कार्य |
-| | मिनीमैक्स एम2.1 | $0.2/1 मिलियन | 5 घंटे की रोलिंग | सबसे सस्ता विकल्प |
-| | किमी K2.5 (मूनशॉट एपीआई) 🆕 | भुगतान-प्रति-उपयोग | कोई नहीं | डायरेक्ट मूनशॉट एपीआई एक्सेस |
-| | किमी K2 | $9/महीना फ्लैट | 10एम टोकन/माह | अनुमानित लागत |
-| **🆓 मुफ़्त** | कोडर | **$0** | असीमित | 5 मॉडल असीमित |
-| | क्वेन | **$0** | असीमित | 4 मॉडल असीमित |
-| | किरो | **$0** | असीमित | क्लाउड सॉनेट/हाइकू (एडब्ल्यूएस बिल्डर) |
-| | लॉन्गकैट फ्लैश-लाइट 🆕 | **$0**(50 मिलियन टोकन/दिन 🔥) | 1 आरपीएस | पृथ्वी पर सबसे बड़ा मुफ़्त कोटा |
-| | परागण एआई 🆕 | **$0**(कोई कुंजी आवश्यक नहीं) | 1 अनुरोध/15s | जीपीटी-5, क्लाउड, डीपसीक, लामा 4 |
-| | क्लाउडफ्लेयर वर्कर्स एआई 🆕 | **$0**(10K न्यूरॉन्स/दिन) | ~150 सम्मान/दिन | 50+ मॉडल, वैश्विक बढ़त |
-| | स्केलवे एआई 🆕 | **$0**(कुल 1 मिलियन टोकन) | दर सीमित | ईयू/जीडीपीआर, क्वेन3 235बी, लामा 70बी | > 🆕**नए मॉडल जोड़े गए (मार्च 2026):**$0.20/$0.50/M पर ग्रोक-4 फास्ट परिवार (1143ms पर बेंचमार्क - जेमिनी 2.5 फ्लैश से 30% तेज), 128K आउटपुट के साथ Z.AI के माध्यम से GLM-5, मिनीमैक्स M2.5 रीजनिंग, डीपसीक V3.2 अद्यतन मूल्य निर्धारण, मूनशॉट डायरेक्ट एपीआई के माध्यम से किमी K2.5। |
+| Tier | Provider | Cost | Quota Reset | Best For |
+| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- |
+| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed |
+| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users |
+| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! |
+| | GitHub Copilot | $10-19/mo | Monthly | GitHub users |
+| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models |
+| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest |
+| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma |
+| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning |
+| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow |
+| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI |
+| | Mistral | Free trial + paid | Rate limited | European AI |
+| | OpenRouter | Pay-per-use | None | 100+ models aggr. |
+| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship |
+| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup |
+| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks |
+| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option |
+| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access |
+| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost |
+| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited |
+| | Qwen | **$0** | Unlimited | 4 models unlimited |
+| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) |
+| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth |
+| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 |
+| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge |
+| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B |
-**💡 $0 कॉम्बो स्टैक - पूर्ण निःशुल्क सेटअप:**```
+> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API.
+**💡 $0 Combo Stack — The Complete Free Setup:**
+
+```
# 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever
+Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
+Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
+Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
+Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
+NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+```
-Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
-Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
-LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
-Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
-Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
-Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
-Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
-Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
-Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
-NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
-Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever.
-````
-
-**शून्य लागत. कोडिंग कभी बंद नहीं होती।**इसे एक ओमनीरूट कॉम्बो के रूप में कॉन्फ़िगर करें और सभी फ़ॉलबैक स्वचालित रूप से होते हैं - कोई मैन्युअल स्विचिंग नहीं।---
+---
---
## 🆓 Free Models — What You Actually Get
-> नीचे दिए गए सभी मॉडल**शून्य क्रेडिट कार्ड की आवश्यकता के साथ 100% निःशुल्क**हैं। जब एक कोटा समाप्त हो जाता है तो ओमनीरूट उनके बीच ऑटो-रूट करता है - उन सभी को एक अटूट $0 कॉम्बो के लिए संयोजित करें।### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
+> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo.
-| मॉडल | उपसर्ग | सीमा | दर सीमा |
-| ------------------- | ------ | ----------------- | ---------------------- |
-| `क्लाउड-सॉनेट-4.5` | `क्र/` |**असीमित**| कोई रिपोर्ट नहीं की गई दैनिक सीमा |
-| `क्लाउड-हाइकु-4.5` | `क्र/` |**Unlimited**| कोई रिपोर्ट नहीं की गई दैनिक सीमा |
-| `क्लाउड-ओपस-4.6` | `क्र/` |**असीमित**| किरो के माध्यम से नवीनतम रचना |### 🟢 QODER MODELS (Free PAT via qodercli)
+### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
-| मॉडल | उपसर्ग | सीमा | दर सीमा |
-| ------------------ | ------ | ----------------- | --------------- |
-| `किमी-के2-सोच` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं |
-| `क्वेन3-कोडर-प्लस` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं |
-| `डीपसीक-आर1` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं |
-| `मिनीमैक्स-एम2.1` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं |
-| `किमी-के2` | `अगर/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं |
+| Model | Prefix | Limit | Rate Limit |
+| ------------------- | ------ | ------------- | --------------------- |
+| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro |
-> अनुशंसित कनेक्शन विधि:**पर्सनल एक्सेस टोकन + `qodercli`**। ब्राउज़र OAuth है
-> प्रयोगात्मक और डिफ़ॉल्ट रूप से अक्षम जब तक कि `QODER_OAUTH_*` पर्यावरण चर कॉन्फ़िगर नहीं किए जाते।### 🟡 QWEN MODELS (Device Code Auth)
+### 🟢 QODER MODELS (Free PAT via qodercli)
-| मॉडल | उपसर्ग | सीमा | दर सीमा |
-| ------------------- | ------ | ----------------- | ------------------- |
-| `क्वेन3-कोडर-प्लस` | `qw/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं |
-| `क्वेन3-कोडर-फ़्लैश` | `qw/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं |
-| `qwen3-कोडर-नेक्स्ट` | `qw/` |**असीमित**| कोई रिपोर्ट की गई सीमा नहीं |
-| `विज़न-मॉडल` | `qw/` |**असीमित**| मल्टीमॉडल (चित्र) |### 🟣 GEMINI CLI (Google OAuth)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------ | ------ | ------------- | --------------- |
+| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap |
+| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap |
+| `deepseek-r1` | `if/` | **Unlimited** | No reported cap |
+| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap |
+| `kimi-k2` | `if/` | **Unlimited** | No reported cap |
-| मॉडल | उपसर्ग | सीमा | Rate Limit |
-| ---------------------- | ------ | -------------------------------- | ----------------- |
-| `मिथुन-3-फ़्लैश-पूर्वावलोकन` | `जीसी/` |**180K टोकन/माह**+ 1K/दिन | मासिक रीसेट |
-| `मिथुन-2.5-प्रो` | `जीसी/` | 180K/माह (साझा पूल) | उच्च गुणवत्ता |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
+> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is
+> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured.
-| टियर | दैनिक सीमा | दर सीमा | नोट्स |
-| ---------- | ----------- | ----------- | ---------------------------------------------------------------- |
-| मुफ़्त (देव) | कोई टोकन सीमा नहीं |**~40 आरपीएम**| 70+ मॉडल; 2025 के मध्य में शुद्ध दर सीमा में परिवर्तन |
+### 🟡 QWEN MODELS (Device Code Auth)
-लोकप्रिय मुफ्त मॉडल: `मूनशोताई/किमी-के2.5` (किमी के2.5), `जेड-एआई/जीएलएम4.7` (जीएलएम 4.7), `डीपसीक-एआई/डीपसीक-वी3.2` (डीपसीक वी3.2), `एनवीडिया/ल्लामा-3.3-70बी-इंस्ट्रक्ट`, `डीपसीक/डीपसीक-आर1`### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------- | ------ | ------------- | ------------------- |
+| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap |
+| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) |
-| टियर | दैनिक सीमा | दर सीमा | नोट्स |
-| ---- | ----------------- | ---------------- | ------------------------------------------------ |
-| मुफ़्त |**1 मिलियन टोकन/दिन**| 60K टीपीएम / 30 आरपीएम | दुनिया का सबसे तेज़ एलएलएम अनुमान; प्रतिदिन रीसेट होता है |
+### 🟣 GEMINI CLI (Google OAuth)
-निःशुल्क उपलब्ध: `लामा-3.3-70बी`, `लामा-3.1-8बी`, `डीपसीक-आर1-डिस्टिल-लामा-70बी`### 🔴 GROQ (Free API Key — console.groq.com)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------------ | ------ | --------------------------- | ------------- |
+| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset |
+| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality |
-| टियर | दैनिक सीमा | दर सीमा | नोट्स |
-| ---- | ----------------- | ---------------- | ------------------------------------------------ |
-| मुफ़्त |**14.4K आरपीडी**| प्रति मॉडल 30 आरपीएम | कोई क्रेडिट कार्ड नहीं; 429 सीमा पर, शुल्क नहीं |
+### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
-निःशुल्क उपलब्ध: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---------- | ------------ | ----------- | ------------------------------------------------------ |
+| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 |
-| मॉडल | उपसर्ग | दैनिक निःशुल्क कोटा | नोट्स |
-| -------------------------------- | ------ | ----------------- | ---------------------- |
-| `लॉन्गकैट-फ्लैश-लाइट` | `एलसी/` |**50M टोकन**💥 | अब तक का सबसे बड़ा मुफ़्त कोटा |
-| 'लॉन्गकैट-फ्लैश-चैट' | `एलसी/` | 500K टोकन | मल्टी-टर्न चैट |
-| 'लॉन्गकैट-फ्लैश-थिंकिंग' | `एलसी/` | 500K टोकन | तर्क/सीओटी |
-| `लॉन्गकैट-फ्लैश-थिंकिंग-2601` | `एलसी/` | 500K टोकन | जनवरी 2026 संस्करण |
-| `लॉन्गकैट-फ्लैश-ओमनी-2603` | `एलसी/` | 500K टोकन | मल्टीमॉडल |
+Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`
-> सार्वजनिक बीटा में रहते हुए 100% निःशुल्क। ईमेल या फोन से [longcat.chat](https://longcat.chat) पर साइन अप करें। प्रतिदिन 00:00 UTC पर रीसेट होता है।### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
-| मॉडल | उपसर्ग | दर सीमा | पीछे प्रदाता |
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---- | ----------------- | ---------------- | ------------------------------------------- |
+| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily |
+
+Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`
+
+### 🔴 GROQ (Free API Key — console.groq.com)
+
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---- | ------------- | ---------------- | ----------------------------------------- |
+| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged |
+
+Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`
+
+### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+
+| Model | Prefix | Daily Free Quota | Notes |
+| ----------------------------- | ------ | ----------------- | ----------------------- |
+| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever |
+| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat |
+| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT |
+| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version |
+| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal |
+
+> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC.
+
+### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+
+| Model | Prefix | Rate Limit | Provider Behind |
| ---------- | ------ | ---------- | ------------------ |
-| 'ओपनाई' | `पोल/` | 1 अनुरोध/15s | जीपीटी-5 |
-| 'क्लाउड' | `पोल/` | 1 अनुरोध/15s | एंथ्रोपिक क्लाउड |
-| 'मिथुन' | `पोल/` | 1 अनुरोध/15s | गूगल जेमिनी |
-| 'डीपसीक' | `पोल/` | 1 अनुरोध/15s | डीपसीक वी3 |
-| 'लामा' | `पोल/` | 1 अनुरोध/15s | मेटा लामा 4 स्काउट |
-| 'मिस्ट्रल' | `पोल/` | 1 अनुरोध/15s | मिस्ट्रल एआई |
+| `openai` | `pol/` | 1 req/15s | GPT-5 |
+| `claude` | `pol/` | 1 req/15s | Anthropic Claude |
+| `gemini` | `pol/` | 1 req/15s | Google Gemini |
+| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 |
+| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout |
+| `mistral` | `pol/` | 1 req/15s | Mistral AI |
-> ✨**शून्य घर्षण:**कोई साइनअप नहीं, कोई एपीआई कुंजी नहीं। खाली कुंजी फ़ील्ड के साथ परागण प्रदाता जोड़ें और यह तुरंत काम करता है।### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately.
-| टियर | दैनिक न्यूरॉन्स | समतुल्य उपयोग | नोट्स |
-| ---- | ----------------- | ------------------------------------------------ | ---------------------- |
-| मुफ़्त |**10,000**| ~150 एलएलएम सम्मान / 500s ऑडियो / 15K एंबेड | वैश्विक बढ़त, 50+ मॉडल |
+### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
-लोकप्रिय मुफ़्त मॉडल: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (मुफ़्त ऑडियो!), `@cf/qwen/qwen2.5-coder-15b-instruct`
+| Tier | Daily Neurons | Equivalent Usage | Notes |
+| ---- | ------------- | --------------------------------------- | ----------------------- |
+| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models |
-> [dash.cloudflare.com](https://dash.cloudflare.com) से एपीआई टोकन + खाता आईडी की आवश्यकता है। प्रदाता सेटिंग्स में खाता आईडी संग्रहीत करें।### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct`
-| टियर | मुफ़्त कोटा | स्थान | नोट्स |
-| ---- | ----------------- | ----------- | -------------------------------------- |
-| मुफ़्त |**1M टोकन**| 🇫🇷पेरिस, ईयू | सीमा के भीतर किसी क्रेडिट कार्ड की आवश्यकता नहीं |
+> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings.
-निःशुल्क उपलब्ध: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `depseek-v3-0324`
+### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
-> ईयू/जीडीपीआर के अनुरूप। [console.scaleway.com](https://console.scaleway.com) पर एपीआई कुंजी प्राप्त करें।
+| Tier | Free Quota | Location | Notes |
+| ---- | ------------- | ------------ | ----------------------------------- |
+| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits |
->**💡 परम निःशुल्क स्टैक (11 प्रदाता, $0 हमेशा के लिए):**
+Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
+
+> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com).
+
+> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):**
>
> ```
-> किरो (केआर/) → क्लाउड सॉनेट/हाइकु अनलिमिटेड
-> कोडर (यदि/) → किमी-के2-सोच, क्वेन3-कोडर-प्लस, डीपसीक-आर1 अनलिमिटेड
-> लॉन्गकैट लाइट (एलसी/) → लॉन्गकैट-फ्लैश-लाइट - 50एम टोकन/दिन 🔥
-> परागण (पोल/) → जीपीटी-5, क्लाउड, डीपसीक, लामा 4 - किसी कुंजी की आवश्यकता नहीं
-> क्वेन (qw/) → क्वेन3-कोडर मॉडल असीमित
-> मिथुन (मिथुन /) → मिथुन 2.5 फ्लैश - 1,500 अनुरोध/दिन निःशुल्क
-> क्लाउडफ्लेयर एआई (सीएफ/) → 50+ मॉडल - 10K न्यूरॉन्स/दिन
-> स्केलवे (scw/) → Qwen3 235B, Llama 70B — 1M मुफ़्त टोकन (EU)
-> ग्रोक (ग्रोक/) → लामा/जेम्मा - 14.4K अनुरोध/दिन अल्ट्रा-फास्ट
-> एनवीडिया एनआईएम (एनवीडिया/) → 70+ खुले मॉडल - 40 आरपीएम हमेशा के लिए
-> सेरेब्रस (सेरेब्रस/) → लामा/क्वेन दुनिया का सबसे तेज़ - 1 मिलियन टोकन/दिन
-> ```## 🎙️ Free Transcription Combo
+> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+> Qwen (qw/) → qwen3-coder models UNLIMITED
+> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free
+> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day
+> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast
+> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+> ```
->**$0**के लिए किसी भी ऑडियो/वीडियो को ट्रांसक्राइब करें - डीपग्राम $200 मुफ़्त, असेंबलीएआई $50 फ़ॉलबैक, ग्रोक व्हिस्पर असीमित आपातकालीन बैकअप के साथ अग्रणी है।
+## 🎙️ Free Transcription Combo
-| प्रदाता | मुफ़्त क्रेडिट | सर्वश्रेष्ठ मॉडल | दर सीमा |
-| ----------------- | ---------------------- | ------------------------------------------------ | -------------------------------- |
-| 🟢**दीपग्राम**|**$200 निःशुल्क**(साइनअप) | `नोवा-3` — सर्वोत्तम सटीकता, 30+ भाषाएँ | निःशुल्क क्रेडिट पर कोई आरपीएम सीमा नहीं |
-| 🔵**असेंबलीएआई**|**$50 निःशुल्क**(साइनअप) | `यूनिवर्सल-3-प्रो` — अध्याय, भावना, पीआईआई | निःशुल्क क्रेडिट पर कोई आरपीएम सीमा नहीं |
-| 🔴**ग्रोक**|**हमेशा के लिए मुफ़्त**| `व्हिस्पर-लार्ज-v3` - ओपनएआई व्हिस्पर | 30 आरपीएम (दर सीमित) |
+> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup.
-**`/डैशबोर्ड/कॉम्बोस` में सुझाया गया कॉम्बो:**```
+| Provider | Free Credits | Best Model | Rate Limit |
+| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- |
+| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits |
+| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits |
+| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) |
+
+**Suggested combo in `/dashboard/combos`:**
+
+```
Name: free-transcription
Strategy: Priority
Nodes:
[1] deepgram/nova-3 → uses $200 free first
[2] assemblyai/universal-3-pro → fallback when Deepgram credits run out
[3] groq/whisper-large-v3 → free forever, emergency fallback
-````
+```
-फिर `/डैशबोर्ड/मीडिया` →**ट्रांसक्रिप्शन**टैब में: कोई भी ऑडियो या वीडियो फ़ाइल अपलोड करें → अपना कॉम्बो एंडपॉइंट चुनें → समर्थित प्रारूपों में ट्रांसक्रिप्शन प्राप्त करें।## 💡 Key Features
+Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats.
-ओमनीरूट v2.0 को केवल एक रिले प्रॉक्सी नहीं, बल्कि एक ऑपरेशनल प्लेटफॉर्म के रूप में बनाया गया है।### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
+## 💡 Key Features
-| फ़ीचर | यह क्या करता है |
-| ---------------------------------- | -------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ |
-| ⚡**ग्रोक-4 फास्ट फ़ैमिली** | $0.20/$0.50/M पर xAI मॉडल - बेंचमार्क 1143ms (जेमिनी 2.5 फ्लैश से 30% तेज) |
-| 🧠**Z.AI के माध्यम से GLM-5** | 128K आउटपुट संदर्भ, $0.5/1M - GLM परिवार का नवीनतम फ्लैगशिप |
-| 🔮**मिनीमैक्स एम2.5** | $0.30/1 मिलियन पर रीज़निंग + एजेंटिक कार्य - एम2.1 से महत्वपूर्ण उन्नयन |
-| 🎯**टूलकॉलिंग फ़्लैग प्रति मॉडल** | रजिस्ट्री में प्रति-मॉडल `टूलकॉलिंग: सही/गलत` - ऑटोकॉम्बो गैर-टूल-सक्षम मॉडल को छोड़ देता है |
-| 🌍**बहुभाषी आशय का पता लगाना** | ऑटोकॉम्बो स्कोरिंग में पीटी/जेडएच/ईएस/एआर कीवर्ड - गैर-अंग्रेजी सामग्री के लिए बेहतर मॉडल चयन |
-| 📊**बेंचमार्क-प्रेरित फ़ॉलबैक** | लाइव अनुरोधों से वास्तविक p95 विलंबता कॉम्बो स्कोरिंग फ़ीड करती है - ऑटोकॉम्बो वास्तविक डेटा से सीखता है |
-| 🔁**डुप्लीकेशन का अनुरोध** | कंटेंट-हैश आधारित डिडअप विंडो - मल्टी-एजेंट सुरक्षित, डुप्लिकेट शुल्क को रोकता है |
-| 🔌**प्लग करने योग्य राउटर रणनीति** | एक्स्टेंसिबल `राउटरस्ट्रैटेजी` इंटरफ़ेस - प्लगइन्स के रूप में कस्टम रूटिंग लॉजिक जोड़ें | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP |
+OmniRoute v3.5 is built as an operational platform, not just a relay proxy.
-| फ़ीचर | यह क्या करता है |
-| -------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------- |
-| 🎮**मॉडल खेल का मैदान** | किसी भी मॉडल का सीधे परीक्षण करने के लिए डैशबोर्ड पेज - प्रदाता/मॉडल/एंडपॉइंट चयनकर्ता, मोनाको संपादक, स्ट्रीमिंग, निरस्त, समय |
-| 🔏**सीएलआई फ़िंगरप्रिंट मिलान** | मूल सीएलआई हस्ताक्षरों से मिलान करने के लिए प्रति-प्रदाता हेडर/बॉडी ऑर्डरिंग - सेटिंग्स> सुरक्षा में प्रति प्रदाता टॉगल करें।**आपका प्रॉक्सी आईपी संरक्षित है** |
-| 🤝**एसीपी सपोर्ट (एजेंट क्लाइंट प्रोटोकॉल)** | सीएलआई एजेंट खोज (कोडेक्स, क्लाउड, गूज़, जेमिनी सीएलआई, ओपनक्लॉ + 9 अधिक), प्रोसेस स्पॉनर, `/api/acp/agents` एंडपॉइंट |
-| 🤖**एसीपी एजेंट्स डैशबोर्ड** | डीबग › एजेंट पेज - किसी भी सीएलआई टूल के लिए इंस्टॉल स्थिति, संस्करण, कस्टम एजेंट फॉर्म के साथ 14 एजेंटों का ग्रिड।**ओपनकोड**उपयोगकर्ताओं को एक "डाउनलोड ओपनकोड.जेसन" बटन मिलता है जो सभी उपलब्ध मॉडलों के साथ उपयोग के लिए तैयार कॉन्फ़िगरेशन को स्वतः उत्पन्न करता है। |
-| 🔧**कस्टम मॉडल `एपीआईफॉर्मेट` रूटिंग** | `apiFormat: "प्रतिक्रियाएं"` के साथ कस्टम मॉडल अब प्रतिक्रिया एपीआई अनुवादक पर सही ढंग से रूट करते हैं |
-| 🏢**कोडेक्स कार्यक्षेत्र अलगाव** | प्रति ईमेल एकाधिक कोडेक्स कार्यस्थान - OAuth कार्यस्थान आईडी द्वारा कनेक्शन को सही ढंग से अलग करता है |
-| 🔄**इलेक्ट्रॉन ऑटो-अपडेट** | डेस्कटॉप ऐप अपडेट की जांच करता है + रीस्टार्ट होने पर ऑटो-इंस्टॉल | ### 🤖 Agent & Protocol Operations (v2.0) |
+### 🆕 New — v3.5.5 Highlights (Apr 2026)
-| फ़ीचर | यह क्या करता है |
-| --------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- |
-| 🔧**एमसीपी सर्वर (25 उपकरण)** | 3 ट्रांसपोर्ट के माध्यम से आईडीई/एजेंट उपकरण: stdio, SSE (`/api/mcp/sse`), स्ट्रीम करने योग्य HTTP (`/api/mcp/stream`)। 18 कोर + 3 मेमोरी + 4 कौशल उपकरण |
-| 🤝**A2A सर्वर (JSON-RPC + SSE)** | सिंक और स्ट्रीमिंग प्रवाह के साथ एजेंट-टू-एजेंट कार्य निष्पादन |
-| 🧭**समेकित समापन बिंदु पृष्ठ** | एंडपॉइंट प्रॉक्सी, एमसीपी, ए2ए और एपीआई एंडपॉइंट टैब के साथ टैब्ड प्रबंधन पृष्ठ |
-| 🎚️**सेवा सक्षम/अक्षम टॉगल** | सेटिंग्स दृढ़ता के साथ एमसीपी और ए2ए के लिए चालू/बंद स्विच (डिफ़ॉल्ट: बंद) |
-| 🛰️**एमसीपी रनटाइम हार्टबीट** | वास्तविक प्रक्रिया स्थिति (पीआईडी, अपटाइम, दिल की धड़कन की उम्र, परिवहन, स्कोप मोड) |
-| 📋**एमसीपी ऑडिट ट्रेल** | सफलता/असफलता और मुख्य एट्रिब्यूशन के साथ फ़िल्टर करने योग्य ऑडिट लॉग |
-| 🔐**एमसीपी स्कोप प्रवर्तन** | नियंत्रित टूल एक्सेस के लिए 10 ग्रैन्युलर स्कोप अनुमतियाँ |
-| 📡**A2A कार्य जीवनचक्र प्रबंधन** | कार्यों को सूचीबद्ध करें/फ़िल्टर करें, घटनाओं/कलाकृतियों का निरीक्षण करें, चल रहे कार्यों को रद्द करें |
-| 📋**एजेंट कार्ड डिस्कवरी** | क्लाइंट ऑटो-डिस्कवरी के लिए `/.well-known/agent.json` |
-| 🧪**प्रोटोकॉल E2E टेस्ट हार्नेस** | वास्तविक MCP SDK + A2A क्लाइंट `test:protocols:e2e` | में प्रवाहित होता है |
-| ⚙️**परिचालन नियंत्रण** | कॉम्बो स्विच करें, लचीलापन प्रोफ़ाइल लागू करें, एक नियंत्रण सतह से ब्रेकर रीसेट करें | ### 🧠 Routing & Intelligence |
+| Feature | What It Does |
+| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation |
+| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs |
+| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions |
+| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection |
+| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts |
-| फ़ीचर | यह क्या करता है |
-| -------------------------------- | ------------------------------------------------------------------------- | ----------------------- |
-| 🎯**स्मार्ट 4-टियर फ़ॉलबैक** | Auto-route: Subscription → API Key → Cheap → Free |
-| 📊**वास्तविक समय कोटा ट्रैकिंग** | लाइव टोकन गिनती + प्रति प्रदाता रीसेट उलटी गिनती |
-| 🔄**प्रारूप अनुवाद** | OpenAI ↔ क्लाउड ↔ जेमिनी ↔ स्कीमा-सुरक्षित रूपांतरण के साथ प्रतिक्रियाएँ |
-| 👥**मल्टी-अकाउंट सपोर्ट** | बुद्धिमान चयन के साथ प्रति प्रदाता एकाधिक खाते |
-| 🔄**ऑटो टोकन रिफ्रेश** | OAuth टोकन पुनः प्रयास के साथ स्वचालित रूप से ताज़ा हो जाते हैं |
-| 🎨**कस्टम कॉम्बो** | 9 संतुलन रणनीतियाँ + फ़ॉलबैक श्रृंखला नियंत्रण |
-| 🌐**वाइल्डकार्ड राउटर** | `प्रदाता/*` डायनेमिक रूटिंग |
-| 🧠**सोच बजट नियंत्रण** | पासथ्रू, ऑटो, कस्टम और अनुकूली तर्क सीमाएँ |
-| 🔀**मॉडल उपनाम** | बिल्ट-इन + कस्टम मॉडल अलियासिंग और माइग्रेशन सुरक्षा |
-| ⚡**पृष्ठभूमि का क्षरण** | कम प्राथमिकता वाले पृष्ठभूमि कार्यों को सस्ते मॉडल पर रूट करें |
-| 🧪**टास्क-अवेयर स्मार्ट रूटिंग** | सामग्री प्रकार (कोडिंग/विज़न/विश्लेषण/सारांशीकरण) द्वारा स्वतः-चयन मॉडल |
-| 🔄**A2A एजेंट वर्कफ़्लोज़** | स्टेटफुल मल्टी-स्टेप एजेंट निष्पादन के लिए नियतात्मक एफएसएम ऑर्केस्ट्रेटर |
-| 🔀**अनुकूली रूटिंग** | टोकन वॉल्यूम और शीघ्र जटिलता के आधार पर गतिशील रणनीति ओवरराइड |
-| 🎲**प्रदाता विविधता** | शैनन एन्ट्रापी स्कोरिंग संतुलन ऑटो-कॉम्बो ट्रैफ़िक वितरण |
-| 💬**सिस्टम प्रॉम्प्ट इंजेक्शन** | वैश्विक व्यवहार नियंत्रण लगातार लागू |
-| 📄**प्रतिक्रियाएं एपीआई संगतता** | कोडेक्स और उन्नत एजेंटिक वर्कफ़्लोज़ के लिए पूर्ण `/v1/responses` समर्थन | ### 🎵 Multi-Modal APIs |
+### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
-| फ़ीचर | यह क्या करता है |
-| --------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- |
-| 🖼️**छवि निर्माण** | `/v1/images/जेनरेशन` क्लाउड और स्थानीय बैकएंड के साथ |
-| 📐**एंबेडिंग** | खोज और RAG पाइपलाइनों के लिए `/v1/embeddings` |
-| 🎤**ऑडियो ट्रांस्क्रिप्शन** | `/v1/ऑडियो/ट्रांसक्रिप्शन` - 7 प्रदाता (डीपग्राम नोवा 3, असेंबलीएआई, ग्रोक व्हिस्पर, हगिंगफेस, इलेवनलैब्स, ओपनएआई, एज़्योर), ऑटो-लैंग्वेज डिटेक्शन, एमपी4/एमपी3/डब्ल्यूएवी सपोर्ट |
-| 🔊**टेक्स्ट-टू-स्पीच** | `/v1/ऑडियो/स्पीच` - सही त्रुटि संदेशों के साथ 10 प्रदाता (इलेवनलैब्स, ओपनएआई, डीपग्राम, कार्टेसिया, प्लेएचटी, हगिंगफेस, एनवीडिया एनआईएम, इनवर्ल्ड, कोक्वी, टोर्टोइज़) |
-| 🎬**वीडियो जेनरेशन** | `/v1/वीडियो/पीढ़ी` (ComfyUI + SD WebUI वर्कफ़्लोज़) |
-| 🎵**संगीत पीढ़ी** | `/v1/संगीत/पीढ़ी` (ComfyUI वर्कफ़्लोज़) |
-| 🛡️**संयम** | `/v1/मॉडरेशन` सुरक्षा जांच |
-| 🔀**पुनर्रैंकिंग** | प्रासंगिकता स्कोरिंग के लिए `/v1/rerank` |
-| 🔍**वेब खोज**🆕 | `/v1/search` - 5 प्रदाता (सर्पर, ब्रेव, पर्प्लेक्सिटी, एक्सा, टैविली), 6,500+ मुफ़्त/माह, ऑटो-फ़ेलओवर, कैश | ### 🛡️ Resilience, Security & Governance |
+| Feature | What It Does |
+| ------------------------------------ | ------------------------------------------------------------------------------------------- |
+| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) |
+| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family |
+| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 |
+| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models |
+| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content |
+| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data |
+| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges |
+| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins |
-| फ़ीचर | यह क्या करता है |
-| ------------------------------------------ | ----------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------- |
-| 🔌**सर्किट तोड़ने वाले** | प्रति-मॉडल यात्रा/सीमा नियंत्रण के साथ पुनर्प्राप्ति |
-| 🎯**एंडपॉइंट-अवेयर मॉडल** | कस्टम मॉडल समर्थित एंडपॉइंट + एपीआई प्रारूप की घोषणा करते हैं |
-| 🛡️**एंटी-थंडरिंग झुंड** | पुनः प्रयास/दर घटनाओं पर म्यूटेक्स + सेमाफोर सुरक्षा |
-| 🧠**सिमेंटिक + सिग्नेचर कैश** | दो कैश परतों के साथ लागत/विलंबता में कमी |
-| ⚡**निष्क्रियता का अनुरोध** | डुप्लिकेट सुरक्षा विंडो |
-| 🔒**टीएलएस फ़िंगरप्रिंट स्पूफिंग** | ब्राउज़र जैसा टीएलएस फ़िंगरप्रिंट -**बॉट डिटेक्शन और अकाउंट फ़्लैगिंग को कम करता है** |
-| 🔏**सीएलआई फ़िंगरप्रिंट मिलान** | मूल सीएलआई अनुरोध हस्ताक्षरों से मेल खाता है -**प्रॉक्सी आईपी को संरक्षित करते हुए प्रतिबंध जोखिम को कम करता है** |
-| 🌐**आईपी फ़िल्टरिंग** | उजागर तैनाती के लिए अनुमति सूची/अवरुद्ध सूची नियंत्रण |
-| 📊**संपादन योग्य दर सीमाएँ** | दृढ़ता के साथ कॉन्फ़िगर करने योग्य वैश्विक/प्रदाता-स्तर की सीमाएं |
-| 📉**सौम्य पतन** | मल्टी-लेयर क्षमता फ़ॉलबैक कोर गेटवे ऑपरेशंस की सुरक्षा करती है |
-| 📜**कॉन्फिग ऑडिट ट्रेल** | डिफ-आधारित परिवर्तन ट्रैकिंग सरल रोलबैक के साथ परिचालन बहाव को रोकती है |
-| ⏳**प्रदाता स्वास्थ्य सिंक** | प्राधिकरण विफलताओं से पहले ट्रिगरिंग अलर्ट सक्रिय टोकन समाप्ति निगरानी |
-| 🚪**प्रतिबंधित खातों को स्वतः अक्षम करें** | ऑपरेशनल सर्किट ब्रेकर स्वचालित रूप से स्थायी रूप से ब्लॉक किए गए टोकन खातों को सील कर देता है |
-| 🔑**एपीआई कुंजी प्रबंधन + स्कोपिंग** | सुरक्षित कुंजी जारी करना/रोटेशन और मॉडल/प्रदाता नियंत्रण |
-| 👁️**स्कोप्ड एपीआई कुंजी का खुलासा**🆕 | `ALLOW_API_KEY_REVEAL` | के माध्यम से एपीआई कुंजियों की ऑप्ट-इन पुनर्प्राप्ति |
-| 🛡️**संरक्षित `/मॉडल`** | मॉडल कैटलॉग के लिए वैकल्पिक प्रमाणीकरण गेटिंग और प्रदाता छिपाना | ### 📊 Observability & Analytics |
+### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP
-| फ़ीचर | यह क्या करता है |
-| ---------------------------------- | ----------------------------------------------------------------------- | ---------------------------- |
-| 📝**अनुरोध + प्रॉक्सी लॉगिंग** | पूर्ण अनुरोध/प्रतिक्रिया और प्रॉक्सी लॉगिंग |
-| 📉**स्ट्रीम किए गए विस्तृत लॉग**🆕 | एसएसई पेलोड स्ट्रीम को यूआई में साफ-सुथरा तरीके से पुनर्निर्माण करता है |
-| 📋**एकीकृत लॉग डैशबोर्ड** | एक पृष्ठ में अनुरोध, प्रॉक्सी, ऑडिट और कंसोल दृश्य |
-| 🔍**टेलीमेट्री के लिए अनुरोध** | p50/p95/p99 विलंबता और अनुरोध अनुरेखण |
-| 🏥**स्वास्थ्य डैशबोर्ड** | अपटाइम, ब्रेकर स्थिति, लॉकआउट, कैश आँकड़े |
-| 💰**लागत ट्रैकिंग** | बजट नियंत्रण और प्रति-मॉडल मूल्य निर्धारण दृश्यता |
-| 📈**एनालिटिक्स विज़ुअलाइज़ेशन** | मॉडल/प्रदाता उपयोग अंतर्दृष्टि और रुझान दृश्य |
-| 🧪**मूल्यांकन ढाँचा** | विन्यास योग्य मिलान रणनीतियों के साथ गोल्डन सेट परीक्षण |
-| 📡**लाइव डायग्नोस्टिक्स**🆕 | सटीक कॉम्बो लाइव परीक्षण के लिए सिमेंटिक कैश बाईपास | ### ☁️ Deployment & Platform |
+| Feature | What It Does |
+| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing |
+| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** |
+| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint |
+| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. |
+| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator |
+| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID |
+| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart |
-| फ़ीचर | यह क्या करता है |
-| --------------------------------- | ----------------------------------------------------------------------------------- | ------------------------------------ |
-| 🌐**कहीं भी तैनात करें** | लोकलहोस्ट, वीपीएस, डॉकर, क्लाउड वातावरण |
-| 🚇**क्लाउडफ्लेयर टनल**🆕 | डैशबोर्ड से एक-क्लिक त्वरित सुरंग एकीकरण |
-| 🔑**एपीआई कुंजी मॉडल फ़िल्टरिंग** | मूल /v1/मॉडल प्रतिक्रिया निर्दिष्ट बियरर संदर्भ भूमिकाओं के माध्यम से फ़िल्टर की गई |
-| ⚡**स्मार्ट कैश बायपास** | कॉन्फ़िगर करने योग्य टीटीएल अनुमान और फ़ोर्स्ड रीफ़ेच नियंत्रण |
-| 🔄**बैकअप/पुनर्स्थापना** | निर्यात/आयात और आपदा पुनर्प्राप्ति प्रवाह |
-| 🧙**ऑनबोर्डिंग विज़ार्ड** | फर्स्ट-रन गाइडेड सेटअप |
-| 🔧**सीएलआई टूल्स डैशबोर्ड** | लोकप्रिय कोडिंग टूल के लिए एक-क्लिक सेटअप |
-| 🎮**मॉडल खेल का मैदान** | डैशबोर्ड से किसी भी प्रदाता/मॉडल/एंडपॉइंट का परीक्षण करें |
-| 🔏**सीएलआई फ़िंगरप्रिंट टॉगल** | सेटिंग्स > सुरक्षा | में प्रति प्रदाता फ़िंगरप्रिंट मिलान |
-| 🌐**i18n (30 भाषाएँ)** | पूर्ण डैशबोर्ड + आरटीएल कवरेज के साथ डॉक्स भाषा समर्थन |
-| 🧹**सभी मॉडल साफ़ करें** | प्रदाता विवरण में एक-क्लिक मॉडल सूची समाशोधन |
-| 👁️**साइडबार नियंत्रण**🆕 | उपस्थिति सेटिंग्स से घटकों और एकीकरणों को छुपाएं |
-| 📋**मुद्दा टेम्पलेट** | बग और सुविधाओं के लिए मानकीकृत GitHub टेम्पलेट |
-| 📂**कस्टम डेटा निर्देशिका** | भंडारण स्थान के लिए `DATA_DIR` ओवरराइड | ### Feature Deep Dive |
+### 🤖 Agent & Protocol Operations (v2.0)
+
+| Feature | What It Does |
+| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
+| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools |
+| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows |
+| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs |
+| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) |
+| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) |
+| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution |
+| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access |
+| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks |
+| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery |
+| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` |
+| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface |
+
+### 🧠 Routing & Intelligence
+
+| Feature | What It Does |
+| ---------------------------------- | ------------------------------------------------------------------------ |
+| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free |
+| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider |
+| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions |
+| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection |
+| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry |
+| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control |
+| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session |
+| 🌐 **Wildcard Router** | `provider/*` dynamic routing |
+| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits |
+| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety |
+| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models |
+| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) |
+| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions |
+| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity |
+| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution |
+| 💬 **System Prompt Injection** | Global behavior controls applied consistently |
+| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows |
+
+### 🎵 Multi-Modal APIs
+
+| Feature | What It Does |
+| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends |
+| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines |
+| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support |
+| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages |
+| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) |
+| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) |
+| 🛡️ **Moderations** | `/v1/moderations` safety checks |
+| 🔀 **Reranking** | `/v1/rerank` for relevance scoring |
+| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache |
+
+### 🛡️ Resilience, Security & Governance
+
+| Feature | What It Does |
+| ----------------------------------- | -------------------------------------------------------------------------------------- |
+| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls |
+| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format |
+| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events |
+| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers |
+| ⚡ **Request Idempotency** | Duplicate protection window |
+| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** |
+| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** |
+| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments |
+| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence |
+| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations |
+| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks |
+| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures |
+| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically |
+| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls |
+| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` |
+| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog |
+
+### 📊 Observability & Analytics
+
+| Feature | What It Does |
+| -------------------------------- | ----------------------------------------------------- |
+| 📝 **Request + Proxy Logging** | Full request/response and proxy logging |
+| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI |
+| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page |
+| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing |
+| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats |
+| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility |
+| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views |
+| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies |
+| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing |
+
+### ☁️ Deployment & Platform
+
+| Feature | What It Does |
+| ------------------------------ | --------------------------------------------------------------------- |
+| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments |
+| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard |
+| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles |
+| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls |
+| 🔄 **Backup/Restore** | Export/import and disaster recovery flows |
+| 🧙 **Onboarding Wizard** | First-run guided setup |
+| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools |
+| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard |
+| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security |
+| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage |
+| 🧹 **Clear All Models** | One-click model list clearing in provider details |
+| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings |
+| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features |
+| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location |
+
+### Feature Deep Dive
#### Smart fallback with practical cost control
@@ -1292,103 +1464,132 @@ Combo: "my-coding-stack"
4. if/kimi-k2-thinking
```
-जब कोटा, दर, या स्वास्थ्य विफल हो जाता है, तो ओमनीरूट मैन्युअल स्विचिंग के बिना स्वचालित रूप से अगले उम्मीदवार के पास चला जाता है।#### Protocol management that is visible and operable
+When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching.
-- MCP + A2A UI और डॉक्स में खोजने योग्य हैं (छिपा हुआ नहीं)
-- प्रोटोकॉल स्थिति एपीआई लाइव परिचालन डेटा को उजागर करते हैं (`/api/mcp/*`, `/api/a2a/*`)
-- डैशबोर्ड में दिन-2 ऑप्स के लिए क्रियाएं शामिल हैं (कॉम्बो टॉगल, ब्रेकर रीसेट, कार्य रद्द करना)#### Translator + validation workflow
+#### Protocol management that is visible and operable
-अनुवादक क्षेत्र में शामिल हैं:
+- MCP + A2A are discoverable in UI and docs (not hidden)
+- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`)
+- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation)
--**खेल का मैदान**: परिवर्तन जांच का अनुरोध करें -**चैट परीक्षक**: पूर्ण अनुरोध/प्रतिक्रिया राउंड-ट्रिप -**टेस्ट बेंच**: एक बार में कई मामले -**लाइव मॉनिटर**: वास्तविक समय यातायात दृश्य
+#### Translator + validation workflow
-साथ ही `npm run test:protocols:e2e` के माध्यम से वास्तविक ग्राहकों के साथ प्रोटोकॉल सत्यापन।
+The Translator area includes:
-> 📖**[एमसीपी सर्वर रीडमी](ओपन-एसएसई/एमसीपी-सर्वर/रीडमी.एमडी)**- टूल संदर्भ, आईडीई कॉन्फ़िगरेशन और क्लाइंट उदाहरण
+- **Playground**: request transformation checks
+- **Chat Tester**: full request/response round-trip
+- **Test Bench**: multiple cases in one run
+- **Live Monitor**: real-time traffic view
+
+Plus protocol validation with real clients via `npm run test:protocols:e2e`.
+
+> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples
>
-> 📖**[A2A सर्वर README](src/lib/a2a/README.md)**- कौशल, JSON-RPC विधियाँ, स्ट्रीमिंग, और कार्य जीवनचक्र## 🧪 Evaluations (Evals)
+> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle
-ओमनीरूट में गोल्डन सेट के मुकाबले एलएलएम प्रतिक्रिया गुणवत्ता का परीक्षण करने के लिए एक अंतर्निहित मूल्यांकन ढांचा शामिल है। डैशबोर्ड में**एनालिटिक्स → इवेल्स**के माध्यम से इसे एक्सेस करें।### Built-in Golden Set
+## 🧪 Evaluations (Evals)
-प्री-लोडेड "ओम्नीरूट गोल्डन सेट" में इसके लिए परीक्षण मामले शामिल हैं:
+OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard.
-- नमस्ते, गणित, भूगोल, कोड जनरेशन
-- JSON प्रारूप अनुपालन, अनुवाद, मार्कडाउन पीढ़ी
-- सुरक्षा इनकार (हानिकारक सामग्री), गिनती, बूलियन तर्क### Evaluation Strategies
+### Built-in Golden Set
-| रणनीति | विवरण | उदाहरण |
-| ---------- | ------------------------------------------------- | ------------------------------ | --- |
-| 'सटीक' | आउटपुट बिल्कुल मेल खाना चाहिए | `"4"` |
-| 'शामिल है' | आउटपुट में सबस्ट्रिंग (केस-असंवेदनशील) होना चाहिए | `"पेरिस"` |
-| 'रेगेक्स' | आउटपुट रेगेक्स पैटर्न से मेल खाना चाहिए | `"1.*2.*3"` |
-| `कस्टम` | कस्टम जेएस फ़ंक्शन सही/गलत लौटाता है | `(आउटपुट) => आउटपुट.लेंथ > 10` | --- |
+The pre-loaded "OmniRoute Golden Set" contains test cases for:
+
+- Greetings, math, geography, code generation
+- JSON format compliance, translation, markdown generation
+- Safety refusal (harmful content), counting, boolean logic
+
+### Evaluation Strategies
+
+| Strategy | Description | Example |
+| ---------- | ------------------------------------------------ | -------------------------------- |
+| `exact` | Output must match exactly | `"4"` |
+| `contains` | Output must contain substring (case-insensitive) | `"Paris"` |
+| `regex` | Output must match regex pattern | `"1.*2.*3"` |
+| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` |
+
+---
## 📖 Setup Guide
### Protocol Setup (MCP + A2A)
-<विवरण>
-<सारांश>🧩 एमसीपी सेटअप (मॉडल संदर्भ प्रोटोकॉल)सारांश>
+
+🧩 MCP Setup (Model Context Protocol)
-stdio मोड में MCP ट्रांसपोर्ट प्रारंभ करें:```bash
+Start MCP transport in stdio mode:
+
+```bash
omniroute --mcp
+```
-````
+Recommended validation flow:
-अनुशंसित सत्यापन प्रवाह:
+1. Connect your MCP client over stdio.
+2. Run `omniroute_get_health`.
+3. Run `omniroute_list_combos`.
+4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit.
-1. अपने MCP क्लाइंट को stdio से कनेक्ट करें।
-2. `omniroute_get_health` चलाएँ।
-3. `omniroute_list_combos` चलाएँ।
-4. दिल की धड़कन, गतिविधि और ऑडिट की पुष्टि करने के लिए `/डैशबोर्ड/एमसीपी` खोलें।
+Useful APIs for automation:
-स्वचालन के लिए उपयोगी एपीआई:
+- `GET /api/mcp/status`
+- `GET /api/mcp/tools`
+- `GET /api/mcp/audit`
+- `GET /api/mcp/audit/stats`
-- `प्राप्त करें /एपीआई/एमसीपी/स्थिति`
-- `प्राप्त करें /एपीआई/एमसीपी/टूल्स`
-- `प्राप्त करें /एपीआई/एमसीपी/ऑडिट`
-- `प्राप्त करें /api/mcp/ऑडिट/आँकड़े`
+
-<विवरण>
-<सारांश>🤝 A2A सेटअप (एजेंट2एजेंट)सारांश>
+
+🤝 A2A Setup (Agent2Agent)
-एजेंट का पता लगाएं:```bash
+Discover the agent:
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
-एक कार्य भेजें:```bash
+Send a task:
+
+```bash
curl -X POST http://localhost:20128/a2a \
- -H 'content-type: application/json' \
- -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+ -H 'content-type: application/json' \
+ -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+```
-````
+Manage lifecycle:
-जीवनचक्र प्रबंधित करें:
-
-- `प्राप्त करें /api/a2a/status`
-- `प्राप्त करें /api/a2a/कार्य`
-- `प्राप्त करें /api/a2a/tasks/:id`
+- `GET /api/a2a/status`
+- `GET /api/a2a/tasks`
+- `GET /api/a2a/tasks/:id`
- `POST /api/a2a/tasks/:id/cancel`
-परिचालन यूआई:
+Operational UI:
-- कार्य/स्थिति/स्ट्रीम अवलोकन और धूम्रपान क्रियाओं के लिए `/डैशबोर्ड/ए2ए`
+- `/dashboard/a2a` for task/state/stream observability and smoke actions
-<विवरण>
-<सारांश>🧪 एंड-टू-एंड प्रोटोकॉल सत्यापनसारांश>
+
-वास्तविक ग्राहकों के साथ दोनों प्रोटोकॉल मान्य करें:```bash
+
+🧪 End-to-end protocol validation
+
+Validate both protocols with real clients:
+
+```bash
npm run test:protocols:e2e
-````
+```
-यह सत्यापित करता है:
+This verifies:
-- एमसीपी एसडीके क्लाइंट कनेक्ट/लिस्ट/कॉल
-- A2A खोज/भेजें/स्ट्रीम/प्राप्त करें/रद्द करें
-- एमसीपी ऑडिट और ए2ए कार्य प्रबंधन एपीआई में डेटा को क्रॉस-चेक करें
+- MCP SDK client connect/list/call
+- A2A discovery/send/stream/get/cancel
+- Cross-check data in MCP audit and A2A task management APIs
-<विवरण>
-<सारांश>💳 सदस्यता प्रदातासारांश>### Claude Code (Pro/Max)
+
+
+
+💳 Subscription Providers
+
+### Claude Code (Pro/Max)
```bash
Dashboard → Providers → Connect Claude Code
@@ -1401,7 +1602,9 @@ Models:
cc/claude-haiku-4-5-20251001
```
-**प्रो टिप:**जटिल कार्यों के लिए ओपस और गति के लिए सॉनेट का उपयोग करें। ओमनीरूट प्रति मॉडल कोटा ट्रैक करता है!### OpenAI Codex (Plus/Pro)
+**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model!
+
+### OpenAI Codex (Plus/Pro)
```bash
Dashboard → Providers → Connect Codex
@@ -1415,20 +1618,22 @@ Models:
#### Codex Account Limit Management (5h + Weekly)
-प्रत्येक कोडेक्स खाते में अब `डैशबोर्ड -> प्रदाता` में नीति टॉगल हैं:
+Each Codex account now has policy toggles in `Dashboard -> Providers`:
-- `5 घंटे` (चालू/बंद): 5 घंटे की विंडो सीमा नीति लागू करें।
-- `साप्ताहिक` (चालू/बंद): साप्ताहिक विंडो सीमा नीति लागू करें।
-- थ्रेसहोल्ड व्यवहार: जब एक सक्षम विंडो >=90% उपयोग तक पहुंच जाती है, तो वह खाता छोड़ दिया जाता है।
-- रोटेशन व्यवहार: ओमनीरूट स्वचालित रूप से अगले पात्र कोडेक्स खाते पर रूट करता है।
-- रीसेट व्यवहार: जब प्रदाता का `resetAt` समय बीत जाता है, तो खाता स्वचालित रूप से फिर से पात्र हो जाता है।
+- `5h` (ON/OFF): enforce the 5-hour window threshold policy.
+- `Weekly` (ON/OFF): enforce the weekly window threshold policy.
+- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped.
+- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically.
+- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically.
-परिदृश्य:
+Scenarios:
-- `5 घंटे चालू` + `साप्ताहिक चालू`: जब कोई भी विंडो सीमा तक पहुंचती है तो खाता छोड़ दिया जाता है।
-- `5 घंटे की छूट` + `साप्ताहिक चालू`: केवल साप्ताहिक उपयोग ही खाते को ब्लॉक कर सकता है।
-- `5 घंटे चालू` + `साप्ताहिक बंद`: केवल 5 घंटे का उपयोग ही खाते को ब्लॉक कर सकता है।
-- `resetAt` पारित: खाता स्वचालित रूप से रोटेशन में पुनः प्रवेश करता है (कोई मैन्युअल पुनः सक्षम नहीं)।### Gemini CLI (FREE 180K/month!)
+- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold.
+- `5h OFF` + `Weekly ON`: only weekly usage can block the account.
+- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account.
+- `resetAt` passed: account re-enters rotation automatically (no manual re-enable).
+
+### Gemini CLI (FREE 180K/month!)
```bash
Dashboard → Providers → Connect Gemini CLI
@@ -1440,7 +1645,9 @@ Models:
gc/gemini-2.5-pro
```
-**सर्वोत्तम मूल्य:**विशाल निःशुल्क स्तर! सशुल्क स्तरों से पहले इसका उपयोग करें।### GitHub Copilot
+**Best Value:** Huge free tier! Use this before paid tiers.
+
+### GitHub Copilot
```bash
Dashboard → Providers → Connect GitHub
@@ -1455,71 +1662,91 @@ Models:
-<विवरण>
-<सारांश>🔑 एपीआई कुंजी प्रदातासारांश>### NVIDIA NIM (FREE developer access — 70+ models)
+
+🔑 API Key Providers
-1. साइन अप करें: [build.nvidia.com](https://build.nvidia.com)
-2. निःशुल्क एपीआई कुंजी प्राप्त करें (1000 अनुमान क्रेडिट शामिल)
-3. डैशबोर्ड → प्रदाता जोड़ें → एनवीडिया एनआईएम:
- - एपीआई कुंजी: `nvapi-your-key`
+### NVIDIA NIM (FREE developer access — 70+ models)
-**मॉडल:**`एनवीडिया/लामा-3.3-70बी-इंस्ट्रक्ट`, `एनवीडिया/मिस्ट्रल-7बी-इंस्ट्रक्ट`, और 50+ अधिक
+1. Sign up: [build.nvidia.com](https://build.nvidia.com)
+2. Get free API key (1000 inference credits included)
+3. Dashboard → Add Provider → NVIDIA NIM:
+ - API Key: `nvapi-your-key`
-**प्रो टिप:**ओपनएआई-संगत एपीआई - ओमनीरूट के प्रारूप अनुवाद के साथ सहजता से काम करता है!### DeepSeek
+**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more
-1. साइन अप करें: [प्लेटफ़ॉर्म.डीपसीक.कॉम](https://platform.डीपसीक.कॉम)
-2. एपीआई कुंजी प्राप्त करें
-3. डैशबोर्ड → प्रदाता जोड़ें → डीपसीक
+**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation!
-**मॉडल:**`डीपसीक/डीपसीक-चैट`, `डीपसीक/डीपसीक-कोडर`### Groq (Free Tier Available!)
+### DeepSeek
-1. साइन अप करें: [console.groq.com](https://console.groq.com)
-2. एपीआई कुंजी प्राप्त करें (फ्री टियर शामिल)
-3. डैशबोर्ड → प्रदाता जोड़ें → ग्रोक
+1. Sign up: [platform.deepseek.com](https://platform.deepseek.com)
+2. Get API key
+3. Dashboard → Add Provider → DeepSeek
-**मॉडल:**`groq/llama-3.3-70b`, `groq/mixtral-8x7b`
+**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder`
-**प्रो टिप:**अल्ट्रा-फास्ट अनुमान - वास्तविक समय कोडिंग के लिए सर्वोत्तम!### OpenRouter (100+ Models)
+### Groq (Free Tier Available!)
-1. साइन अप करें: [openrouter.ai](https://openrouter.ai)
-2. एपीआई कुंजी प्राप्त करें
-3. डैशबोर्ड → प्रदाता जोड़ें → ओपनराउटर
+1. Sign up: [console.groq.com](https://console.groq.com)
+2. Get API key (free tier included)
+3. Dashboard → Add Provider → Groq
-**मॉडल:**एक ही एपीआई कुंजी के माध्यम से सभी प्रमुख प्रदाताओं से 100+ मॉडल तक पहुंचें।
+**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b`
-**डैशबोर्ड व्यवहार:**ओपनराउटर मॉडल**उपलब्ध मॉडल**से प्रबंधित किए जाते हैं। मैन्युअल ऐड, आयात और ऑटो-सिंक सभी एक ही सूची को अपडेट करते हैं।
+**Pro Tip:** Ultra-fast inference — best for real-time coding!
-<विवरण>
-<सारांश>💰 सस्ते प्रदाता (बैकअप)सारांश>### GLM-4.7 (Daily reset, $0.6/1M)
+### OpenRouter (100+ Models)
-1. साइन अप करें: [झिपु एआई](https://open.bigmodel.cn/)
-2. कोडिंग योजना से एपीआई कुंजी प्राप्त करें
-3. डैशबोर्ड → एपीआई कुंजी जोड़ें:
- - प्रदाता: `glm`
- - एपीआई कुंजी: `आपकी-कुंजी`
+1. Sign up: [openrouter.ai](https://openrouter.ai)
+2. Get API key
+3. Dashboard → Add Provider → OpenRouter
-**उपयोग:**`glm/glm-4.7`
+**Models:** Access 100+ models from all major providers through a single API key.
-**प्रो टिप:**कोडिंग प्लान 1/7 लागत पर 3× कोटा प्रदान करता है! प्रतिदिन सुबह 10:00 बजे रीसेट करें।### MiniMax M2.1 (5h reset, $0.20/1M)
+**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list.
-1. साइन अप करें: [मिनीमैक्स](https://www.minimax.io/)
-2. एपीआई कुंजी प्राप्त करें
-3. डैशबोर्ड → एपीआई कुंजी जोड़ें
+
-**उपयोग करें:**`मिनीमैक्स/मिनीमैक्स-एम2.1`
+
+💰 Cheap Providers (Backup)
-**Pro Tip:**Cheapest option for long context (1M tokens)!### Kimi K2 ($9/month flat)
+### GLM-4.7 (Daily reset, $0.6/1M)
-1. सदस्यता लें: [मूनशॉट एआई](https://platform.moonshot.ai/)
-2. एपीआई कुंजी प्राप्त करें
-3. डैशबोर्ड → एपीआई कुंजी जोड़ें
+1. Sign up: [Zhipu AI](https://open.bigmodel.cn/)
+2. Get API key from Coding Plan
+3. Dashboard → Add API Key:
+ - Provider: `glm`
+ - API Key: `your-key`
-**उपयोग करें:**`किमी/किमी-नवीनतम`
+**Use:** `glm/glm-4.7`
-**प्रो टिप:**10एम टोकन के लिए निश्चित $9/माह = $0.90/1एम प्रभावी लागत!
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM.
-<विवरण>
-<सारांश>🆓 मुफ़्त प्रदाता (आपातकालीन बैकअप)सारांश>### Qoder (5 FREE models via OAuth)
+### MiniMax M2.1 (5h reset, $0.20/1M)
+
+1. Sign up: [MiniMax](https://www.minimax.io/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `minimax/MiniMax-M2.1`
+
+**Pro Tip:** Cheapest option for long context (1M tokens)!
+
+### Kimi K2 ($9/month flat)
+
+1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `kimi/kimi-latest`
+
+**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost!
+
+
+
+
+🆓 FREE Providers (Emergency Backup)
+
+### Qoder (5 FREE models via OAuth)
```bash
Dashboard → Connect Qoder
@@ -1560,8 +1787,10 @@ Models:
-<विवरण>
-<सारांश>🎨 कॉम्बो बनाएंसारांश>### Example 1: Maximize Subscription → Cheap Backup
+
+🎨 Create Combos
+
+### Example 1: Maximize Subscription → Cheap Backup
```
Dashboard → Combos → Create New
@@ -1589,8 +1818,10 @@ Cost: $0 forever!
-<विवरण>
-<सारांश>🔧 सीएलआई एकीकरणसारांश>### Cursor IDE
+
+🔧 CLI Integration
+
+### Cursor IDE
```
Settings → Models → Advanced:
@@ -1601,7 +1832,9 @@ Settings → Models → Advanced:
### Claude Code
-एक-क्लिक कॉन्फ़िगरेशन के लिए डैशबोर्ड में**सीएलआई टूल्स**पृष्ठ का उपयोग करें, या `~/.claude/settings.json` को मैन्युअल रूप से संपादित करें।### Codex CLI
+Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually.
+
+### Codex CLI
```bash
export OPENAI_BASE_URL="http://localhost:20128"
@@ -1612,12 +1845,15 @@ codex "your prompt"
### OpenClaw
-**विकल्प 1 - डैशबोर्ड (अनुशंसित):**```
+**Option 1 — Dashboard (recommended):**
+
+```
Dashboard → CLI Tools → OpenClaw → Select Model → Apply
+```
-````
+**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`:
-**विकल्प 2 - मैनुअल:**`~/.openclaw/openclaw.json` संपादित करें:```json
+```json
{
"models": {
"providers": {
@@ -1629,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply
}
}
}
-````
+```
-> **ध्यान दें:**ओपनक्लाव केवल स्थानीय ओमनीरूट के साथ काम करता है। IPv6 रिज़ॉल्यूशन समस्याओं से बचने के लिए `लोकलहोस्ट` के बजाय `127.0.0.1` का उपयोग करें।### Cline / Continue / RooCode
+> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues.
+
+### Cline / Continue / RooCode
```
Settings → API Configuration:
@@ -1643,15 +1881,17 @@ Settings → API Configuration:
### OpenCode
-**चरण 1:**एक कस्टम प्रदाता के रूप में ओम्निरूट जोड़ें:```bash
+**Step 1:** Add OmniRoute as a custom provider:
+
+```bash
opencode
/connect
-
# Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key
+```
-````
+**Step 2:** Create/edit `opencode.json` in your project root:
-**चरण 2:**अपने प्रोजेक्ट रूट में `opencode.json` बनाएं/संपादित करें:```json
+```json
{
"$schema": "https://opencode.ai/config.json",
"provider": {
@@ -1669,118 +1909,130 @@ opencode
}
}
}
-````
+```
-**चरण 3:**ओपनकोड में मॉडल का चयन करें:```bash
+**Step 3:** Select the model in OpenCode:
+
+```bash
/models
-
# Select any OmniRoute model from the list
+```
-````
+> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard.
->**टिप:**अपने ओमनीरूट `/v1/models` एंडपॉइंट में उपलब्ध किसी भी मॉडल को `मॉडल` अनुभाग में जोड़ें। अपने ओमनीरूट डैशबोर्ड से `प्रदाता/मॉडल-आईडी` प्रारूप का उपयोग करें।
+
---
## समस्या निवारण
-<विवरण>
-<सारांश>समस्या निवारण मार्गदर्शिका का विस्तार करने के लिए क्लिक करेंसारांश>
+
+Click to expand troubleshooting guide
-**"भाषा मॉडल ने संदेश प्रदान नहीं किया"**
+**"Language model did not provide messages"**
-- प्रदाता कोटा समाप्त → डैशबोर्ड कोटा ट्रैकर की जाँच करें
-- समाधान: कॉम्बो फ़ॉलबैक का उपयोग करें या सस्ते स्तर पर स्विच करें
+- Provider quota exhausted → Check dashboard quota tracker
+- Solution: Use combo fallback or switch to cheaper tier
-**दर सीमित करना**
+**Rate limiting**
-- सदस्यता कोटा ख़त्म → GLM/MiniMax पर फ़ॉलबैक
-- कॉम्बो जोड़ें: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Subscription quota out → Fallback to GLM/MiniMax
+- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-**OAuth टोकन समाप्त हो गया**
+**OAuth token expired**
-- ओम्निरूट द्वारा स्वतः ताज़ा
-- यदि समस्या बनी रहती है: डैशबोर्ड → प्रदाता → पुनः कनेक्ट करें
+- Auto-refreshed by OmniRoute
+- If issues persist: Dashboard → Provider → Reconnect
-**उच्च लागत**
+**High costs**
-- डैशबोर्ड → लागत में उपयोग के आँकड़े जाँचें
-- प्राथमिक मॉडल को जीएलएम/मिनीमैक्स पर स्विच करें
-- गैर-महत्वपूर्ण कार्यों के लिए फ्री टियर (मिथुन सीएलआई, क्यूडर) का उपयोग करें
+- Check usage stats in Dashboard → Costs
+- Switch primary model to GLM/MiniMax
+- Use free tier (Gemini CLI, Qoder) for non-critical tasks
-**डैशबोर्ड/एपीआई पोर्ट गलत हैं**
+**Dashboard/API ports are wrong**
-- `पोर्ट` कैनोनिकल बेस पोर्ट है (और डिफ़ॉल्ट रूप से एपीआई पोर्ट)
-- `API_PORT` केवल OpenAI-संगत API श्रोता को ओवरराइड करता है
-- `DASHBOARD_PORT` केवल डैशबोर्ड/Next.js श्रोता को ओवरराइड करता है
-- अपने डैशबोर्ड/सार्वजनिक URL पर `NEXT_PUBLIC_BASE_URL` सेट करें (OAuth कॉलबैक के लिए)
+- `PORT` is the canonical base port (and API port by default)
+- `API_PORT` overrides only OpenAI-compatible API listener
+- `DASHBOARD_PORT` overrides only dashboard/Next.js listener
+- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks)
-**क्लाउड सिंक त्रुटियाँ**
+**Cloud sync errors**
-- अपने चल रहे उदाहरण के लिए `BASE_URL` बिंदुओं को सत्यापित करें
-- अपने अपेक्षित क्लाउड एंडपॉइंट पर `CLOUD_URL` बिंदुओं को सत्यापित करें
-- `NEXT_PUBLIC_*` मानों को सर्वर-साइड मानों के साथ संरेखित रखें
+- Verify `BASE_URL` points to your running instance
+- Verify `CLOUD_URL` points to your expected cloud endpoint
+- Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**पहला लॉगिन काम नहीं कर रहा**
+**First login not working**
-- `.env` में `INITIAL_PASSWORD` जांचें
-- यदि सेट नहीं है, तो फ़ॉलबैक पासवर्ड `123456` है
+- Check `INITIAL_PASSWORD` in `.env`
+- If unset, fallback password is `123456`
-**कोई अनुरोध लॉग नहीं**
+**No request logs**
-- अनुरोध कलाकृतियों को प्रति अनुरोध एक JSON फ़ाइल के रूप में `DATA_DIR/call_logs/` पर लिखा जाता है
-- यदि आपको विस्तृत प्रति-स्टेज पेलोड की आवश्यकता है तो डैशबोर्ड → लॉग → अनुरोध लॉग से पाइपलाइन कैप्चर सक्षम करें
-- यदि आप `logs/application/app.log` में एप्लिकेशन कंसोल लॉग भी चाहते हैं तो `APP_LOG_TO_FILE=true` सेट करें
-- `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, और `CALL_LOG_MAX_ENTRIES` को आवश्यकतानुसार समायोजित करें
+- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request
+- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads
+- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log`
+- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed
-**कनेक्शन परीक्षण OpenAI-संगत प्रदाताओं के लिए "अमान्य" दिखाता है**
+**Connection test shows "Invalid" for OpenAI-compatible providers**
-- कई प्रदाता `/मॉडल` समापन बिंदु को उजागर नहीं करते हैं
-- ओमनीरूट v1.0.6+ में चैट पूर्णता के माध्यम से फ़ॉलबैक सत्यापन शामिल है
-- सुनिश्चित करें कि आधार URL में `/v1` प्रत्यय शामिल है### 🔐 OAuth on a Remote Server
+- Many providers don't expose a `/models` endpoint
+- OmniRoute v1.0.6+ includes fallback validation via chat completions
+- Ensure base URL includes `/v1` suffix
-
-
+### 🔐 OAuth on a Remote Server
->**⚠️ वीपीएस, डॉकर या किसी रिमोट सर्वर पर ओमनीरूट चलाने वाले उपयोगकर्ताओं के लिए महत्वपूर्ण**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
+
+
-**एंटीग्रेविटी**और**जेमिनी सीएलआई**प्रदाता**Google OAuth 2.0**का उपयोग करते हैं। Google को ऐप के Google क्लाउड कंसोल में पूर्व-पंजीकृत यूआरआई में से एक से सटीक मिलान करने के लिए OAuth प्रवाह में `redirect_uri` की आवश्यकता है।
+> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server**
-ओम्निरूट में बंडल किए गए OAuth क्रेडेंशियल**केवल `लोकलहोस्ट`* के लिए पंजीकृत हैं। जब आप किसी दूरस्थ सर्वर (उदाहरण के लिए `https://omniroute.myserver.com`) पर ओमनीरूट एक्सेस करते हैं, तो Google प्रमाणीकरण को अस्वीकार कर देता है:```
+#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
+
+The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console.
+
+The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solution: Configure your own OAuth credentials
-आपको अपने सर्वर के यूआरआई के साथ Google क्लाउड कंसोल में एक**OAuth 2.0 क्लाइंट आईडी**बनाना होगा।#### Step-by-step
+You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI.
-**1. Google क्लाउड कंसोल खोलें**
+#### Step-by-step
-यहां जाएं: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
+**1. Open Google Cloud Console**
-**2. एक नया OAuth 2.0 क्लाइंट आईडी बनाएं**
+Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
--**"+ क्रेडेंशियल बनाएं"**→**"OAuth क्लाइंट आईडी"**पर क्लिक करें
+**2. Create a new OAuth 2.0 Client ID**
-- एप्लिकेशन प्रकार:**"वेब एप्लिकेशन"**
-- नाम: कुछ भी जो आपको पसंद हो (जैसे `ओम्नीरूट रिमोट`)
+- Click **"+ Create Credentials"** → **"OAuth client ID"**
+- Application type: **"Web application"**
+- Name: anything you like (e.g. `OmniRoute Remote`)
-**3. अधिकृत रीडायरेक्ट यूआरआई जोड़ें**
+**3. Add Authorized Redirect URIs**
-**"अधिकृत रीडायरेक्ट यूआरआई"**फ़ील्ड में, जोड़ें:```
+In the **"Authorized redirect URIs"** field, add:
+
+```
https://your-server.com/callback
+```
-````
+> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`).
-> `your-server.com` को अपने सर्वर के डोमेन या आईपी से बदलें (यदि आवश्यक हो तो पोर्ट शामिल करें, उदाहरण के लिए `http://45.33.32.156:20128/callback`)।
+**4. Save and copy the credentials**
-**4. क्रेडेंशियल सहेजें और कॉपी करें**
+After creating, Google will show the **Client ID** and **Client Secret**.
-बनाने के बाद, Google**क्लाइंट आईडी**और**क्लाइंट सीक्रेट**दिखाएगा।
+**5. Set environment variables**
-**5. पर्यावरण चर सेट करें**
+In your `.env` (or Docker environment variables):
-आपके `.env` (या डॉकर पर्यावरण चर) में:```bash
+```bash
# For Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
@@ -1789,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
-````
+```
-**6. ओम्निरूट को पुनरारंभ करें**```bash
+**6. Restart OmniRoute**
+```bash
# npm:
-
npm run dev
# Docker:
-
docker restart omniroute
+```
-````
+**7. Try connecting again**
-**7. पुनः कनेक्ट करने का प्रयास करें**
+Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth
-डैशबोर्ड → प्रदाता → एंटीग्रेविटी (या जेमिनी सीएलआई) → OAuth
+Google will now redirect correctly to `https://your-server.com/callback`.
-Google अब `https://your-server.com/callback` पर सही ढंग से रीडायरेक्ट करेगा।---
+---
#### Temporary workaround (without custom credentials)
-यदि आप अभी अपना स्वयं का क्रेडेंशियल सेट नहीं करना चाहते हैं, तो आप अभी भी**मैन्युअल यूआरएल प्रवाह**का उपयोग कर सकते हैं:
+If you don't want to set up your own credentials right now, you can still use the **manual URL flow**:
-1. ओमनीरूट Google प्राधिकरण URL खोलता है
-2. अधिकृत करने के बाद, Google `लोकलहोस्ट` पर रीडायरेक्ट करने का प्रयास करता है (जो रिमोट सर्वर पर विफल रहता है)
-3.**अपने ब्राउज़र के एड्रेस बार से पूरा यूआरएल कॉपी करें**(भले ही पेज लोड न हो)
-4. उस यूआरएल को ओमनीरूट कनेक्शन मोडल में दिखाए गए फ़ील्ड में पेस्ट करें
-5.**"कनेक्ट"**पर क्लिक करें
+1. OmniRoute opens the Google authorization URL
+2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server)
+3. **Copy the full URL** from your browser's address bar (even if the page doesn't load)
+4. Paste that URL into the field shown in the OmniRoute connection modal
+5. Click **"Connect"**
-> यह काम करता है क्योंकि यूआरएल में प्राधिकरण कोड इस बात पर ध्यान दिए बिना मान्य है कि रीडायरेक्ट पेज लोड किया गया है या नहीं।---
+> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded.
-<विवरण>
-<सारांश>🇧🇷 पुर्तगाली भाषा में वर्साओसारांश>#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+---
-प्रमाणित करने के लिए**एंटीग्रेविटी**और**मिथुन सीएलआई**का उपयोग करें**Google OAuth 2.0**का उपयोग करें। Google को एक `redirect_uri` का उपयोग करना चाहिए जो बिना किसी प्रवाह के OAuth सेजा**exatamente**का उपयोग करता है और आपको Google क्लाउड कंसोल के लिए यूआरआई डाउनलोड करने की आवश्यकता है।
+
+🇧🇷 Versão em Português
-जैसा कि OAuth का क्रेडेंशियल है, कोई ओम्निरूट एस्टाओ कैडस्ट्रास नहीं है**'लोकलहोस्ट'**के लिए एपेनास। एक सर्विडोर रिमोट (उदा: `https://omniroute.meuservidor.com`) पर ओम्निरूट का उपयोग कैसे करें, या Google एक ऑटेंटिका कॉम को पुनः प्राप्त करता है:```
+#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+
+Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo.
+
+As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solução: Configure suas próprias credenciais OAuth
-आपका सटीक विवरण**OAuth 2.0 क्लाइंट आईडी**आपके सर्वर पर यूआरआई के साथ Google क्लाउड कंसोल नहीं है।#### Passo a passo
+Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor.
-**1. Google क्लाउड कंसोल तक पहुंच**
+#### Passo a passo
-अब्राहम: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
+**1. Acesse o Google Cloud Console**
-**2. नया OAuth 2.0 क्लाइंट आईडी देखें**
+Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-- उन्हें क्लिक करें**"+ क्रेडेंशियल बनाएं"**→**"OAuth क्लाइंट आईडी"**
-- आवेदन टिप:**"वेब एप्लिकेशन"**
-- नोम: एस्कोल्हा क्वाल्कर नोम (उदा: `ओम्नीरूट रिमोट`)
+**2. Crie um novo OAuth 2.0 Client ID**
-**3. अधिकृत रीडायरेक्ट यूआरआई के रूप में एडिकियोन**
+- Clique em **"+ Create Credentials"** → **"OAuth client ID"**
+- Tipo de aplicativo: **"Web application"**
+- Nome: escolha qualquer nome (ex: `OmniRoute Remote`)
-कोई शिकायत नहीं**"अधिकृत रीडायरेक्ट यूआरआई"**, आदि:```
+**3. Adicione as Authorized Redirect URIs**
+
+No campo **"Authorized redirect URIs"**, adicione:
+
+```
https://seu-servidor.com/callback
+```
-````
+> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`).
-> स्थानापन्न `seu-servidor.com` अपने आईपी को अपने सर्वर पर रखें (इसमें एक आवश्यक पोर्ट भी शामिल है, उदाहरण के लिए: `http://45.33.32.156:20128/callback`)।
+**4. Salve e copie as credenciais**
-**4. साख के रूप में सहेजें और कॉपी करें**
+Após criar, o Google mostrará o **Client ID** e o **Client Secret**.
-एपोस क्रियर, Google द्वारा**क्लाइंट आईडी**और**क्लाइंट सीक्रेट**।
+**5. Configure as variáveis de ambiente**
-**5. परिवेश परिवर्तन**के रूप में कॉन्फ़िगर करें
+No seu `.env` (ou nas variáveis de ambiente do Docker):
-कोई सेउ `.env` (आप डॉकर के परिवेश को कैसे बदलते हैं):```bash
+```bash
# Para Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
@@ -1868,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
-````
+```
-**6. ओम्निरूट का नवीनीकरण**```bash
+**6. Reinicie o OmniRoute**
+```bash
# Se usando npm:
-
npm run dev
# Se usando Docker:
-
docker restart omniroute
+```
-````
+**7. Tente conectar novamente**
-**7. नए सिरे से संपर्क करें**
+Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth
-डैशबोर्ड → प्रदाता → एंटीग्रेविटी (या जेमिनी सीएलआई) → OAuth
+Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará.
-यदि आप `https://seu-servidor.com/callback` और एक प्रामाणिक कार्य के लिए Google पुनर्निर्देशन करते हैं।---
+---
#### Workaround temporário (sem configurar credenciais próprias)
-यदि आप पहले से ही उचित क्रेडेंशियल प्राप्त नहीं करना चाहते हैं, तो आपके लिए फ्लक्सो का उपयोग करना संभव है**यूआरएल का मैनुअल**:
+Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**:
-1. Google पर स्वचालित URL का उपयोग करके ओम्निरूट का उपयोग करें
-2. आप स्वचालित रूप से काम कर सकते हैं, या Google `लोकलहोस्ट` को पुनः प्राप्त कर सकता है (यदि कोई सर्वर रिमोट नहीं है)
-3.**एक यूआरएल को पूरा कॉपी करें**अपने ब्राउजर से दोबारा डाउनलोड करें (मुझे लगता है कि एक पेज अभी भी उपलब्ध है)
-4. ओम्निरूट से जुड़ने के लिए कोई भी यूआरएल नहीं है
-5. उन्हें क्लिक करें**"कनेक्ट"**
+1. O OmniRoute abrirá a URL de autorização do Google
+2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto)
+3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue)
+4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute
+5. Clique em **"Connect"**
-> यह समाधान यूआरएल को स्वचालित रूप से डाउनलोड करने के लिए काम कर रहा है और आपके द्वारा किए गए रीडायरेक्ट को स्वतंत्र रूप से वैध बनाता है।
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+
+
---
@@ -1906,64 +2171,73 @@ docker restart omniroute
## 🛠️ Tech Stack
-<विवरण>
-<सारांश>तकनीकी स्टैक विवरण का विस्तार करने के लिए क्लिक करेंसारांश>
+
+Click to expand tech stack details
--**रनटाइम**: Node.js 18-22 LTS (⚠️ Node.js 24+**समर्थित नहीं**है - `better-sqlite3` मूल बायनेरिज़ असंगत हैं)
--**भाषा**: टाइपस्क्रिप्ट 5.9 -**100% टाइपस्क्रिप्ट**`src/` और `open-sse/` में (v2.0 के बाद से कोर मॉड्यूल में शून्य `कोई भी`)
--**फ्रेमवर्क**: नेक्स्ट.जेएस 16 + रिएक्ट 19 + टेलविंड सीएसएस 4
--**डेटाबेस**: LowDB (JSON) + SQLite (डोमेन स्थिति + प्रॉक्सी लॉग + MCP ऑडिट + रूटिंग निर्णय)
--**स्कीमा**: ज़ॉड (एमसीपी टूल I/O सत्यापन, एपीआई अनुबंध)
--**प्रोटोकॉल**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
--**स्ट्रीमिंग**: सर्वर-भेजे गए इवेंट (एसएसई)
--**प्रामाणिक**: OAuth 2.0 (PKCE) + JWT + API कुंजियाँ + MCP स्कोप्ड प्राधिकरण
--**परीक्षण**: Node.js टेस्ट रनर + विटेस्ट (यूनिट, एकीकरण, E2E सहित 900+ परीक्षण)
--**सीआई/सीडी**: गिटहब क्रियाएँ (ऑटो एनपीएम प्रकाशन + रिलीज पर डॉकर हब)
--**वेबसाइट**: [omniroute.online](https://omniroute.online)
--**पैकेज**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
--**डॉकर**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
--**लचीलापन**: सर्किट ब्रेकर, एक्सपोनेंशियल बैकऑफ़, एंटी-थंडरिंग झुंड, टीएलएस स्पूफिंग, ऑटो-कॉम्बो सेल्फ-हीलिंग
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible)
+- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0)
+- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4
+- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions)
+- **Schemas**: Zod (MCP tool I/O validation, API contracts)
+- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
+- **Streaming**: Server-Sent Events (SSE)
+- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization
+- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E)
+- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release)
+- **Website**: [omniroute.online](https://omniroute.online)
+- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
+- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
+- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing
+
+
---
## दस्तावेज़
-| दस्तावेज़ | विवरण |
-| ------------------------------------------------ | ---------------------------------------------------------------- |
-| [उपयोगकर्ता गाइड](docs/USER_GUIDE.md) | प्रदाता, कॉम्बो, सीएलआई एकीकरण, तैनाती |
-| [एपीआई संदर्भ](docs/API_REFERENCE.md) | उदाहरण सहित सभी समापन बिंदु |
-| [एमसीपी सर्वर](ओपन-एसएसई/एमसीपी-सर्वर/रीडमी.एमडी) | 16 एमसीपी उपकरण, आईडीई कॉन्फ़िगरेशन, पायथन/टीएस/गो क्लाइंट |
-| [ए2ए सर्वर](src/lib/a2a/README.md) | JSON-RPC 2.0 प्रोटोकॉल, कौशल, स्ट्रीमिंग, कार्य प्रबंधन |
-| [ऑटो-कॉम्बो इंजन](docs/auto-combo.md) | 6-कारक स्कोरिंग, मोड पैक, स्व-उपचार |
-| [समस्या निवारण](docs/TROUBLESHOOTING.md) | सामान्य समस्याएँ एवं समाधान |
-| [आर्किटेक्चर](docs/ARCHITECTURE.md) | सिस्टम आर्किटेक्चर और आंतरिक |
-| [योगदान](CONTRIBUTING.md) | विकास सेटअप और दिशानिर्देश |
-| [OpenAPI Spec](docs/openapi.yaml) | ओपनएपीआई 3.0 विशिष्टता |
-| [सुरक्षा नीति](सुरक्षा.एमडी) | भेद्यता रिपोर्टिंग और सुरक्षा प्रथाएं |
-| [VM परिनियोजन](docs/VM_DEPLOYMENT_GUIDE.md) | संपूर्ण गाइड: VM + nginx + Cloudflare सेटअप |
-| [फीचर्स गैलरी](docs/FEATURES.md) | स्क्रीनशॉट के साथ विजुअल डैशबोर्ड टूर |
-| [रिलीज़ चेकलिस्ट](docs/RELEASE_CHECKLIST.md) | प्री-रिलीज़ सत्यापन चरण |---
+| Document | Description |
+| ---------------------------------------------- | --------------------------------------------------- |
+| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment |
+| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples |
+| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients |
+| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt |
+| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing |
+| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation |
+| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions |
+| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals |
+| [Contributing](CONTRIBUTING.md) | Development setup and guidelines |
+| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification |
+| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices |
+| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup |
+| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots |
+| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps |
+
+---
## 🗺️ Roadmap
-ओम्निरूट ने कई विकास चरणों में**210+ सुविधाओं की योजना बनाई है**। यहां प्रमुख क्षेत्र हैं:
+OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas:
-| श्रेणी | नियोजित विशेषताएं | हाइलाइट्स |
-| -------------------------------- | ---------------- | -------------------------------------------------------------------------------------------------- |
-| 🧠**रूटिंग और इंटेलिजेंस**| 25+ | न्यूनतम-विलंबता रूटिंग, टैग-आधारित रूटिंग, कोटा प्रीफ़्लाइट, पी2सी खाता चयन |
-| 🔒**सुरक्षा एवं अनुपालन**| 20+ | एसएसआरएफ हार्डनिंग, क्रेडेंशियल क्लोकिंग, प्रति समापन बिंदु दर-सीमा, प्रबंधन कुंजी स्कोपिंग |
-| 📊**अवलोकनशीलता**| 15+ | ओपन टेलीमेट्री एकीकरण, वास्तविक समय कोटा निगरानी, प्रति मॉडल लागत ट्रैकिंग |
-| 🔄**प्रदाता एकीकरण**| 20+ | डायनेमिक मॉडल रजिस्ट्री, प्रदाता कूलडाउन, मल्टी-अकाउंट कोडेक्स, कोपायलट कोटा पार्सिंग |
-| ⚡**प्रदर्शन**| 15+ | दोहरी कैश परत, शीघ्र कैश, प्रतिक्रिया कैश, स्ट्रीमिंग कीपलाइव, बैच एपीआई |
-| 🌐**पारिस्थितिकी तंत्र**| 10+ | वेबसॉकेट एपीआई, कॉन्फिग हॉट-रीलोड, वितरित कॉन्फिग स्टोर, वाणिज्यिक मोड |### 🔜 Coming Soon
+| Category | Planned Features | Highlights |
+| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- |
+| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection |
+| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping |
+| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model |
+| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing |
+| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API |
+| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode |
-- 🔗**ओपनकोड इंटीग्रेशन**- ओपनकोड एआई कोडिंग आईडीई के लिए मूल प्रदाता समर्थन
-- 🔗**TRAE एकीकरण**- TRAE AI विकास ढांचे के लिए पूर्ण समर्थन
-- 📦**बैच एपीआई**- थोक अनुरोधों के लिए अतुल्यकालिक बैच प्रोसेसिंग
-- 🎯**टैग-आधारित रूटिंग**- कस्टम टैग और मेटाडेटा के आधार पर रूट अनुरोध
-- 💰**न्यूनतम-लागत रणनीति**— स्वचालित रूप से सबसे सस्ते उपलब्ध प्रदाता का चयन करें
+### 🔜 Coming Soon
-> 📝 पूर्ण सुविधा विशिष्टताएँ [`docs/new-features/`](docs/new-features/) में उपलब्ध हैं (217 विस्तृत विवरण)---
+- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE
+- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework
+- 📦 **Batch API** — Asynchronous batch processing for bulk requests
+- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata
+- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider
+
+> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs)
+
+---
## 👥 Contributors
@@ -1971,18 +2245,20 @@ docker restart omniroute
### How to Contribute
-1. रिपॉजिटरी को फोर्क करें
-2. अपनी फीचर शाखा बनाएं (`git checkout -b फीचर/अमेजिंग-फीचर`)
-3. अपने परिवर्तन प्रतिबद्ध करें (`गिट कमिट -एम 'अद्भुत सुविधा जोड़ें'`)
-4. शाखा में पुश करें (`गिट पुश ओरिजिन फीचर/अद्भुत-फीचर`)
-5. एक पुल अनुरोध खोलें
+1. Fork the repository
+2. Create your feature branch (`git checkout -b feature/amazing-feature`)
+3. Commit your changes (`git commit -m 'Add amazing feature'`)
+4. Push to the branch (`git push origin feature/amazing-feature`)
+5. Open a Pull Request
-विस्तृत दिशानिर्देशों के लिए [CONTRIBUTING.md](CONTRIBUTING.md) देखें।### Releasing a New Version
+See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines.
+
+### Releasing a New Version
```bash
# Create a release — npm publish happens automatically
gh release create v2.0.0 --title "v2.0.0" --generate-notes
-````
+```
---
@@ -1994,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes
## 🙏 Acknowledgments
-**[9router](https://github.com/decolua/9router)**को**[decolua](https://github.com/decolua)**द्वारा विशेष धन्यवाद - मूल परियोजना जिसने इस फोर्क को प्रेरित किया। ओमनीरूट अतिरिक्त सुविधाओं, मल्टी-मोडल एपीआई और पूर्ण टाइपस्क्रिप्ट पुनर्लेखन के साथ उस अविश्वसनीय नींव पर आधारित है।
+Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite.
-**[CLIPrxyAPI](https://github.com/router-for-me/CLIPrxyAPI)**को विशेष धन्यवाद - मूल गो कार्यान्वयन जिसने इस जावास्क्रिप्ट पोर्ट को प्रेरित किया।---
+Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port.
+
+---
## लाइसेंस
-एमआईटी लाइसेंस - विवरण के लिए [लाइसेंस](लाइसेंस) देखें।---
+MIT License - see [LICENSE](LICENSE) for details.
+
+---
Built with ❤️ for developers who code 24/7
diff --git a/docs/i18n/hi/docs/ARCHITECTURE.md b/docs/i18n/hi/docs/ARCHITECTURE.md
index 47de92cee1..adb69b5ed7 100644
--- a/docs/i18n/hi/docs/ARCHITECTURE.md
+++ b/docs/i18n/hi/docs/ARCHITECTURE.md
@@ -4,80 +4,93 @@
---
-_अंतिम अद्यतन: 2026-03-28_## Executive Summary
-ओमनीरूट एक स्थानीय एआई रूटिंग गेटवे और नेक्स्ट.जेएस पर निर्मित डैशबोर्ड है।
-यह एक एकल ओपनएआई-संगत एंडपॉइंट (`/v1/*`) प्रदान करता है और अनुवाद, फ़ॉलबैक, टोकन रिफ्रेश और उपयोग ट्रैकिंग के साथ कई अपस्ट्रीम प्रदाताओं के बीच ट्रैफ़िक को रूट करता है।
-मुख्य क्षमताएं:
+_Last updated: 2026-03-28_
-- सीएलआई/टूल्स के लिए ओपनएआई-संगत एपीआई सतह (28 प्रदाता)
-- प्रदाता प्रारूपों में अनुरोध/प्रतिक्रिया अनुवाद
-- मॉडल कॉम्बो फ़ॉलबैक (मल्टी-मॉडल अनुक्रम)
-- खाता-स्तरीय फ़ॉलबैक (प्रति प्रदाता बहु-खाता)
-- OAuth + एपीआई-कुंजी प्रदाता कनेक्शन प्रबंधन
-- `/v1/embeddings` के माध्यम से एम्बेडिंग पीढ़ी (6 प्रदाता, 9 मॉडल)
-- `/v1/images/पीढ़ी` के माध्यम से छवि निर्माण (4 प्रदाता, 9 मॉडल)
-- तर्क मॉडल के लिए थिंक टैग पार्सिंग (`<थिंक>...थिंक>`) सोचें
-- सख्त ओपनएआई एसडीके संगतता के लिए प्रतिक्रिया स्वच्छता
-- क्रॉस-प्रदाता अनुकूलता के लिए भूमिका सामान्यीकरण (डेवलपर→सिस्टम, सिस्टम→उपयोगकर्ता)।
-- संरचित आउटपुट रूपांतरण (json_schema → जेमिनी रिस्पॉन्सस्कीमा)
-- प्रदाताओं, चाबियाँ, उपनाम, कॉम्बो, सेटिंग्स, मूल्य निर्धारण के लिए स्थानीय दृढ़ता
-- उपयोग/लागत ट्रैकिंग और अनुरोध लॉगिंग
-- मल्टी-डिवाइस/स्टेट सिंक के लिए वैकल्पिक क्लाउड सिंक
-- एपीआई एक्सेस नियंत्रण के लिए आईपी अनुमति सूची/ब्लॉकलिस्ट
-- सोच बजट प्रबंधन (पासथ्रू/ऑटो/कस्टम/अनुकूली)
-- वैश्विक प्रणाली शीघ्र इंजेक्शन
-- सत्र ट्रैकिंग और फ़िंगरप्रिंटिंग
-- प्रदाता-विशिष्ट प्रोफाइल के साथ प्रति-खाता बढ़ी हुई दर सीमित करना
-- प्रदाता लचीलेपन के लिए सर्किट ब्रेकर पैटर्न
-- म्यूटेक्स लॉकिंग के साथ एंटी-थंडरिंग झुंड सुरक्षा
-- हस्ताक्षर-आधारित अनुरोध डिडुप्लीकेशन कैश
-- डोमेन परत: मॉडल उपलब्धता, लागत नियम, फ़ॉलबैक नीति, लॉकआउट नीति
-- डोमेन स्थिति दृढ़ता (फ़ॉलबैक, बजट, लॉकआउट, सर्किट ब्रेकर के लिए SQLite राइट-थ्रू कैश)
-- केंद्रीकृत अनुरोध मूल्यांकन के लिए नीति इंजन (लॉकआउट → बजट → फ़ॉलबैक)
-- p50/p95/p99 विलंबता एकत्रीकरण के साथ टेलीमेट्री का अनुरोध करें
-- एंड-टू-एंड ट्रेसिंग के लिए सहसंबंध आईडी (एक्स-रिक्वेस्ट-आईडी)।
-- एपीआई कुंजी के अनुसार ऑप्ट-आउट के साथ अनुपालन ऑडिट लॉगिंग
-- एलएलएम गुणवत्ता आश्वासन के लिए इवल फ्रेमवर्क
-- वास्तविक समय सर्किट ब्रेकर स्थिति के साथ लचीलापन यूआई डैशबोर्ड
-- मॉड्यूलर OAuth प्रदाता (`src/lib/oauth/providers/` के अंतर्गत 12 व्यक्तिगत मॉड्यूल)
+## Executive Summary
-प्राथमिक रनटाइम मॉडल:
+OmniRoute is a local AI routing gateway and dashboard built on Next.js.
+It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking.
-- `src/app/api/*` के अंतर्गत Next.js ऐप रूट डैशबोर्ड एपीआई और संगतता एपीआई दोनों को लागू करते हैं
-- `src/sse/*` + `open-sse/*` में एक साझा SSE/रूटिंग कोर प्रदाता निष्पादन, अनुवाद, स्ट्रीमिंग, फ़ॉलबैक और उपयोग को संभालता है## Scope and Boundaries
+Core capabilities:
+
+- OpenAI-compatible API surface for CLI/tools (28 providers)
+- Request/response translation across provider formats
+- Model combo fallback (multi-model sequence)
+- Account-level fallback (multi-account per provider)
+- OAuth + API-key provider connection management
+- Embedding generation via `/v1/embeddings` (6 providers, 9 models)
+- Image generation via `/v1/images/generations` (4 providers, 9 models)
+- Think tag parsing (`
...`) for reasoning models
+- Response sanitization for strict OpenAI SDK compatibility
+- Role normalization (developer→system, system→user) for cross-provider compatibility
+- Structured output conversion (json_schema → Gemini responseSchema)
+- Local persistence for providers, keys, aliases, combos, settings, pricing
+- Usage/cost tracking and request logging
+- Optional cloud sync for multi-device/state sync
+- IP allowlist/blocklist for API access control
+- Thinking budget management (passthrough/auto/custom/adaptive)
+- Global system prompt injection
+- Session tracking and fingerprinting
+- Per-account enhanced rate limiting with provider-specific profiles
+- Circuit breaker pattern for provider resilience
+- Anti-thundering herd protection with mutex locking
+- Signature-based request deduplication cache
+- Domain layer: model availability, cost rules, fallback policy, lockout policy
+- Context Relay: session handoff summaries for account rotation continuity
+- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers)
+- Policy engine for centralized request evaluation (lockout → budget → fallback)
+- Request telemetry with p50/p95/p99 latency aggregation
+- Correlation ID (X-Request-Id) for end-to-end tracing
+- Compliance audit logging with opt-out per API key
+- Eval framework for LLM quality assurance
+- Resilience UI dashboard with real-time circuit breaker status
+- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`)
+
+Primary runtime model:
+
+- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs
+- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage
+
+## Scope and Boundaries
### In Scope
-- स्थानीय गेटवे रनटाइम
-- डैशबोर्ड प्रबंधन एपीआई
-- प्रदाता प्रमाणीकरण और टोकन ताज़ा करें
-- अनुवाद और एसएसई स्ट्रीमिंग का अनुरोध करें
-- स्थानीय स्थिति + उपयोग की दृढ़ता
-- वैकल्पिक क्लाउड सिंक ऑर्केस्ट्रेशन### Out of Scope
+- Local gateway runtime
+- Dashboard management APIs
+- Provider authentication and token refresh
+- Request translation and SSE streaming
+- Local state + usage persistence
+- Optional cloud sync orchestration
-- `NEXT_PUBLIC_CLOUD_URL` के पीछे क्लाउड सेवा कार्यान्वयन
-- स्थानीय प्रक्रिया के बाहर प्रदाता एसएलए/नियंत्रण विमान
-- बाहरी सीएलआई बायनेरिज़ स्वयं (क्लाउड सीएलआई, कोडेक्स सीएलआई, आदि)## Dashboard Surface (Current)
+### Out of Scope
-`src/app/(डैशबोर्ड)/डैशबोर्ड/` के अंतर्गत मुख्य पृष्ठ:
+- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL`
+- Provider SLA/control plane outside local process
+- External CLI binaries themselves (Claude CLI, Codex CLI, etc.)
-- `/डैशबोर्ड` - त्वरित शुरुआत + प्रदाता अवलोकन
-- `/डैशबोर्ड/एंडपॉइंट` - एंडपॉइंट प्रॉक्सी + एमसीपी + ए2ए + एपीआई एंडपॉइंट टैब
-- `/डैशबोर्ड/प्रदाता` - प्रदाता कनेक्शन और क्रेडेंशियल
-- `/डैशबोर्ड/कॉम्बोस` - कॉम्बो रणनीतियाँ, टेम्पलेट, मॉडल रूटिंग नियम
-- `/डैशबोर्ड/लागत` - लागत एकत्रीकरण और मूल्य निर्धारण दृश्यता
-- `/डैशबोर्ड/एनालिटिक्स` - उपयोग विश्लेषण और मूल्यांकन
-- `/डैशबोर्ड/सीमाएँ` - कोटा/दर नियंत्रण
-- `/डैशबोर्ड/क्ली-टूल्स` - सीएलआई ऑनबोर्डिंग, रनटाइम डिटेक्शन, कॉन्फिग जेनरेशन
-- `/डैशबोर्ड/एजेंट` - पता चला एसीपी एजेंट + कस्टम एजेंट पंजीकरण
-- `/डैशबोर्ड/मीडिया` - छवि/वीडियो/संगीत खेल का मैदान
-- `/डैशबोर्ड/सर्च-टूल्स` - खोज प्रदाता परीक्षण और इतिहास
-- `/डैशबोर्ड/स्वास्थ्य` - अपटाइम, सर्किट ब्रेकर, दर सीमा
-- `/डैशबोर्ड/लॉग्स` - अनुरोध/प्रॉक्सी/ऑडिट/कंसोल लॉग
-- `/डैशबोर्ड/सेटिंग्स` - सिस्टम सेटिंग्स टैब (सामान्य, रूटिंग, कॉम्बो डिफ़ॉल्ट, आदि)
-- `/डैशबोर्ड/एपीआई-मैनेजर` - एपीआई कुंजी जीवनचक्र और मॉडल अनुमतियाँ## High-Level System Context
+## Dashboard Surface (Current)
+
+Main pages under `src/app/(dashboard)/dashboard/`:
+
+- `/dashboard` — quick start + provider overview
+- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs
+- `/dashboard/providers` — provider connections and credentials
+- `/dashboard/combos` — combo strategies, templates, model routing rules
+- `/dashboard/costs` — cost aggregation and pricing visibility
+- `/dashboard/analytics` — usage analytics and evaluations
+- `/dashboard/limits` — quota/rate controls
+- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation
+- `/dashboard/agents` — detected ACP agents + custom agent registration
+- `/dashboard/media` — image/video/music playground
+- `/dashboard/search-tools` — search provider testing and history
+- `/dashboard/health` — uptime, circuit breakers, rate limits
+- `/dashboard/logs` — request/proxy/audit/console logs
+- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.)
+- `/dashboard/api-manager` — API key lifecycle and model permissions
+
+## High-Level System Context
```mermaid
flowchart LR
@@ -129,139 +142,151 @@ flowchart LR
## 1) API and Routing Layer (Next.js App Routes)
-मुख्य निर्देशिकाएँ:
+Main directories:
-- अनुकूलता एपीआई के लिए `src/app/api/v1/*` और `src/app/api/v1beta/*`
-- प्रबंधन/कॉन्फ़िगरेशन एपीआई के लिए `src/app/api/*`
-- अगला `next.config.mjs` मैप `/v1/*` से `/api/v1/*` में फिर से लिखता है
+- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs
+- `src/app/api/*` for management/configuration APIs
+- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*`
-महत्वपूर्ण अनुकूलता मार्ग:
+Important compatibility routes:
- `src/app/api/v1/chat/completions/route.ts`
- `src/app/api/v1/messages/route.ts`
- `src/app/api/v1/responses/route.ts`
-- `src/app/api/v1/models/route.ts` - इसमें `कस्टम: ट्रू` के साथ कस्टम मॉडल शामिल हैं
-- `src/app/api/v1/embeddings/route.ts` - एम्बेडिंग जेनरेशन (6 प्रदाता)
-- `src/app/api/v1/images/जेनरेशन/रूट.ts` - छवि निर्माण (एंटीग्रेविटी/नेबियस सहित 4+ प्रदाता)
+- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true`
+- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers)
+- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius)
- `src/app/api/v1/messages/count_tokens/route.ts`
-- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` - प्रति-प्रदाता समर्पित चैट
-- `src/app/api/v1/providers/[provider]/embeddings/route.ts` - प्रति-प्रदाता समर्पित एम्बेडिंग
-- `src/app/api/v1/providers/[provider]/images/nations/route.ts` - प्रति-प्रदाता समर्पित छवियां
+- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat
+- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings
+- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images
- `src/app/api/v1beta/models/route.ts`
- `src/app/api/v1beta/models/[...path]/route.ts`
-प्रबंधन डोमेन:
+Management domains:
-- प्रामाणिक/सेटिंग्स: `src/app/api/auth/*`, `src/app/api/settings/*`
-- प्रदाता/कनेक्शन: `src/app/api/प्रदाता*`
-- प्रदाता नोड्स: `src/app/api/provider-nodes*`
-- कस्टम मॉडल: `src/app/api/provider-models` (प्राप्त करें/पोस्ट करें/हटाएं)
-- मॉडल कैटलॉग: `src/app/api/models/route.ts` (GET)
-- प्रॉक्सी कॉन्फ़िगरेशन: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
+- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*`
+- Providers/connections: `src/app/api/providers*`
+- Provider nodes: `src/app/api/provider-nodes*`
+- Custom models: `src/app/api/provider-models` (GET/POST/DELETE)
+- Model catalog: `src/app/api/models/route.ts` (GET)
+- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
- OAuth: `src/app/api/oauth/*`
-- कुंजी/उपनाम/कॉम्बोस/मूल्य निर्धारण: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing`
-- उपयोग: `src/app/api/usage/*`
-- सिंक/क्लाउड: `src/app/api/sync/*`, `src/app/api/cloud/*`
-- सीएलआई टूलींग सहायक: `src/app/api/cli-tools/*`
-- आईपी फ़िल्टर: `src/app/api/settings/ip-filter` (प्राप्त/पुट)
-- सोच बजट: `src/app/api/settings/thinking-budget` (प्राप्त/पुट)
-- सिस्टम प्रॉम्प्ट: `src/app/api/settings/system-prompt` (GET/PUT)
-- सत्र: `src/app/api/sessions` (प्राप्त करें)
-- दर सीमा: `src/app/api/दर-सीमा` (प्राप्त करें)
-- लचीलापन: `src/app/api/resilience` (GET/PATCH) - प्रदाता प्रोफाइल, सर्किट ब्रेकर, दर सीमा स्थिति
-- लचीलापन रीसेट: `src/app/api/resilience/reset` (POST) - रीसेट ब्रेकर + कूलडाउन
-- कैश आँकड़े: `src/app/api/cache/stats` (प्राप्त करें/हटाएँ)
-- मॉडल उपलब्धता: `src/app/api/मॉडल/उपलब्धता` (प्राप्त करें/पोस्ट करें)
-- टेलीमेट्री: `src/app/api/टेलीमेट्री/सारांश` (प्राप्त करें)
-- बजट: `src/app/api/usage/budget` (प्राप्त करें/पोस्ट करें)
-- फ़ॉलबैक चेन: `src/app/api/फ़ॉलबैक/चेन` (प्राप्त करें/पोस्ट करें/हटाएं)
-- अनुपालन ऑडिट: `src/app/api/compliance/audit-log` (GET)
-- इवल्स: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET)
-- नीतियां: `src/app/api/policies` (प्राप्त करें/पोस्ट करें)## 2) SSE + Translation Core
+- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing`
+- Usage: `src/app/api/usage/*`
+- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*`
+- CLI tooling helpers: `src/app/api/cli-tools/*`
+- IP filter: `src/app/api/settings/ip-filter` (GET/PUT)
+- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT)
+- System prompt: `src/app/api/settings/system-prompt` (GET/PUT)
+- Sessions: `src/app/api/sessions` (GET)
+- Rate limits: `src/app/api/rate-limits` (GET)
+- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state
+- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns
+- Cache stats: `src/app/api/cache/stats` (GET/DELETE)
+- Model availability: `src/app/api/models/availability` (GET/POST)
+- Telemetry: `src/app/api/telemetry/summary` (GET)
+- Budget: `src/app/api/usage/budget` (GET/POST)
+- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE)
+- Compliance audit: `src/app/api/compliance/audit-log` (GET)
+- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET)
+- Policies: `src/app/api/policies` (GET/POST)
-मुख्य प्रवाह मॉड्यूल:
+## 2) SSE + Translation Core
-- प्रविष्टि: `src/sse/handlers/chat.ts`
-- कोर ऑर्केस्ट्रेशन: `open-sse/handlers/chatCore.ts`
-- प्रदाता निष्पादन एडाप्टर: `ओपन-एसएसई/निष्पादक/*`
-- प्रारूप पहचान/प्रदाता कॉन्फिगरेशन: `open-sse/services/provider.ts`
-- मॉडल पार्स/रिज़ॉल्यूशन: `src/sse/services/model.ts`, `open-sse/services/model.ts`
-- खाता फ़ॉलबैक तर्क: `open-sse/services/accountFallback.ts`
-- अनुवाद रजिस्ट्री: `open-sse/translator/index.ts`
-- स्ट्रीम परिवर्तन: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
-- उपयोग निष्कर्षण/सामान्यीकरण: `open-sse/utils/usageTracking.ts`
-- टैग पार्सर सोचें: `open-sse/utils/thinkTagParser.ts`
-- एंबेडिंग हैंडलर: `open-sse/handlers/embeddings.ts`
-- एंबेडिंग प्रदाता रजिस्ट्री: `open-sse/config/embeddingRegistry.ts`
-- छवि निर्माण हैंडलर: `open-sse/handlers/imageGeneration.ts`
-- छवि प्रदाता रजिस्ट्री: `open-sse/config/imageRegistry.ts`
-- रिस्पॉन्स सेनिटाइजेशन: `ओपन-एसएसई/हैंडलर्स/रिस्पांससैनिटाइजर.टीएस`
-- भूमिका सामान्यीकरण: `open-sse/services/roleNormalizer.ts`
+Main flow modules:
-सेवाएँ (व्यावसायिक तर्क):
+- Entry: `src/sse/handlers/chat.ts`
+- Core orchestration: `open-sse/handlers/chatCore.ts`
+- Provider execution adapters: `open-sse/executors/*`
+- Format detection/provider config: `open-sse/services/provider.ts`
+- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts`
+- Account fallback logic: `open-sse/services/accountFallback.ts`
+- Translation registry: `open-sse/translator/index.ts`
+- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
+- Usage extraction/normalization: `open-sse/utils/usageTracking.ts`
+- Think tag parser: `open-sse/utils/thinkTagParser.ts`
+- Embedding handler: `open-sse/handlers/embeddings.ts`
+- Embedding provider registry: `open-sse/config/embeddingRegistry.ts`
+- Image generation handler: `open-sse/handlers/imageGeneration.ts`
+- Image provider registry: `open-sse/config/imageRegistry.ts`
+- Response sanitization: `open-sse/handlers/responseSanitizer.ts`
+- Role normalization: `open-sse/services/roleNormalizer.ts`
-- खाता चयन/स्कोरिंग: `open-sse/services/accountSelector.ts`
-- संदर्भ जीवनचक्र प्रबंधन: `open-sse/services/contextManager.ts`
-- आईपी फ़िल्टर प्रवर्तन: `open-sse/services/ipFilter.ts`
-- सत्र ट्रैकिंग: `open-sse/services/sessionManager.ts`
-- डुप्लिकेशन अनुरोध: `open-sse/services/signatureCache.ts`
-- सिस्टम प्रॉम्प्ट इंजेक्शन: `open-sse/services/systemPrompt.ts`
-- सोच बजट प्रबंधन: `open-sse/services/thinkingBudget.ts`
-- वाइल्डकार्ड मॉडल रूटिंग: `open-sse/services/wildcardRouter.ts`
-- दर सीमा प्रबंधन: `open-sse/services/rateLimitManager.ts`
-- सर्किट ब्रेकर: `open-sse/services/circuitBreaker.ts`
+Services (business logic):
-डोमेन परत मॉड्यूल:
+- Account selection/scoring: `open-sse/services/accountSelector.ts`
+- Context lifecycle management: `open-sse/services/contextManager.ts`
+- IP filter enforcement: `open-sse/services/ipFilter.ts`
+- Session tracking: `open-sse/services/sessionManager.ts`
+- Request deduplication: `open-sse/services/signatureCache.ts`
+- System prompt injection: `open-sse/services/systemPrompt.ts`
+- Thinking budget management: `open-sse/services/thinkingBudget.ts`
+- Wildcard model routing: `open-sse/services/wildcardRouter.ts`
+- Rate limit management: `open-sse/services/rateLimitManager.ts`
+- Circuit breaker: `open-sse/services/circuitBreaker.ts`
+- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy
+- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions
-- मॉडल उपलब्धता: `src/lib/domain/modelAvailability.ts`
-- लागत नियम/बजट: `src/lib/domain/costRules.ts`
-- फ़ॉलबैक नीति: `src/lib/domain/fallbackPolicy.ts`
-- कॉम्बो रिज़ॉल्वर: `src/lib/domain/comboResolver.ts`
-- लॉकआउट नीति: `src/lib/domain/lockoutPolicy.ts`
-- नीति इंजन: `src/domain/policyEngine.ts` - केंद्रीकृत लॉकआउट → बजट → फ़ॉलबैक मूल्यांकन
-- त्रुटि कोड कैटलॉग: `src/lib/domain/errorCodes.ts`
-- अनुरोध आईडी: `src/lib/domain/requestId.ts`
-- फ़ेच टाइमआउट: `src/lib/domain/fetchTimeout.ts`
-- अनुरोध टेलीमेट्री: `src/lib/domain/requestTelemetry.ts`
-- अनुपालन/ऑडिट: `src/lib/domain/compliance/index.ts`
-- इवल रनर: `src/lib/domain/evalRunner.ts`
-- डोमेन स्थिति दृढ़ता: `src/lib/db/domainState.ts` - फ़ॉलबैक चेन, बजट, लागत इतिहास, लॉकआउट स्थिति, सर्किट ब्रेकर के लिए SQLite CRUD
+Domain layer modules:
-OAuth प्रदाता मॉड्यूल (`src/lib/oauth/providers/` के अंतर्गत 12 व्यक्तिगत फ़ाइलें):
+- Model availability: `src/lib/domain/modelAvailability.ts`
+- Cost rules/budgets: `src/lib/domain/costRules.ts`
+- Fallback policy: `src/lib/domain/fallbackPolicy.ts`
+- Combo resolver: `src/lib/domain/comboResolver.ts`
+- Lockout policy: `src/lib/domain/lockoutPolicy.ts`
+- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation
+- Error codes catalog: `src/lib/domain/errorCodes.ts`
+- Request ID: `src/lib/domain/requestId.ts`
+- Fetch timeout: `src/lib/domain/fetchTimeout.ts`
+- Request telemetry: `src/lib/domain/requestTelemetry.ts`
+- Compliance/audit: `src/lib/domain/compliance/index.ts`
+- Eval runner: `src/lib/domain/evalRunner.ts`
+- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers
-- रजिस्ट्री सूचकांक: `src/lib/oauth/providers/index.ts`
-- व्यक्तिगत प्रदाता: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
-- पतला आवरण: `src/lib/oauth/providers.ts` - अलग-अलग मॉड्यूल से पुनः निर्यात## 3) Persistence Layer
+OAuth provider modules (12 individual files under `src/lib/oauth/providers/`):
-प्राथमिक अवस्था DB (SQLite):
+- Registry index: `src/lib/oauth/providers/index.ts`
+- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
+- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules
-- कोर इन्फ्रा: `src/lib/db/core.ts` (बेहतर-sqlite3, माइग्रेशन, वाल)
-- पुनः निर्यात पहलू: `src/lib/localDb.ts` (कॉलर्स के लिए पतली अनुकूलता परत)
-- फ़ाइल: `${DATA_DIR}/storage.sqlite` (या `$XDG_CONFIG_HOME/omniroute/storage.sqlite` सेट होने पर, अन्यथा `~/.omniroute/storage.sqlite`)
-- इकाइयां (टेबल + केवी नेमस्पेस): प्रोवाइडरकनेक्शन्स, प्रोवाइडरनोड्स, मॉडलएलियासेस, कॉम्बो, एपीआईकीज़, सेटिंग्स, मूल्य निर्धारण,**कस्टममॉडल**,**प्रॉक्सीकॉन्फिग**,**आईपीफिल्टर**,**थिंकिंगबजट**,**सिस्टमप्रॉम्प्ट**
+## 3) Persistence Layer
-उपयोग दृढ़ता:
+Primary state DB (SQLite):
-- मुखौटा: `src/lib/usageDb.ts` (`src/lib/usage/*` में विघटित मॉड्यूल)
-- `storage.sqlite` में SQLite तालिकाएँ: `usage_history`, `call_logs`, `proxy_logs`
+- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL)
+- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers)
+- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`)
+- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt**
+
+Usage persistence:
+
+- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`)
+- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs`
- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `
/logs/...`)
-- मौजूद होने पर लीगेसी JSON फ़ाइलें स्टार्टअप माइग्रेशन द्वारा SQLite में माइग्रेट की जाती हैं
+- legacy JSON files are migrated to SQLite by startup migrations when present
-डोमेन स्थिति DB (SQLite):
+Domain State DB (SQLite):
-- `src/lib/db/domainState.ts` - डोमेन स्थिति के लिए CRUD संचालन
-- टेबल्स (`src/lib/db/core.ts` में निर्मित): `domain_fallback_चेन्स`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
-- राइट-थ्रू कैश पैटर्न: इन-मेमोरी मैप्स रनटाइम पर आधिकारिक होते हैं; उत्परिवर्तन SQLite के साथ समकालिक रूप से लिखे जाते हैं; कोल्ड स्टार्ट पर राज्य को डीबी से बहाल किया जाता है## 4) Auth + Security Surfaces
+- `src/lib/db/domainState.ts` — CRUD operations for domain state
+- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
+- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start
+
+## 4) Auth + Security Surfaces
- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts`
-- एपीआई कुंजी निर्माण/सत्यापन: `src/shared/utils/apiKey.ts`
-- प्रदाता रहस्य `providerConnections` प्रविष्टियों में बने रहे
-- `open-sse/utils/proxyFetch.ts` (env vars) और `open-sse/utils/networkProxy.ts` (प्रति-प्रदाता या वैश्विक रूप से कॉन्फ़िगर करने योग्य) के माध्यम से आउटबाउंड प्रॉक्सी समर्थन## 5) Cloud Sync
+- API key generation/verification: `src/shared/utils/apiKey.ts`
+- Provider secrets persisted in `providerConnections` entries
+- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global)
-- शेड्यूलर init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
-- आवधिक कार्य: `src/shared/services/cloudSyncScheduler.ts`
-- आवधिक कार्य: `src/shared/services/modelSyncScheduler.ts`
-- नियंत्रण मार्ग: `src/app/api/sync/cloud/route.ts`## Request Lifecycle (`/v1/chat/completions`)
+## 5) Cloud Sync
+
+- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
+- Periodic task: `src/shared/services/cloudSyncScheduler.ts`
+- Periodic task: `src/shared/services/modelSyncScheduler.ts`
+- Control route: `src/app/api/sync/cloud/route.ts`
+
+## Request Lifecycle (`/v1/chat/completions`)
```mermaid
sequenceDiagram
@@ -338,7 +363,9 @@ flowchart TD
Q -- No --> R[Return all unavailable]
```
-फ़ॉलबैक निर्णय स्थिति कोड और त्रुटि-संदेश अनुमानों का उपयोग करके `open-sse/services/accountFallback.ts` द्वारा संचालित होते हैं। कॉम्बो रूटिंग एक अतिरिक्त गार्ड जोड़ता है: प्रदाता-स्कोप्ड 400s जैसे अपस्ट्रीम सामग्री-ब्लॉक और भूमिका-सत्यापन विफलताओं को मॉडल-स्थानीय विफलताओं के रूप में माना जाता है ताकि बाद में कॉम्बो लक्ष्य अभी भी चल सकें।## OAuth Onboarding and Token Refresh Lifecycle
+Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run.
+
+## OAuth Onboarding and Token Refresh Lifecycle
```mermaid
sequenceDiagram
@@ -368,7 +395,9 @@ sequenceDiagram
Test-->>UI: validation result
```
-लाइव ट्रैफ़िक के दौरान रिफ्रेश को निष्पादक `refreshCredentials()` के माध्यम से `open-sse/handlers/chatCore.ts` के अंदर निष्पादित किया जाता है।## Cloud Sync Lifecycle (Enable / Sync / Disable)
+Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`.
+
+## Cloud Sync Lifecycle (Enable / Sync / Disable)
```mermaid
sequenceDiagram
@@ -503,12 +532,14 @@ erDiagram
}
```
-भौतिक भंडारण फ़ाइलें:
+Physical storage files:
-- प्राथमिक रनटाइम DB: `${DATA_DIR}/storage.sqlite`
-- अनुरोध लॉग लाइनें: `${DATA_DIR}/log.txt` (कॉम्पैट/डीबग आर्टिफैक्ट)
-- संरचित कॉल पेलोड अभिलेखागार: `${DATA_DIR}/call_logs/`
-- वैकल्पिक अनुवादक/अनुरोध डिबग सत्र: `/logs/...`## Deployment Topology
+- primary runtime DB: `${DATA_DIR}/storage.sqlite`
+- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact)
+- structured call payload archives: `${DATA_DIR}/call_logs/`
+- optional translator/request debug sessions: `/logs/...`
+
+## Deployment Topology
```mermaid
flowchart LR
@@ -543,205 +574,249 @@ flowchart LR
### Route and API Modules
-- `src/app/api/v1/*`, `src/app/api/v1beta/*`: अनुकूलता एपीआई
-- `src/app/api/v1/providers/[provider]/*`: प्रति-प्रदाता समर्पित मार्ग (चैट, एम्बेडिंग, चित्र)
-- `src/app/api/providers*`: प्रदाता CRUD, सत्यापन, परीक्षण
-- `src/app/api/provider-nodes*`: कस्टम संगत नोड प्रबंधन
-- `src/app/api/provider-models`: कस्टम मॉडल प्रबंधन (CRUD)
-- `src/app/api/models/route.ts`: मॉडल कैटलॉग एपीआई (उपनाम + कस्टम मॉडल)
-- `src/app/api/oauth/*`: OAuth/डिवाइस-कोड प्रवाह
-- `src/app/api/keys*`: स्थानीय एपीआई कुंजी जीवनचक्र
-- `src/app/api/models/alias`: उपनाम प्रबंधन
-- `src/app/api/combos*`: फ़ॉलबैक कॉम्बो प्रबंधन
-- `src/app/api/pricing`: लागत गणना के लिए मूल्य निर्धारण ओवरराइड हो जाता है
-- `src/app/api/settings/proxy`: प्रॉक्सी कॉन्फ़िगरेशन (प्राप्त/पुट/हटाएं)
-- `src/app/api/settings/proxy/test`: आउटबाउंड प्रॉक्सी कनेक्टिविटी टेस्ट (POST)
-- `src/app/api/usage/*`: एपीआई का उपयोग और लॉग
-- `src/app/api/sync/*` + `src/app/api/cloud/*`: क्लाउड सिंक और क्लाउड-फेसिंग हेल्पर्स
-- `src/app/api/cli-tools/*`: स्थानीय सीएलआई कॉन्फ़िगरेशन लेखक/चेकर्स
-- `src/app/api/settings/ip-filter`: आईपी अनुमति सूची/ब्लॉकलिस्ट (प्राप्त/पुट)
-- `src/app/api/settings/thinking-budget`: थिंकिंग टोकन बजट कॉन्फ़िगरेशन (GET/PUT)
-- `src/app/api/settings/system-prompt`: ग्लोबल सिस्टम प्रॉम्प्ट (GET/PUT)
-- `src/app/api/sessions`: सक्रिय सत्र सूची (GET)
-- `src/app/api/rate-limits`: प्रति-खाता दर सीमा स्थिति (GET)### Routing and Execution Core
+- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs
+- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images)
+- `src/app/api/providers*`: provider CRUD, validation, testing
+- `src/app/api/provider-nodes*`: custom compatible node management
+- `src/app/api/provider-models`: custom model management (CRUD)
+- `src/app/api/models/route.ts`: model catalog API (aliases + custom models)
+- `src/app/api/oauth/*`: OAuth/device-code flows
+- `src/app/api/keys*`: local API key lifecycle
+- `src/app/api/models/alias`: alias management
+- `src/app/api/combos*`: fallback combo management
+- `src/app/api/pricing`: pricing overrides for cost calculation
+- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE)
+- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST)
+- `src/app/api/usage/*`: usage and logs APIs
+- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers
+- `src/app/api/cli-tools/*`: local CLI config writers/checkers
+- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT)
+- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT)
+- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT)
+- `src/app/api/sessions`: active session listing (GET)
+- `src/app/api/rate-limits`: per-account rate limit status (GET)
-- `src/sse/handlers/chat.ts`: अनुरोध पार्स, कॉम्बो हैंडलिंग, खाता चयन लूप
-- `ओपन-एसएसई/हैंडलर/चैटकोर.टीएस`: अनुवाद, निष्पादक प्रेषण, पुनः प्रयास/रीफ्रेश हैंडलिंग, स्ट्रीम सेटअप
-- `ओपन-एसएसई/निष्पादक/*`: प्रदाता-विशिष्ट नेटवर्क और प्रारूप व्यवहार### Translation Registry and Format Converters
+### Routing and Execution Core
-- `open-sse/translator/index.ts`: अनुवादक रजिस्ट्री और ऑर्केस्ट्रेशन
-- अनुवादकों से अनुरोध: `ओपन-एसएसई/अनुवादक/अनुरोध/*`
-- प्रतिक्रिया अनुवादक: `ओपन-एसएसई/अनुवादक/प्रतिक्रिया/*`
-- प्रारूप स्थिरांक: `open-sse/translator/formats.ts`### Persistence
+- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop
+- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup
+- `open-sse/executors/*`: provider-specific network and format behavior
-- `src/lib/db/*`: SQLite पर लगातार कॉन्फ़िगरेशन/स्थिति और डोमेन दृढ़ता
-- `src/lib/localDb.ts`: डीबी मॉड्यूल के लिए अनुकूलता पुनः निर्यात
-- `src/lib/usageDb.ts`: SQLite तालिकाओं के शीर्ष पर उपयोग इतिहास/कॉल लॉग मुखौटा## Provider Executor Coverage (Strategy Pattern)
+### Translation Registry and Format Converters
-प्रत्येक प्रदाता के पास `BaseExecutor` (`open-sse/executors/base.ts` में) का विस्तार करने वाला एक विशेष निष्पादक होता है, जो URL निर्माण, हेडर निर्माण, घातीय बैकऑफ़ के साथ पुनः प्रयास, क्रेडेंशियल रिफ्रेश हुक और `execute()` ऑर्केस्ट्रेशन विधि प्रदान करता है।
+- `open-sse/translator/index.ts`: translator registry and orchestration
+- Request translators: `open-sse/translator/request/*`
+- Response translators: `open-sse/translator/response/*`
+- Format constants: `open-sse/translator/formats.ts`
-| निष्पादक | प्रदाता(ओं) | विशेष हैंडलिंग |
-| --------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------- |
-| `डिफ़ॉल्ट निष्पादक` | ओपनएआई, क्लाउड, जेमिनी, क्वेन, क्यूडर, ओपनराउटर, जीएलएम, किमी, मिनीमैक्स, डीपसीक, ग्रोक, एक्सएआई, मिस्ट्रल, पर्प्लेक्सिटी, टुगेदर, फायरवर्क्स, सेरेब्रा, कोहेरे, एनवीआईडीआईए | प्रति प्रदाता डायनामिक यूआरएल/हेडर कॉन्फिगरेशन |
-| 'एंटीग्रेविटी एक्ज़ीक्यूटर' | गूगल एंटीग्रेविटी | कस्टम प्रोजेक्ट/सत्र आईडी, पुनः प्रयास करें-पार्सिंग के बाद |
-| `कोडेक्स एक्ज़ीक्यूटर` | ओपनएआई कोडेक्स | सिस्टम निर्देश इंजेक्ट करता है, तर्क करने का प्रयास करता है |
-| `कर्सर निष्पादक` | कर्सर आईडीई | कनेक्टआरपीसी प्रोटोकॉल, प्रोटोबफ एन्कोडिंग, चेकसम के माध्यम से हस्ताक्षर करने का अनुरोध |
-| 'GithubExecutor' | गिटहब कोपायलट | कोपायलट टोकन ताज़ा करें, VSCode-नकल हेडर |
-| `कीरो एक्ज़ीक्यूटर` | एडब्ल्यूएस कोडव्हिस्परर/किरो | एडब्ल्यूएस इवेंटस्ट्रीम बाइनरी प्रारूप → एसएसई रूपांतरण |
-| `जेमिनीसीएलआईएक्सक्यूटर` | जेमिनी सीएलआई | Google OAuth टोकन ताज़ा चक्र |
+### Persistence
-अन्य सभी प्रदाता (कस्टम संगत नोड्स सहित) `DefaultExecutor` का उपयोग करते हैं।## Provider Compatibility Matrix
+- `src/lib/db/*`: persistent config/state and domain persistence on SQLite
+- `src/lib/localDb.ts`: compatibility re-export for DB modules
+- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables
-| प्रदाता | प्रारूप | प्रामाणिक | स्ट्रीम | नॉन-स्ट्रीम | टोकन ताज़ा करें | उपयोग एपीआई |
-| --------------------- | -------------------- | ------------------------ | ----------------- | ----------- | --------------- | ------------------- | ------------------------------ |
-| क्लाउड | क्लाउड | एपीआई कुंजी / OAuth | ✅ | ✅ | ✅ | ⚠️ केवल एडमिन |
-| मिथुन | मिथुन | एपीआई कुंजी / OAuth | ✅ | ✅ | ✅ | ⚠️ क्लाउड कंसोल |
-| जेमिनी सीएलआई | मिथुन-क्ली | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
-| प्रतिगुरुत्वाकर्षण | प्रतिगुरुत्वाकर्षण | OAuth | ✅ | ✅ | ✅ | ✅ पूर्ण कोटा एपीआई |
-| ओपनएआई | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ |
-| कोडेक्स | openai-प्रतिक्रियाएं | OAuth | ✅ मजबूर | ❌ | ✅ | ✅ दर सीमा |
-| गिटहब कोपायलट | ओपनाई | OAuth + सहपायलट टोकन | ✅ | ✅ | ✅ | ✅ कोटा स्नैपशॉट |
-| कर्सर | कर्सर | कस्टम चेकसम | ✅ | ✅ | ❌ | ❌ |
-| किरो | किरो | एडब्ल्यूएस एसएसओ ओआईडीसी | ✅ (इवेंटस्ट्रीम) | ❌ | ✅ | ✅ उपयोग सीमा |
-| क्वेन | ओपनाई | OAuth | ✅ | ✅ | ✅ | ⚠️ प्रति अनुरोध |
-| कोडर | ओपनाई | OAuth (बेसिक) | ✅ | ✅ | ✅ | ⚠️ प्रति अनुरोध |
-| ओपनराउटर | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ |
-| जीएलएम/किमी/मिनीमैक्स | क्लाउड | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ |
-| डीपसीक | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ |
-| ग्रोक | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ |
-| एक्सएआई (ग्रोक) | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ |
-| मिस्ट्रल | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ |
-| उलझन | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ |
-| एक साथ एआई | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ |
-| आतिशबाजी एआई | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ |
-| सेरेब्रस | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ |
-| सहभागी | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ |
-| एनवीडिया एनआईएम | ओपनाई | एपीआई कुंजी | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage |
+## Provider Executor Coverage (Strategy Pattern)
-पता लगाए गए स्रोत प्रारूपों में शामिल हैं:
+Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method.
-- `ओपनाई`
-- `ओपनई-प्रतिक्रियाएँ`
-- `क्लाउड`
-- 'मिथुन'
+| Executor | Provider(s) | Special Handling |
+| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- |
+| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider |
+| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing |
+| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort |
+| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum |
+| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers |
+| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion |
+| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle |
-लक्ष्य प्रारूपों में शामिल हैं:
+All other providers (including custom compatible nodes) use the `DefaultExecutor`.
-- ओपनएआई चैट/प्रतिक्रियाएं
- -क्लाउड
-- मिथुन/मिथुन-सीएलआई/एंटीग्रेविटी लिफाफा
-- किरो
-- कर्सर
+## Provider Compatibility Matrix
-अनुवाद**हब प्रारूप के रूप में ओपनएआई**का उपयोग करते हैं - सभी रूपांतरण मध्यवर्ती के रूप में ओपनएआई से गुजरते हैं:```
+| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API |
+| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ |
+| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only |
+| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API |
+| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits |
+| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots |
+| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ |
+| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits |
+| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request |
+| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request |
+| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ |
+| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+
+## Format Translation Coverage
+
+Detected source formats include:
+
+- `openai`
+- `openai-responses`
+- `claude`
+- `gemini`
+
+Target formats include:
+
+- OpenAI chat/Responses
+- Claude
+- Gemini/Gemini-CLI/Antigravity envelope
+- Kiro
+- Cursor
+
+Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate:
+
+```
Source Format → OpenAI (hub) → Target Format
+```
-````
+Translations are selected dynamically based on source payload shape and provider target format.
-स्रोत पेलोड आकार और प्रदाता लक्ष्य प्रारूप के आधार पर अनुवादों का चयन गतिशील रूप से किया जाता है।
+Additional processing layers in the translation pipeline:
-अनुवाद पाइपलाइन में अतिरिक्त प्रसंस्करण परतें:
+- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance
+- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE)
+- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field
+- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema`
--**प्रतिक्रिया स्वच्छता**- सख्त एसडीके अनुपालन सुनिश्चित करने के लिए ओपनएआई-प्रारूप प्रतिक्रियाओं (स्ट्रीमिंग और गैर-स्ट्रीमिंग दोनों) से गैर-मानक फ़ील्ड हटा देता है
--**भूमिका सामान्यीकरण**- गैर-ओपनएआई लक्ष्यों के लिए `डेवलपर` → `सिस्टम` को रूपांतरित करता है; सिस्टम भूमिका को अस्वीकार करने वाले मॉडलों के लिए `सिस्टम` → `उपयोगकर्ता` को मर्ज करता है (जीएलएम, ईआरएनआईई)
--**टैग निष्कर्षण के बारे में सोचें**- पार्स `<सोच>...सोच>` सामग्री से `reasoning_content` फ़ील्ड में ब्लॉक करता है
--**संरचित आउटपुट**- OpenAI `response_format.json_schema` को जेमिनी के `responseMimeType` + `responseSchema` में परिवर्तित करता है## Supported API Endpoints
+## Supported API Endpoints
-| समापन बिंदु | प्रारूप | हैंडलर |
-| -------------------------------------------------- | ------------------ | ---------------------------------------------------------------------------------- |
-| `पोस्ट /v1/चैट/समापन` | ओपनएआई चैट | `src/sse/handlers/chat.ts` |
-| `पोस्ट /v1/संदेश` | क्लाउड संदेश | वही हैंडलर (स्वतः पता चला) |
-| `पोस्ट /v1/प्रतिक्रियाएँ` | ओपनएआई प्रतिक्रियाएँ | `open-sse/handlers/responsesHandler.ts` |
-| `पोस्ट /v1/एम्बेडिंग` | ओपनएआई एंबेडिंग्स | `open-sse/handlers/embeddings.ts` |
-| `प्राप्त करें /v1/एम्बेडिंग्स` | मॉडल सूची | एपीआई मार्ग |
-| `पोस्ट /v1/छवियां/पीढ़ी` | OpenAI छवियाँ | `ओपन-एसएसई/हैंडलर/इमेजजेनरेशन.टीएस` |
-| `प्राप्त करें /v1/छवियां/पीढ़ी` | मॉडल सूची | एपीआई मार्ग |
-| `पोस्ट /v1/प्रदाता/{प्रदाता}/चैट/समापन` | ओपनएआई चैट | मॉडल सत्यापन के साथ प्रति-प्रदाता समर्पित |
-| `पोस्ट /v1/प्रदाता/{प्रदाता}/एम्बेडिंग्स` | ओपनएआई एंबेडिंग्स | मॉडल सत्यापन के साथ प्रति-प्रदाता समर्पित |
-| `पोस्ट /v1/प्रदाता/{प्रदाता}/छवियां/पीढ़ी` | OpenAI छवियाँ | मॉडल सत्यापन के साथ प्रति-प्रदाता समर्पित |
-| `POST /v1/messages/count_tokens` | क्लाउड टोकन गिनती | एपीआई मार्ग |
-| `प्राप्त करें /v1/मॉडल` | OpenAI मॉडल सूची | एपीआई मार्ग (चैट + एम्बेडिंग + छवि + कस्टम मॉडल) |
-| `प्राप्त करें /एपीआई/मॉडल/कैटलॉग` | कैटलॉग | प्रदाता + प्रकार | द्वारा समूहीकृत सभी मॉडल
-| `POST /v1beta/models/*:streamGenerateContent` | मिथुन राशि के जातक | एपीआई मार्ग |
-| `प्राप्त/पुट/डिलीट /एपीआई/सेटिंग्स/प्रॉक्सी` | प्रॉक्सी कॉन्फिग | नेटवर्क प्रॉक्सी कॉन्फ़िगरेशन |
-| `पोस्ट /एपीआई/सेटिंग्स/प्रॉक्सी/टेस्ट` | प्रॉक्सी कनेक्टिविटी | प्रॉक्सी स्वास्थ्य/कनेक्टिविटी परीक्षण समापन बिंदु |
-| `प्राप्त करें/पोस्ट करें/हटाएं /एपीआई/प्रदाता-मॉडल` | प्रदाता मॉडल | प्रदाता मॉडल मेटाडेटा समर्थन कस्टम और प्रबंधित उपलब्ध मॉडल |## Bypass Handler
+| Endpoint | Format | Handler |
+| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- |
+| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` |
+| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) |
+| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` |
+| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` |
+| `GET /v1/embeddings` | Model listing | API route |
+| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` |
+| `GET /v1/images/generations` | Model listing | API route |
+| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation |
+| `POST /v1/messages/count_tokens` | Claude Token Count | API route |
+| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) |
+| `GET /api/models/catalog` | Catalog | All models grouped by provider + type |
+| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route |
+| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration |
+| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint |
+| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models |
-बाईपास हैंडलर (`ओपन-एसएसई/यूटिल्स/बायपासहैंडलर.टीएस`) क्लाउड सीएलआई से ज्ञात "थ्रोअवे" अनुरोधों को रोकता है - वार्मअप पिंग, शीर्षक निष्कर्षण और टोकन गिनती - और अपस्ट्रीम प्रदाता टोकन का उपभोग किए बिना एक**नकली प्रतिक्रिया**देता है। यह तभी ट्रिगर होता है जब `User-Agent` में `claude-cli` होता है।## Request Logger Pipeline
+## Bypass Handler
-अनुरोध लकड़हारा (`open-sse/utils/requestLogger.ts`) एक 7-चरण डीबग लॉगिंग पाइपलाइन प्रदान करता है, जो डिफ़ॉल्ट रूप से अक्षम है, `ENABLE_REQUEST_LOGS=true` के माध्यम से सक्षम है:```
+The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`.
+
+## Request Logger Pipeline
+
+The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`:
+
+```
1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json
→ 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt
-````
+```
-प्रत्येक अनुरोध सत्र के लिए फ़ाइलें `/logs//` पर लिखी जाती हैं।## Failure Modes and Resilience
+Files are written to `/logs//` for each request session.
+
+## Failure Modes and Resilience
## 1) Account/Provider Availability
-- क्षणिक/दर/प्रामाणिक त्रुटियों पर प्रदाता खाता ठंडा हो गया
-- अनुरोध विफल होने से पहले खाता फ़ॉलबैक
-- वर्तमान मॉडल/प्रदाता पथ समाप्त होने पर कॉम्बो मॉडल फ़ॉलबैक## 2) Token Expiry
+- provider account cooldown on transient/rate/auth errors
+- account fallback before failing request
+- combo model fallback when current model/provider path is exhausted
-- ताज़ा करने योग्य प्रदाताओं के लिए पुनः प्रयास के साथ पूर्व-जांच और ताज़ा करें
-- कोर पथ में ताज़ा प्रयास के बाद 401/403 पुनः प्रयास करें## 3) Stream Safety
+## 2) Token Expiry
-- डिस्कनेक्ट-अवेयर स्ट्रीम नियंत्रक
-- एंड-ऑफ-स्ट्रीम फ्लश और `[DONE]` हैंडलिंग के साथ अनुवाद स्ट्रीम
-- प्रदाता उपयोग मेटाडेटा अनुपलब्ध होने पर उपयोग अनुमान फ़ॉलबैक## 4) Cloud Sync Degradation
+- pre-check and refresh with retry for refreshable providers
+- 401/403 retry after refresh attempt in core path
-- समन्वयन त्रुटियाँ सामने आती हैं लेकिन स्थानीय रनटाइम जारी रहता है
-- शेड्यूलर में पुनः प्रयास-सक्षम तर्क है, लेकिन आवधिक निष्पादन वर्तमान में डिफ़ॉल्ट रूप से एकल-प्रयास सिंक को कॉल करता है## 5) Data Integrity
+## 3) Stream Safety
-- स्टार्टअप पर SQLite स्कीमा माइग्रेशन और ऑटो-अपग्रेड हुक
-- लीगेसी JSON → SQLite माइग्रेशन संगतता पथ## Observability and Operational Signals
+- disconnect-aware stream controller
+- translation stream with end-of-stream flush and `[DONE]` handling
+- usage estimation fallback when provider usage metadata is missing
-रनटाइम दृश्यता स्रोत:
+## 4) Cloud Sync Degradation
-- कंसोल `src/sse/utils/logger.ts` से लॉग करता है
-- SQLite में प्रति-अनुरोध उपयोग समुच्चय (`use_history`, `call_logs`, `proxy_logs`)
-- जब `settings.detailed_logs_enabled=true` होता है तो SQLite (`request_detail_logs`) में चार चरण वाला विस्तृत पेलोड कैप्चर होता है
-- पाठ्य अनुरोध स्थिति लॉग इन `log.txt` (वैकल्पिक/कॉम्पैट)
-- `ENABLE_REQUEST_LOGS=true` होने पर `लॉग/` के अंतर्गत वैकल्पिक गहन अनुरोध/अनुवाद लॉग
-- यूआई खपत के लिए डैशबोर्ड उपयोग समापन बिंदु (`/api/usage/*`)।
+- sync errors are surfaced but local runtime continues
+- scheduler has retry-capable logic, but periodic execution currently calls single-attempt sync by default
-विस्तृत अनुरोध पेलोड कैप्चर प्रति रूटेड कॉल को चार JSON पेलोड चरणों तक संग्रहीत करता है:
+## 5) Data Integrity
-- ग्राहक से प्राप्त कच्चा अनुरोध
-- अनुवादित अनुरोध वास्तव में अपस्ट्रीम भेजा गया
-- प्रदाता प्रतिक्रिया JSON के रूप में पुनर्निर्मित; स्ट्रीम की गई प्रतिक्रियाओं को अंतिम सारांश और स्ट्रीम मेटाडेटा में संकलित किया जाता है
-- ओम्निरूट द्वारा लौटाई गई अंतिम ग्राहक प्रतिक्रिया; स्ट्रीम की गई प्रतिक्रियाएँ उसी संक्षिप्त सारांश रूप में संग्रहीत की जाती हैं## Security-Sensitive Boundaries
+- SQLite schema migrations and auto-upgrade hooks at startup
+- legacy JSON → SQLite migration compatibility path
-- JWT सीक्रेट (`JWT_SECRET`) डैशबोर्ड सत्र कुकी सत्यापन/हस्ताक्षर को सुरक्षित करता है
-- प्रारंभिक पासवर्ड बूटस्ट्रैप (`INITIAL_PASSWORD`) को प्रथम-रन प्रावधान के लिए स्पष्ट रूप से कॉन्फ़िगर किया जाना चाहिए
-- एपीआई कुंजी HMAC सीक्रेट (`API_KEY_SECRET`) उत्पन्न स्थानीय एपीआई कुंजी प्रारूप को सुरक्षित करता है
-- प्रदाता रहस्य (एपीआई कुंजी/टोकन) स्थानीय डीबी में बने रहते हैं और उन्हें फ़ाइल सिस्टम स्तर पर संरक्षित किया जाना चाहिए
-- क्लाउड सिंक एंडपॉइंट एपीआई कुंजी ऑथ + मशीन आईडी सेमेन्टिक्स पर निर्भर करते हैं## Environment and Runtime Matrix
+## Observability and Operational Signals
-कोड द्वारा सक्रिय रूप से उपयोग किए जाने वाले पर्यावरण चर:
+Runtime visibility sources:
-- ऐप/ऑथ: `JWT_SECRET`, `INITIAL_PASSWORD`
-- भंडारण: `DATA_DIR`
-- संगत नोड व्यवहार: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
-- वैकल्पिक स्टोरेज बेस ओवरराइड (Linux/macOS जब `DATA_DIR` अनसेट होता है): `XDG_CONFIG_HOME`
-- सुरक्षा हैशिंग: `API_KEY_SECRET`, `MACHINE_ID_SALT`
-- लॉगिंग: `ENABLE_REQUEST_LOGS`
-- सिंक/क्लाउड यूआरएल: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
-- आउटबाउंड प्रॉक्सी: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` और लोअरकेस वेरिएंट
-- SOCKS5 फ़ीचर फ़्लैग: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
-- प्लेटफ़ॉर्म/रनटाइम सहायक (ऐप-विशिष्ट कॉन्फ़िगरेशन नहीं): `एप्लिकेशन डेटा`, `NODE_ENV`, `पोर्ट`, `होस्टनाम`## Known Architectural Notes
+- console logs from `src/sse/utils/logger.ts`
+- per-request usage aggregates in SQLite (`usage_history`, `call_logs`, `proxy_logs`)
+- four-stage detailed payload captures in SQLite (`request_detail_logs`) when `settings.detailed_logs_enabled=true`
+- textual request status log in `log.txt` (optional/compat)
+- optional deep request/translation logs under `logs/` when `ENABLE_REQUEST_LOGS=true`
+- dashboard usage endpoints (`/api/usage/*`) for UI consumption
-1. `usageDb` और `localDb` लीगेसी फ़ाइल माइग्रेशन के साथ समान आधार निर्देशिका नीति (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) साझा करते हैं।
-2. `/api/v1/route.ts` सिमेंटिक बहाव से बचने के लिए `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) द्वारा उपयोग किए जाने वाले समान एकीकृत कैटलॉग बिल्डर को सौंपता है।
-3. अनुरोध लकड़हारा सक्षम होने पर पूर्ण हेडर/बॉडी लिखता है; लॉग निर्देशिका को संवेदनशील मानें।
-4. क्लाउड व्यवहार सही `NEXT_PUBLIC_BASE_URL` और क्लाउड एंडपॉइंट रीचैबिलिटी पर निर्भर करता है।
-5. `ओपन-एसएसई/` निर्देशिका को `@omniroute/ओपन-एसएसई`**एनपीएम वर्कस्पेस पैकेज**के रूप में प्रकाशित किया गया है। स्रोत कोड इसे `@omniroute/open-sse/...` (Next.js `transpilePackages` द्वारा हल) के माध्यम से आयात करता है। इस दस्तावेज़ में फ़ाइल पथ अभी भी स्थिरता के लिए निर्देशिका नाम `open-sse/` का उपयोग करते हैं।
-6. डैशबोर्ड में चार्ट सुलभ, इंटरैक्टिव एनालिटिक्स विज़ुअलाइज़ेशन (मॉडल उपयोग बार चार्ट, सफलता दर के साथ प्रदाता ब्रेकडाउन टेबल) के लिए**रिचार्ट्स**(एसवीजी-आधारित) का उपयोग करते हैं।
-7. E2E परीक्षण**Playwright**(`test/e2e/`) का उपयोग करते हैं, `npm run test:e2e` के माध्यम से चलते हैं। यूनिट परीक्षण**नोड.जेएस टेस्ट रनर**(`टेस्ट/यूनिट/`) का उपयोग करते हैं, जो `एनपीएम रन टेस्ट: यूनिट` के माध्यम से चलते हैं। `src/` के अंतर्गत स्रोत कोड**टाइपस्क्रिप्ट**(`.ts`/`.tsx`) है; `ओपन-एसएसई/` कार्यक्षेत्र जावास्क्रिप्ट (`.जेएस`) बना हुआ है।
-8. सेटिंग्स पृष्ठ को 5 टैब में व्यवस्थित किया गया है: सुरक्षा, रूटिंग (6 वैश्विक रणनीतियाँ: भरण-प्रथम, राउंड-रॉबिन, पी2सी, यादृच्छिक, कम से कम उपयोग किया गया, लागत-अनुकूलित), लचीलापन (संपादन योग्य दर सीमा, सर्किट ब्रेकर, नीतियां), एआई (सोच बजट, सिस्टम प्रॉम्प्ट, प्रॉम्प्ट कैश), उन्नत (प्रॉक्सी)।## Operational Verification Checklist
+Detailed request payload capture stores up to four JSON payload stages per routed call:
-- स्रोत से निर्माण: `एनपीएम रन बिल्ड`
-- डॉकर छवि बनाएँ: `docker build -t omniroute।`
-- सेवा प्रारंभ करें और सत्यापित करें:
-- `प्राप्त करें /एपीआई/सेटिंग्स`
-- `प्राप्त करें /api/v1/मॉडल`
-- जब `PORT=20128` हो तो CLI लक्ष्य आधार URL `http://:20128/v1` होना चाहिए
+- raw request received from the client
+- translated request actually sent upstream
+- provider response reconstructed as JSON; streamed responses are compacted to the final summary plus stream metadata
+- final client response returned by OmniRoute; streamed responses are stored in the same compact summary form
+
+## Security-Sensitive Boundaries
+
+- JWT secret (`JWT_SECRET`) secures dashboard session cookie verification/signing
+- Initial password bootstrap (`INITIAL_PASSWORD`) should be explicitly configured for first-run provisioning
+- API key HMAC secret (`API_KEY_SECRET`) secures generated local API key format
+- Provider secrets (API keys/tokens) are persisted in local DB and should be protected at filesystem level
+- Cloud sync endpoints rely on API key auth + machine id semantics
+
+## Environment and Runtime Matrix
+
+Environment variables actively used by code:
+
+- App/auth: `JWT_SECRET`, `INITIAL_PASSWORD`
+- Storage: `DATA_DIR`
+- Compatible node behavior: `ALLOW_MULTI_CONNECTIONS_PER_COMPAT_NODE`
+- Optional storage base override (Linux/macOS when `DATA_DIR` unset): `XDG_CONFIG_HOME`
+- Security hashing: `API_KEY_SECRET`, `MACHINE_ID_SALT`
+- Logging: `ENABLE_REQUEST_LOGS`
+- Sync/cloud URLing: `NEXT_PUBLIC_BASE_URL`, `NEXT_PUBLIC_CLOUD_URL`
+- Outbound proxy: `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, `NO_PROXY` and lowercase variants
+- SOCKS5 feature flags: `ENABLE_SOCKS5_PROXY`, `NEXT_PUBLIC_ENABLE_SOCKS5_PROXY`
+- Platform/runtime helpers (not app-specific config): `APPDATA`, `NODE_ENV`, `PORT`, `HOSTNAME`
+
+## Known Architectural Notes
+
+1. `usageDb` and `localDb` share the same base directory policy (`DATA_DIR` -> `XDG_CONFIG_HOME/omniroute` -> `~/.omniroute`) with legacy file migration.
+2. `/api/v1/route.ts` delegates to the same unified catalog builder used by `/api/v1/models` (`src/app/api/v1/models/catalog.ts`) to avoid semantic drift.
+3. Request logger writes full headers/body when enabled; treat log directory as sensitive.
+4. Cloud behavior depends on correct `NEXT_PUBLIC_BASE_URL` and cloud endpoint reachability.
+5. The `open-sse/` directory is published as the `@omniroute/open-sse` **npm workspace package**. Source code imports it via `@omniroute/open-sse/...` (resolved by Next.js `transpilePackages`). File paths in this document still use the directory name `open-sse/` for consistency.
+6. Charts in the dashboard use **Recharts** (SVG-based) for accessible, interactive analytics visualizations (model usage bar charts, provider breakdown tables with success rates).
+7. E2E tests use **Playwright** (`tests/e2e/`), run via `npm run test:e2e`. Unit tests use **Node.js test runner** (`tests/unit/`), run via `npm run test:unit`. Source code under `src/` is **TypeScript** (`.ts`/`.tsx`); the `open-sse/` workspace remains JavaScript (`.js`).
+8. Settings page is organized into 5 tabs: Security, Routing (6 global strategies: fill-first, round-robin, p2c, random, least-used, cost-optimized), Resilience (editable rate limits, circuit breaker, policies, **Context Relay** handoff config), AI (thinking budget, system prompt, prompt cache), Advanced (proxy).
+9. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated, `chat.ts` injects the handoff after account resolution. Handoff data lives in `context_handoffs` SQLite table. This split is intentional because only `chat.ts` knows whether the actual account changed.
+10. **Proxy enforcement** is now comprehensive: `tokenHealthCheck.ts` resolves proxy per connection, `/api/providers/validate` uses `runWithProxyContext`, and `proxyFetch.ts` uses `undici.fetch()` to maintain dispatcher compatibility on Node 22.
+11. **Node.js 24+ detection**: `/api/settings/require-login` returns `nodeVersion` and `nodeCompatible` fields. The login page renders a warning banner when the runtime is incompatible.
+
+## Operational Verification Checklist
+
+- Build from source: `npm run build`
+- Build Docker image: `docker build -t omniroute .`
+- Start service and verify:
+- `GET /api/settings`
+- `GET /api/v1/models`
+- CLI target base URL should be `http://:20128/v1` when `PORT=20128`
diff --git a/docs/i18n/hi/docs/FEATURES.md b/docs/i18n/hi/docs/FEATURES.md
index 7cce5a61c0..46627da387 100644
--- a/docs/i18n/hi/docs/FEATURES.md
+++ b/docs/i18n/hi/docs/FEATURES.md
@@ -4,102 +4,168 @@
---
-ओमनीरूट डैशबोर्ड के प्रत्येक अनुभाग के लिए विज़ुअल गाइड।---
+
+
+Visual guide to every section of the OmniRoute dashboard.
+
+---
## 🔌 Providers
-एआई प्रदाता कनेक्शन प्रबंधित करें: OAuth प्रदाता (क्लाउड कोड, कोडेक्स, जेमिनी सीएलआई), एपीआई कुंजी प्रदाता (ग्रोक, डीपसीक, ओपनराउटर), और मुफ्त प्रदाता (क्यूडर, क्वेन, किरो)। किरो खातों में क्रेडिट बैलेंस ट्रैकिंग - शेष क्रेडिट, कुल भत्ता और डैशबोर्ड → उपयोग में दिखाई देने वाली नवीनीकरण तिथि शामिल है।
+Manage AI provider connections: OAuth providers (Claude Code, Codex, Gemini CLI), API key providers (Groq, DeepSeek, OpenRouter), and free providers (Qoder, Qwen, Kiro). Kiro accounts include credit balance tracking — remaining credits, total allowance, and renewal date visible in Dashboard → Usage.
+
+
---
## 🎨 Combos
-6 रणनीतियों के साथ मॉडल रूटिंग कॉम्बो बनाएं: प्राथमिकता, भारित, राउंड-रॉबिन, यादृच्छिक, कम से कम उपयोग किया गया और लागत-अनुकूलित। प्रत्येक कॉम्बो स्वचालित फ़ॉलबैक के साथ कई मॉडलों को श्रृंखलाबद्ध करता है और इसमें त्वरित टेम्पलेट और तत्परता जांच शामिल होती है।
+Create model routing combos with 13 strategies: priority, weighted, round-robin, random, least-used, cost-optimized, strict-random, auto, fill-first, p2c, lkgp, context-optimized, and **context-relay**. Each combo chains multiple models with automatic fallback and includes quick templates and readiness checks.
+
+
---
## 📊 Analytics
-टोकन खपत, लागत अनुमान, गतिविधि हीटमैप, साप्ताहिक वितरण चार्ट और प्रति-प्रदाता विश्लेषण के साथ व्यापक उपयोग विश्लेषण।
+Comprehensive usage analytics with token consumption, cost estimates, activity heatmaps, weekly distribution charts, and per-provider breakdowns.
+
+
---
## 🏥 System Health
-वास्तविक समय की निगरानी: अपटाइम, मेमोरी, संस्करण, विलंबता प्रतिशत (p50/p95/p99), कैश आँकड़े, और प्रदाता सर्किट ब्रेकर स्थिति।
+Real-time monitoring: uptime, memory, version, latency percentiles (p50/p95/p99), cache statistics, and provider circuit breaker states.
+
+
---
## 🔧 Translator Playground
-एपीआई अनुवादों को डीबग करने के लिए चार मोड:**प्लेग्राउंड**(फॉर्मेट कनवर्टर),**चैट टेस्टर**(लाइव अनुरोध),**टेस्ट बेंच**(बैच टेस्ट), और**लाइव मॉनिटर**(रियल-टाइम स्ट्रीम)।
+Four modes for debugging API translations: **Playground** (format converter), **Chat Tester** (live requests), **Test Bench** (batch tests), and **Live Monitor** (real-time stream).
+
+
---
## 🎮 Model Playground _(v2.0.9+)_
-किसी भी मॉडल का सीधे डैशबोर्ड से परीक्षण करें। प्रदाता, मॉडल और समापन बिंदु का चयन करें, मोनाको संपादक के साथ संकेत लिखें, वास्तविक समय में प्रतिक्रियाओं को स्ट्रीम करें, मध्य-धारा को निरस्त करें, और समय मेट्रिक्स देखें।---
+Test any model directly from the dashboard. Select provider, model, and endpoint, write prompts with Monaco Editor, stream responses in real-time, abort mid-stream, and view timing metrics.
+
+---
## 🎨 Themes _(v2.0.5+)_
-संपूर्ण डैशबोर्ड के लिए अनुकूलन योग्य रंग थीम। 7 पूर्व निर्धारित रंगों (कोरल, नीला, लाल, हरा, बैंगनी, नारंगी, सियान) में से चुनें या कोई भी हेक्स रंग चुनकर एक कस्टम थीम बनाएं। प्रकाश, अंधेरा और सिस्टम मोड का समर्थन करता है।---
+Customizable color themes for the entire dashboard. Choose from 7 preset colors (Coral, Blue, Red, Green, Violet, Orange, Cyan) or create a custom theme by picking any hex color. Supports light, dark, and system mode.
+
+---
## ⚙️ Settings
-टैब के साथ व्यापक सेटिंग पैनल:
+Comprehensive settings panel with tabs:
--**सामान्य**- सिस्टम स्टोरेज, बैकअप प्रबंधन (निर्यात/आयात डेटाबेस) -**प्रकटन**- थीम चयनकर्ता (गहरा/प्रकाश/सिस्टम), रंग थीम प्रीसेट और कस्टम रंग, स्वास्थ्य लॉग दृश्यता, साइडबार आइटम दृश्यता नियंत्रण -**सुरक्षा**- एपीआई एंडपॉइंट सुरक्षा, कस्टम प्रदाता अवरोधन, आईपी फ़िल्टरिंग, सत्र जानकारी -**रूटिंग**- मॉडल उपनाम, पृष्ठभूमि कार्य गिरावट -**लचीलापन**- दर सीमा दृढ़ता, सर्किट ब्रेकर ट्यूनिंग, प्रतिबंधित खातों को स्वचालित रूप से अक्षम करें, प्रदाता समाप्ति की निगरानी -**उन्नत**- कॉन्फ़िगरेशन ओवरराइड, कॉन्फ़िगरेशन ऑडिट ट्रेल, फ़ॉलबैक डिग्रेडेशन मोड
+- **General** — System storage, backup management (export/import database)
+- **Appearance** — Theme selector (dark/light/system), color theme presets and custom colors, health log visibility, sidebar item visibility controls
+- **Security** — API endpoint protection, custom provider blocking, IP filtering, session info
+- **Routing** — Model aliases, background task degradation
+- **Resilience** — Rate limit persistence, circuit breaker tuning, auto-disable banned accounts, provider expiration monitoring, **Context Relay** handoff threshold and summary model configuration
+- **Advanced** — Configuration overrides, configuration audit trail, fallback degradation mode
+
+
---
## 🔧 CLI Tools
-एआई कोडिंग टूल के लिए एक-क्लिक कॉन्फ़िगरेशन: क्लाउड कोड, कोडेक्स सीएलआई, जेमिनी सीएलआई, ओपनक्लाव, किलो कोड, एंटीग्रेविटी, क्लाइन, कंटिन्यू, कर्सर और फैक्ट्री ड्रॉयड। सुविधाएँ स्वचालित कॉन्फ़िगरेशन लागू/रीसेट, कनेक्शन प्रोफ़ाइल और मॉडल मैपिंग।
+One-click configuration for AI coding tools: Claude Code, Codex CLI, Gemini CLI, OpenClaw, Kilo Code, Antigravity, Cline, Continue, Cursor, and Factory Droid. Features automated config apply/reset, connection profiles, and model mapping.
+
+
---
## 🤖 CLI Agents _(v2.0.11+)_
-सीएलआई एजेंटों की खोज और प्रबंधन के लिए डैशबोर्ड। 14 अंतर्निहित एजेंटों (कोडेक्स, क्लाउड, गूज़, जेमिनी सीएलआई, ओपनक्लाव, एडर, ओपनकोड, क्लाइन, क्वेन कोड, फोर्जकोड, अमेज़ॅन क्यू, ओपन इंटरप्रेटर, कर्सर सीएलआई, वार्प) का ग्रिड दिखाता है:
+Dashboard for discovering and managing CLI agents. Shows a grid of 14 built-in agents (Codex, Claude, Goose, Gemini CLI, OpenClaw, Aider, OpenCode, Cline, Qwen Code, ForgeCode, Amazon Q, Open Interpreter, Cursor CLI, Warp) with:
--**इंस्टॉलेशन स्थिति**- संस्करण पहचान के साथ स्थापित / नहीं मिला -**Protocol badges**— stdio, HTTP, etc. -**कस्टम एजेंट**- फॉर्म के माध्यम से किसी भी सीएलआई टूल को पंजीकृत करें (नाम, बाइनरी, वर्जन कमांड, स्पॉन आर्ग्स) -**सीएलआई फ़िंगरप्रिंट मिलान**- मूल सीएलआई अनुरोध हस्ताक्षरों से मिलान करने के लिए प्रति-प्रदाता टॉगल करता है, प्रॉक्सी आईपी को संरक्षित करते हुए प्रतिबंध जोखिम को कम करता है---
+- **Installation status** — Installed / Not Found with version detection
+- **Protocol badges** — stdio, HTTP, etc.
+- **Custom agents** — Register any CLI tool via form (name, binary, version command, spawn args)
+- **CLI Fingerprint Matching** — Per-provider toggle to match native CLI request signatures, reducing ban risk while preserving proxy IP
+
+---
+
+## 🔗 Context Relay _(v3.5.5+)_
+
+A combo strategy that preserves session continuity when account rotation happens mid-conversation. Before the active account is exhausted, OmniRoute generates a structured handoff summary in the background. After the next request resolves to a different account, the summary is injected as a system message so the new account continues with full context.
+
+Configurable via combo-level or global settings:
+- **Handoff Threshold** — Quota usage percentage that triggers summary generation (default 85%)
+- **Max Messages For Summary** — How much recent history to condense
+- **Summary Model** — Optional override model for generating the handoff summary
+
+Currently supports Codex account rotation. See [Context Relay documentation](features/context-relay.md).
+
+---
+
+## 🛡️ Proxy Hardening _(v3.5.5+)_
+
+Comprehensive proxy configuration enforcement across the entire request pipeline:
+
+- **Token Health Check** — Background OAuth refresh now resolves proxy config per connection, preventing failures in proxy-required environments
+- **API Key Validation** — Provider key validation (`POST /api/providers/validate`) routes through `runWithProxyContext`, honoring provider-level and global proxy settings
+- **undici Dispatcher Fix** — Proxy dispatchers use undici's own fetch implementation instead of Node's built-in fetch, resolving `invalid onRequestStart method` errors on Node.js 22
+- **Node.js Version Detection** — Login page proactively detects incompatible Node.js versions (24+) and displays a warning banner with instructions to use Node 22 LTS
+
+---
## 🖼️ Media _(v2.0.3+)_
-डैशबोर्ड से चित्र, वीडियो और संगीत उत्पन्न करें। OpenAI, xAI, टुगेदर, हाइपरबोलिक, SD WebUI, ComfyUI, AnimateDiff, स्टेबल ऑडियो ओपन और MusicGen को सपोर्ट करता है।---
+Generate images, videos, and music from the dashboard. Supports OpenAI, xAI, Together, Hyperbolic, SD WebUI, ComfyUI, AnimateDiff, Stable Audio Open, and MusicGen.
+
+---
## 📝 Request Logs
-प्रदाता, मॉडल, खाता और एपीआई कुंजी द्वारा फ़िल्टरिंग के साथ वास्तविक समय अनुरोध लॉगिंग। स्थिति कोड, टोकन उपयोग, विलंबता और प्रतिक्रिया विवरण दिखाता है।
+Real-time request logging with filtering by provider, model, account, and API key. Shows status codes, token usage, latency, and response details.
+
+
---
## 🌐 API Endpoint
-क्षमता विश्लेषण के साथ आपका एकीकृत एपीआई एंडपॉइंट: चैट पूर्णताएं, प्रतिक्रिया एपीआई, एंबेडिंग, छवि निर्माण, रीरैंकिंग, ऑडियो ट्रांसक्रिप्शन, टेक्स्ट-टू-स्पीच, मॉडरेशन और पंजीकृत एपीआई कुंजी। रिमोट एक्सेस के लिए क्लाउडफ्लेयर क्विक टनल इंटीग्रेशन और क्लाउड प्रॉक्सी सपोर्ट।
+Your unified API endpoint with capability breakdown: Chat Completions, Responses API, Embeddings, Image Generation, Reranking, Audio Transcription, Text-to-Speech, Moderations, and registered API keys. Cloudflare Quick Tunnel integration and cloud proxy support for remote access.
+
+
---
## 🔑 API Key Management
-एपीआई कुंजी बनाएं, दायरा बढ़ाएं और निरस्त करें। प्रत्येक कुंजी को पूर्ण पहुंच या केवल-पढ़ने की अनुमति वाले विशिष्ट मॉडल/प्रदाताओं तक सीमित किया जा सकता है। उपयोग ट्रैकिंग के साथ विज़ुअल कुंजी प्रबंधन।---
+Create, scope, and revoke API keys. Each key can be restricted to specific models/providers with full access or read-only permissions. Visual key management with usage tracking.
+
+---
## 📋 Audit Log
-कार्रवाई प्रकार, अभिनेता, लक्ष्य, आईपी पता और टाइमस्टैम्प द्वारा फ़िल्टरिंग के साथ प्रशासनिक कार्रवाई ट्रैकिंग। पूर्ण सुरक्षा घटना इतिहास.---
+Administrative action tracking with filtering by action type, actor, target, IP address, and timestamp. Full security event history.
+
+---
## 🖥️ Desktop Application
-विंडोज़, मैकओएस और लिनक्स के लिए नेटिव इलेक्ट्रॉन डेस्कटॉप ऐप। सिस्टम ट्रे एकीकरण, ऑफ़लाइन समर्थन, ऑटो-अपडेट और एक-क्लिक इंस्टॉल के साथ ओमनीरूट को एक स्टैंडअलोन एप्लिकेशन के रूप में चलाएं।
+Native Electron desktop app for Windows, macOS, and Linux. Run OmniRoute as a standalone application with system tray integration, offline support, auto-update, and one-click install.
-मुख्य विशेषताएं:
+Key features:
-- सर्वर तत्परता मतदान (कोल्ड स्टार्ट पर कोई खाली स्क्रीन नहीं)
-- पोर्ट प्रबंधन के साथ सिस्टम ट्रे
-- सामग्री सुरक्षा नीति
-- सिंगल-इंस्टेंस लॉक
-- पुनरारंभ पर स्वतः अद्यतन
-- प्लेटफ़ॉर्म-सशर्त यूआई (मैकओएस ट्रैफिक लाइट, विंडोज़/लिनक्स डिफ़ॉल्ट टाइटलबार)
-- कठोर इलेक्ट्रॉन बिल्ड पैकेजिंग - स्टैंडअलोन बंडल में सिम्लिंक्ड `नोड_मॉड्यूल` का पता लगाया जाता है और पैकेजिंग से पहले खारिज कर दिया जाता है, जिससे बिल्ड मशीन पर रनटाइम निर्भरता को रोका जा सकता है (v2.5.5+)
+- Server readiness polling (no blank screen on cold start)
+- System tray with port management
+- Content Security Policy
+- Single-instance lock
+- Auto-update on restart
+- Platform-conditional UI (macOS traffic lights, Windows/Linux default titlebar)
+- Hardened Electron build packaging — symlinked `node_modules` in the standalone bundle is detected and rejected before packaging, preventing runtime dependency on the build machine (v2.5.5+)
-📖 संपूर्ण दस्तावेज़ीकरण के लिए [`electron/README.md`](../electron/README.md) देखें।
+📖 See [`electron/README.md`](../electron/README.md) for full documentation.
diff --git a/docs/i18n/hi/docs/TROUBLESHOOTING.md b/docs/i18n/hi/docs/TROUBLESHOOTING.md
index 5a582de15d..5fb25a9ebb 100644
--- a/docs/i18n/hi/docs/TROUBLESHOOTING.md
+++ b/docs/i18n/hi/docs/TROUBLESHOOTING.md
@@ -4,68 +4,142 @@
---
-ओम्निरूट के लिए सामान्य समस्याएं और समाधान।---
+
+
+Common problems and solutions for OmniRoute.
+
+---
## Quick Fixes
-| समस्या | समाधान |
-| ------------------------------------- | ------------------------------------------------------------------------------- | --- |
-| पहला लॉगिन काम नहीं कर रहा | `INITIAL_PASSWORD` को `.env` में सेट करें (कोई हार्डकोडेड डिफ़ॉल्ट नहीं) |
-| गलत पोर्ट पर डैशबोर्ड खुलता है | `PORT=20128` और `NEXT_PUBLIC_BASE_URL=http://localhost:20128` सेट करें |
-| `लॉग/` के अंतर्गत कोई अनुरोध लॉग नहीं | `ENABLE_REQUEST_LOGS=true` सेट करें |
-| EACCES: अनुमति अस्वीकृत | `~/.omniroute` को ओवरराइड करने के लिए `DATA_DIR=/path/to/writable/dir` सेट करें |
-| रूटिंग रणनीति सहेजी नहीं जा रही | v1.4.11+ पर अपडेट करें (सेटिंग्स दृढ़ता के लिए ज़ोड स्कीमा फिक्स) | --- |
+| Problem | Solution |
+| ----------------------------- | ------------------------------------------------------------------ |
+| First login not working | Set `INITIAL_PASSWORD` in `.env` (no hardcoded default) |
+| Dashboard opens on wrong port | Set `PORT=20128` and `NEXT_PUBLIC_BASE_URL=http://localhost:20128` |
+| No request logs under `logs/` | Set `ENABLE_REQUEST_LOGS=true` |
+| EACCES: permission denied | Set `DATA_DIR=/path/to/writable/dir` to override `~/.omniroute` |
+| Routing strategy not saving | Update to v1.4.11+ (Zod schema fix for settings persistence) |
+| Login crash / blank page | You may be on Node.js 24+ — see [Node.js Compatibility](#nodejs-compatibility) below |
+| Proxy "fetch failed" | Ensure proxy config is set at the correct level — see [Proxy Issues](#proxy-issues) below |
+
+---
+
+## Node.js Compatibility
+
+
+
+### Login page crashes or shows "Module self-registration" error
+
+**Cause:** You are running Node.js 24+. The `better-sqlite3` native binary is not compatible with Node.js 24, which causes a fatal crash when the server tries to initialize the database.
+
+**Symptoms:**
+- Login page shows a blank screen or a server error
+- Console shows `Error: Module did not self-register` or similar native binding errors
+- Starting with v3.5.5, the login page shows an **orange warning banner** with your Node version if incompatibility is detected
+
+**Fix:**
+
+1. Install Node.js 22 LTS (recommended):
+ ```bash
+ nvm install 22
+ nvm use 22
+ ```
+2. Verify your version: `node --version` should show `v22.x.x`
+3. Reinstall OmniRoute: `npm install -g omniroute`
+4. Restart: `omniroute`
+
+> **Supported versions:** Node.js 18, 20, or 22 LTS. Node.js 24+ is **not supported**.
+
+---
+
+## Proxy Issues
+
+
+
+### Provider validation shows "fetch failed"
+
+**Cause:** The API key validation endpoint (`POST /api/providers/validate`) was previously bypassing proxy configuration, causing failures in environments that require proxy routing.
+
+**Fix (v3.5.5+):** This is now fixed. Provider validation routes through `runWithProxyContext`, honoring provider-level and global proxy settings automatically.
+
+### Token health check fails with "fetch failed"
+
+**Cause:** Background OAuth token refresh was not resolving proxy configuration per connection.
+
+**Fix (v3.5.5+):** The token health check scheduler now resolves proxy config per connection before attempting refresh. Update to v3.5.5+.
+
+### SOCKS5 proxy returns "invalid onRequestStart method"
+
+**Cause:** On Node.js 22, the undici@8 dispatcher is incompatible with Node's built-in `fetch()` implementation.
+
+**Fix (v3.5.5+):** OmniRoute now uses undici's own `fetch()` function when a proxy dispatcher is active, ensuring consistent behavior. Update to v3.5.5+.
+
+---
## Provider Issues
### "Language model did not provide messages"
-**कारण:**प्रदाता कोटा समाप्त हो गया।
+**Cause:** Provider quota exhausted.
-**ठीक करें:**
+**Fix:**
-1. डैशबोर्ड कोटा ट्रैकर की जाँच करें
-2. फ़ॉलबैक टियर वाले कॉम्बो का उपयोग करें
-3. सस्ते/मुफ़्त स्तर पर स्विच करें### Rate Limiting
+1. Check dashboard quota tracker
+2. Use a combo with fallback tiers
+3. Switch to cheaper/free tier
-**कारण:**सदस्यता कोटा समाप्त हो गया।
+### Rate Limiting
-**ठीक करें:**
+**Cause:** Subscription quota exhausted.
-- फ़ॉलबैक जोड़ें: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-- सस्ते बैकअप के रूप में GLM/MiniMax का उपयोग करें### OAuth Token Expired
+**Fix:**
-ओम्निरूट स्वचालित रूप से टोकन ताज़ा करता है। यदि समस्याएँ बनी रहती हैं:
+- Add fallback: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
+- Use GLM/MiniMax as cheap backup
-1. डैशबोर्ड → प्रदाता → पुनः कनेक्ट करें
-2. प्रदाता कनेक्शन हटाएं और पुनः जोड़ें---
+### OAuth Token Expired
+
+OmniRoute auto-refreshes tokens. If issues persist:
+
+1. Dashboard → Provider → Reconnect
+2. Delete and re-add the provider connection
+
+---
## Cloud Issues
### Cloud Sync Errors
-1. अपने चल रहे उदाहरण के लिए `BASE_URL` बिंदुओं को सत्यापित करें (उदाहरण के लिए, `http://localhost:20128`)
-2. अपने क्लाउड एंडपॉइंट पर `CLOUD_URL` बिंदुओं को सत्यापित करें (उदाहरण के लिए, `https://omniroute.dev`)
-3. `NEXT_PUBLIC_*` मानों को सर्वर-साइड मानों के साथ संरेखित रखें### Cloud `stream=false` Returns 500
+1. Verify `BASE_URL` points to your running instance (e.g., `http://localhost:20128`)
+2. Verify `CLOUD_URL` points to your cloud endpoint (e.g., `https://omniroute.dev`)
+3. Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**लक्षण:**गैर-स्ट्रीमिंग कॉल के लिए क्लाउड एंडपॉइंट पर `अप्रत्याशित टोकन 'डी'...`।
+### Cloud `stream=false` Returns 500
-**कारण:**अपस्ट्रीम एसएसई पेलोड लौटाता है जबकि ग्राहक JSON की अपेक्षा करता है।
+**Symptom:** `Unexpected token 'd'...` on cloud endpoint for non-streaming calls.
-**समाधान:**क्लाउड डायरेक्ट कॉल के लिए `stream=true` का उपयोग करें। स्थानीय रनटाइम में SSE→JSON फ़ॉलबैक शामिल है।### Cloud Says Connected but "Invalid API key"
+**Cause:** Upstream returns SSE payload while client expects JSON.
-1. स्थानीय डैशबोर्ड से एक नई कुंजी बनाएं (`/api/keys`)
-2. क्लाउड सिंक चलाएँ: क्लाउड सक्षम करें → अभी सिंक करें
-3. पुरानी/गैर-सिंक की गई कुंजियाँ अभी भी क्लाउड पर `401` लौटा सकती हैं---
+**Workaround:** Use `stream=true` for cloud direct calls. Local runtime includes SSE→JSON fallback.
+
+### Cloud Says Connected but "Invalid API key"
+
+1. Create a fresh key from local dashboard (`/api/keys`)
+2. Run cloud sync: Enable Cloud → Sync Now
+3. Old/non-synced keys can still return `401` on cloud
+
+---
## Docker Issues
### CLI Tool Shows Not Installed
-1. रनटाइम फ़ील्ड जांचें: `कर्ल http://localhost:20128/api/cli-tools/runtime/codex | jq`
-2. पोर्टेबल मोड के लिए: छवि लक्ष्य `रनर-सीएलआई` (बंडल सीएलआई) का उपयोग करें
-3. होस्ट माउंट मोड के लिए: `CLI_EXTRA_PATHS` सेट करें और होस्ट बिन निर्देशिका को केवल पढ़ने के लिए माउंट करें
-4. यदि `इंस्टॉल = सही` और `रनने योग्य = गलत`: बाइनरी पाया गया था लेकिन स्वास्थ्य जांच विफल रही### Quick Runtime Validation
+1. Check runtime fields: `curl http://localhost:20128/api/cli-tools/runtime/codex | jq`
+2. For portable mode: use image target `runner-cli` (bundled CLIs)
+3. For host mount mode: set `CLI_EXTRA_PATHS` and mount host bin directory as read-only
+4. If `installed=true` and `runnable=false`: binary was found but failed healthcheck
+
+### Quick Runtime Validation
```bash
curl -s http://localhost:20128/api/cli-tools/codex-settings | jq '{installed,runnable,commandPath,runtimeMode,reason}'
@@ -79,16 +153,20 @@ curl -s http://localhost:20128/api/cli-tools/openclaw-settings | jq '{installed,
### High Costs
-1. डैशबोर्ड → उपयोग में उपयोग के आँकड़े जाँचें
-2. प्राथमिक मॉडल को जीएलएम/मिनीमैक्स पर स्विच करें
-3. गैर-महत्वपूर्ण कार्यों के लिए फ्री टियर (मिथुन सीएलआई, क्यूडर) का उपयोग करें
-4. प्रति एपीआई कुंजी लागत बजट निर्धारित करें: डैशबोर्ड → एपीआई कुंजी → बजट---
+1. Check usage stats in Dashboard → Usage
+2. Switch primary model to GLM/MiniMax
+3. Use free tier (Gemini CLI, Qoder) for non-critical tasks
+4. Set cost budgets per API key: Dashboard → API Keys → Budget
+
+---
## Debugging
### Enable Request Logs
-अपनी `.env` फ़ाइल में `ENABLE_REQUEST_LOGS=true` सेट करें। लॉग `लॉग/` निर्देशिका के अंतर्गत दिखाई देते हैं।### Check Provider Health
+Set `ENABLE_REQUEST_LOGS=true` in your `.env` file. Logs appear under `logs/` directory.
+
+### Check Provider Health
```bash
# Health dashboard
@@ -100,98 +178,135 @@ curl http://localhost:20128/api/monitoring/health
### Runtime Storage
-- मुख्य स्थिति: `${DATA_DIR}/storage.sqlite` (प्रदाता, कॉम्बो, उपनाम, कुंजियाँ, सेटिंग्स)
-- उपयोग: `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) में SQLite टेबल + वैकल्पिक `${DATA_DIR}/log.txt` और `${DATA_DIR}/call_logs/`
-- अनुरोध लॉग: `/logs/...` (जब `ENABLE_REQUEST_LOGS=true`)---
+- Main state: `${DATA_DIR}/storage.sqlite` (providers, combos, aliases, keys, settings)
+- Usage: SQLite tables in `storage.sqlite` (`usage_history`, `call_logs`, `proxy_logs`) + optional `${DATA_DIR}/log.txt` and `${DATA_DIR}/call_logs/`
+- Request logs: `/logs/...` (when `ENABLE_REQUEST_LOGS=true`)
+
+---
## Circuit Breaker Issues
### Provider stuck in OPEN state
-जब किसी प्रदाता का सर्किट ब्रेकर खुला होता है, तो कूलडाउन समाप्त होने तक अनुरोध अवरुद्ध हो जाते हैं।
+When a provider's circuit breaker is OPEN, requests are blocked until the cooldown expires.
-**ठीक करें:**
+**Fix:**
-1.**डैशबोर्ड → सेटिंग्स → लचीलापन**पर जाएं 2. प्रभावित प्रदाता के लिए सर्किट ब्रेकर कार्ड की जाँच करें 3. सभी ब्रेकर साफ़ करने के लिए**रीसेट ऑल**पर क्लिक करें, या कूलडाउन समाप्त होने तक प्रतीक्षा करें 4. रीसेट करने से पहले सत्यापित करें कि प्रदाता वास्तव में उपलब्ध है### Provider keeps tripping the circuit breaker
+1. Go to **Dashboard → Settings → Resilience**
+2. Check the circuit breaker card for the affected provider
+3. Click **Reset All** to clear all breakers, or wait for the cooldown to expire
+4. Verify the provider is actually available before resetting
-यदि कोई प्रदाता बार-बार खुली स्थिति में प्रवेश करता है:
+### Provider keeps tripping the circuit breaker
-1. विफलता पैटर्न के लिए**डैशबोर्ड → स्वास्थ्य → प्रदाता स्वास्थ्य**की जाँच करें 2.**सेटिंग्स → लचीलापन → प्रदाता प्रोफाइल**पर जाएं और विफलता सीमा बढ़ाएं
-2. जांचें कि क्या प्रदाता ने एपीआई सीमाएं बदल दी हैं या पुनः प्रमाणीकरण की आवश्यकता है
-3. विलंबता टेलीमेट्री की समीक्षा करें - उच्च विलंबता टाइमआउट-आधारित विफलताओं का कारण बन सकती है---
+If a provider repeatedly enters OPEN state:
+
+1. Check **Dashboard → Health → Provider Health** for the failure pattern
+2. Go to **Settings → Resilience → Provider Profiles** and increase the failure threshold
+3. Check if the provider has changed API limits or requires re-authentication
+4. Review latency telemetry — high latency may cause timeout-based failures
+
+---
## Audio Transcription Issues
### "Unsupported model" error
-- सुनिश्चित करें कि आप सही उपसर्ग का उपयोग कर रहे हैं: `डीपग्राम/नोवा-3` या `असेंबलीई/बेस्ट`
-- सत्यापित करें कि प्रदाता**डैशबोर्ड → प्रदाता**में जुड़ा हुआ है### Transcription returns empty or fails
+- Ensure you're using the correct prefix: `deepgram/nova-3` or `assemblyai/best`
+- Verify the provider is connected in **Dashboard → Providers**
-- समर्थित ऑडियो प्रारूप जांचें: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm`
-- सत्यापित करें कि फ़ाइल का आकार प्रदाता सीमा के भीतर है (आमतौर पर <25MB)
-- प्रदाता कार्ड में प्रदाता एपीआई कुंजी वैधता की जांच करें---
+### Transcription returns empty or fails
+
+- Check supported audio formats: `mp3`, `wav`, `m4a`, `flac`, `ogg`, `webm`
+- Verify file size is within provider limits (typically < 25MB)
+- Check provider API key validity in the provider card
+
+---
## Translator Debugging
-प्रारूप अनुवाद समस्याओं को डीबग करने के लिए**डैशबोर्ड → अनुवादक**का उपयोग करें:
+Use **Dashboard → Translator** to debug format translation issues:
-| मोड | कब उपयोग करें |
-| ---------------- | ------------------------------------------------------------------------------------------------------------- | ------------------------ |
-| **खेल का मैदान** | इनपुट/आउटपुट स्वरूपों की साथ-साथ तुलना करें - यह कैसे अनुवादित होता है यह देखने के लिए एक असफल अनुरोध चिपकाएँ |
-| **Chat Tester** | लाइव संदेश भेजें और हेडर सहित पूर्ण अनुरोध/प्रतिक्रिया पेलोड का निरीक्षण करें |
-| **टेस्ट बेंच** | यह पता लगाने के लिए कि कौन से अनुवाद टूटे हुए हैं, सभी प्रारूप संयोजनों में बैच परीक्षण चलाएँ |
-| **लाइव मॉनिटर** | रुक-रुक कर होने वाली अनुवाद समस्याओं को पकड़ने के लिए वास्तविक समय अनुरोध प्रवाह देखें | ### Common format issues |
+| Mode | When to Use |
+| ---------------- | -------------------------------------------------------------------------------------------- |
+| **Playground** | Compare input/output formats side by side — paste a failing request to see how it translates |
+| **Chat Tester** | Send live messages and inspect the full request/response payload including headers |
+| **Test Bench** | Run batch tests across format combinations to find which translations are broken |
+| **Live Monitor** | Watch real-time request flow to catch intermittent translation issues |
--**सोच टैग दिखाई नहीं दे रहे हैं**- जांचें कि क्या लक्ष्य प्रदाता सोच और सोच बजट सेटिंग का समर्थन करता है -**टूल कॉल ड्रॉपिंग**— कुछ प्रारूप अनुवाद असमर्थित फ़ील्ड को हटा सकते हैं; खेल का मैदान मोड में सत्यापित करें -**सिस्टम प्रॉम्प्ट गायब**— क्लाउड और जेमिनी हैंडल सिस्टम प्रॉम्प्ट अलग-अलग होते हैं; अनुवाद आउटपुट की जाँच करें -**एसडीके ऑब्जेक्ट के बजाय कच्ची स्ट्रिंग लौटाता है**- v1.1.0 में फिक्स्ड: रिस्पॉन्स सैनिटाइज़र अब गैर-मानक फ़ील्ड (`x_groq`, `usage_breakdown`, आदि) को हटा देता है जो OpenAI SDK पायडेंटिक सत्यापन विफलताओं का कारण बनता है -**GLM/ERNIE `सिस्टम' भूमिका को अस्वीकार करता है**- v1.1.0 में फिक्स्ड: रोल नॉर्मलाइज़र स्वचालित रूप से असंगत मॉडल के लिए सिस्टम संदेशों को उपयोगकर्ता संदेशों में मर्ज कर देता है -**'डेवलपर' की भूमिका पहचानी नहीं गई**- v1.1.0 में फिक्स्ड: गैर-ओपनएआई प्रदाताओं के लिए स्वचालित रूप से `सिस्टम' में कनवर्ट किया गया
--**`json_schema`जेमिनी के साथ काम नहीं कर रहा है**- v1.1.0 में फिक्स्ड:`response_format`को अब जेमिनी के`responseMimeType`+`responseSchema` में बदल दिया गया है---
+### Common format issues
+
+- **Thinking tags not appearing** — Check if the target provider supports thinking and the thinking budget setting
+- **Tool calls dropping** — Some format translations may strip unsupported fields; verify in Playground mode
+- **System prompt missing** — Claude and Gemini handle system prompts differently; check translation output
+- **SDK returns raw string instead of object** — Fixed in v1.1.0: response sanitizer now strips non-standard fields (`x_groq`, `usage_breakdown`, etc.) that cause OpenAI SDK Pydantic validation failures
+- **GLM/ERNIE rejects `system` role** — Fixed in v1.1.0: role normalizer automatically merges system messages into user messages for incompatible models
+- **`developer` role not recognized** — Fixed in v1.1.0: automatically converted to `system` for non-OpenAI providers
+- **`json_schema` not working with Gemini** — Fixed in v1.1.0: `response_format` is now converted to Gemini's `responseMimeType` + `responseSchema`
+
+---
## Resilience Settings
### Auto rate-limit not triggering
-- ऑटो दर-सीमा केवल एपीआई कुंजी प्रदाताओं पर लागू होती है (OAuth/सदस्यता पर नहीं)
-- सत्यापित करें**सेटिंग्स → लचीलापन → प्रदाता प्रोफाइल**में ऑटो-दर-सीमा सक्षम है
-- जांचें कि क्या प्रदाता `429` स्टेटस कोड या `रीट्री-आफ्टर` हेडर लौटाता है### Tuning exponential backoff
+- Auto rate-limit only applies to API key providers (not OAuth/subscription)
+- Verify **Settings → Resilience → Provider Profiles** has auto-rate-limit enabled
+- Check if the provider returns `429` status codes or `Retry-After` headers
-प्रदाता प्रोफ़ाइल इन सेटिंग्स का समर्थन करती हैं:
+### Tuning exponential backoff
--**आधार विलंब**— पहली विफलता के बाद प्रारंभिक प्रतीक्षा समय (डिफ़ॉल्ट: 1 सेकंड) -**अधिकतम विलंब**— अधिकतम प्रतीक्षा समय सीमा (डिफ़ॉल्ट: 30s) -**गुणक**- लगातार विफलता के बाद विलंब को कितना बढ़ाया जाए (डिफ़ॉल्ट: 2x)### Anti-thundering herd
+Provider profiles support these settings:
-जब कई समवर्ती अनुरोध एक दर-सीमित प्रदाता से टकराते हैं, तो ओमनीरूट अनुरोधों को क्रमबद्ध करने और कैस्केडिंग विफलताओं को रोकने के लिए म्यूटेक्स + ऑटो रेट-लिमिटिंग का उपयोग करता है। यह एपीआई कुंजी प्रदाताओं के लिए स्वचालित है।---
+- **Base delay** — Initial wait time after first failure (default: 1s)
+- **Max delay** — Maximum wait time cap (default: 30s)
+- **Multiplier** — How much to increase delay per consecutive failure (default: 2x)
+
+### Anti-thundering herd
+
+When many concurrent requests hit a rate-limited provider, OmniRoute uses mutex + auto rate-limiting to serialize requests and prevent cascading failures. This is automatic for API key providers.
+
+---
## Optional RAG / LLM failure taxonomy (16 problems)
-कुछ ओमनीरूट उपयोगकर्ता गेटवे को RAG या एजेंट स्टैक के सामने रखते हैं। उन सेटअपों में एक अजीब पैटर्न देखना आम है: ओम्नीरूट स्वस्थ दिखता है (प्रदाता ऊपर, रूटिंग प्रोफाइल ठीक, कोई दर सीमा अलर्ट नहीं) लेकिन अंतिम उत्तर अभी भी गलत है।
+Some OmniRoute users place the gateway in front of RAG or agent stacks. In those setups it is common to see a strange pattern: OmniRoute looks healthy (providers up, routing profiles ok, no rate limit alerts) but the final answer is still wrong.
-व्यवहार में ये घटनाएं आम तौर पर डाउनस्ट्रीम आरएजी पाइपलाइन से आती हैं, गेटवे से नहीं।
+In practice these incidents usually come from the downstream RAG pipeline, not from the gateway itself.
-यदि आप उन विफलताओं का वर्णन करने के लिए एक साझा शब्दावली चाहते हैं तो आप डब्लूएफजीवाई प्रॉब्लममैप का उपयोग कर सकते हैं, एक बाहरी एमआईटी लाइसेंस टेक्स्ट संसाधन जो सोलह आवर्ती आरएजी / एलएलएम विफलता पैटर्न को परिभाषित करता है। उच्च स्तर पर इसमें शामिल हैं:
+If you want a shared vocabulary to describe those failures you can use the WFGY ProblemMap, an external MIT license text resource that defines sixteen recurring RAG / LLM failure patterns. At a high level it covers:
-- पुनर्प्राप्ति बहाव और टूटी हुई संदर्भ सीमाएँ
-- खाली या बासी इंडेक्स और वेक्टर स्टोर
-- एम्बेडिंग बनाम सिमेंटिक बेमेल
-- शीघ्र असेंबली और संदर्भ विंडो समस्याएँ
-- तर्क पतन और अतिआत्मविश्वासपूर्ण उत्तर
-- लंबी श्रृंखला और एजेंट समन्वय विफलताएँ
-- मल्टी एजेंट मेमोरी और रोल ड्रिफ्ट
-- परिनियोजन और बूटस्ट्रैप ऑर्डरिंग समस्याएं
+- retrieval drift and broken context boundaries
+- empty or stale indexes and vector stores
+- embedding versus semantic mismatch
+- prompt assembly and context window issues
+- logic collapse and overconfident answers
+- long chain and agent coordination failures
+- multi agent memory and role drift
+- deployment and bootstrap ordering problems
-विचार सरल है:
+The idea is simple:
-1. जब आप किसी खराब प्रतिक्रिया की जांच करते हैं, तो कैप्चर करें:
- - उपयोगकर्ता कार्य और अनुरोध
- - ओमनीरूट में रूट या प्रदाता कॉम्बो
- - डाउनस्ट्रीम में उपयोग किया गया कोई भी RAG संदर्भ (पुनर्प्राप्त दस्तावेज़, टूल कॉल, आदि)
-2. घटना को एक या दो WFGY समस्या मानचित्र संख्याओं ('नंबर 1' ... 'नंबर 16') पर मैप करें।
-3. नंबर को अपने डैशबोर्ड, रनबुक, या घटना ट्रैकर में ओमनीरूट लॉग के बगल में संग्रहीत करें।
-4. यह तय करने के लिए संबंधित WFGY पृष्ठ का उपयोग करें कि आपको अपने RAG स्टैक, रिट्रीवर या रूटिंग रणनीति को बदलने की आवश्यकता है या नहीं।
+1. When you investigate a bad response, capture:
+ - user task and request
+ - route or provider combo in OmniRoute
+ - any RAG context used downstream (retrieved documents, tool calls, etc)
+2. Map the incident to one or two WFGY ProblemMap numbers (`No.1` … `No.16`).
+3. Store the number in your own dashboard, runbook, or incident tracker next to the OmniRoute logs.
+4. Use the corresponding WFGY page to decide whether you need to change your RAG stack, retriever, or routing strategy.
-पूर्ण पाठ और ठोस व्यंजन यहां उपलब्ध हैं (एमआईटी लाइसेंस, केवल पाठ):
+Full text and concrete recipes live here (MIT license, text only):
-[डब्ल्यूएफजीवाई प्रॉब्लममैप रीडमी](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md)
+[WFGY ProblemMap README](https://github.com/onestardao/WFGY/blob/main/ProblemMap/README.md)
-यदि आप ओमनीरूट के पीछे आरएजी या एजेंट पाइपलाइन नहीं चलाते हैं तो आप इस अनुभाग को अनदेखा कर सकते हैं।---
+You can ignore this section if you do not run RAG or agent pipelines behind OmniRoute.
+
+---
## Still Stuck?
--**गिटहब मुद्दे**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**आर्किटेक्चर**: आंतरिक विवरण के लिए [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) देखें -**एपीआई संदर्भ**: सभी समापन बिंदुओं के लिए [`docs/API_REFERENCE.md`](API_REFERENCE.md) देखें -**स्वास्थ्य डैशबोर्ड**: वास्तविक समय प्रणाली की स्थिति के लिए**डैशबोर्ड → स्वास्थ्य**जांचें -**अनुवादक**: प्रारूप संबंधी समस्याओं को डीबग करने के लिए**डैशबोर्ड → अनुवादक**का उपयोग करें
+- **GitHub Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **Architecture**: See [`docs/ARCHITECTURE.md`](ARCHITECTURE.md) for internal details
+- **API Reference**: See [`docs/API_REFERENCE.md`](API_REFERENCE.md) for all endpoints
+- **Health Dashboard**: Check **Dashboard → Health** for real-time system status
+- **Translator**: Use **Dashboard → Translator** to debug format issues
diff --git a/docs/i18n/hi/llm.txt b/docs/i18n/hi/llm.txt
new file mode 100644
index 0000000000..7a4c81f081
--- /dev/null
+++ b/docs/i18n/hi/llm.txt
@@ -0,0 +1,476 @@
+# OmniRoute (हिन्दी)
+
+🌐 **Languages:** 🇺🇸 [English](../../../llm.txt) · 🇪🇸 [es](../es/llm.txt) · 🇫🇷 [fr](../fr/llm.txt) · 🇩🇪 [de](../de/llm.txt) · 🇮🇹 [it](../it/llm.txt) · 🇷🇺 [ru](../ru/llm.txt) · 🇨🇳 [zh-CN](../zh-CN/llm.txt) · 🇯🇵 [ja](../ja/llm.txt) · 🇰🇷 [ko](../ko/llm.txt) · 🇸🇦 [ar](../ar/llm.txt) · 🇮🇳 [hi](../hi/llm.txt) · 🇮🇳 [in](../in/llm.txt) · 🇹🇭 [th](../th/llm.txt) · 🇻🇳 [vi](../vi/llm.txt) · 🇮🇩 [id](../id/llm.txt) · 🇲🇾 [ms](../ms/llm.txt) · 🇳🇱 [nl](../nl/llm.txt) · 🇵🇱 [pl](../pl/llm.txt) · 🇸🇪 [sv](../sv/llm.txt) · 🇳🇴 [no](../no/llm.txt) · 🇩🇰 [da](../da/llm.txt) · 🇫🇮 [fi](../fi/llm.txt) · 🇵🇹 [pt](../pt/llm.txt) · 🇷🇴 [ro](../ro/llm.txt) · 🇭🇺 [hu](../hu/llm.txt) · 🇧🇬 [bg](../bg/llm.txt) · 🇸🇰 [sk](../sk/llm.txt) · 🇺🇦 [uk-UA](../uk-UA/llm.txt) · 🇮🇱 [he](../he/llm.txt) · 🇵🇭 [phi](../phi/llm.txt) · 🇧🇷 [pt-BR](../pt-BR/llm.txt) · 🇨🇿 [cs](../cs/llm.txt) · 🇹🇷 [tr](../tr/llm.txt)
+
+---
+
+
+> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 60+ AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (25 tools), A2A v0.3 protocol, Memory/Skills systems, and an Electron desktop app.
+
+## अवलोकन
+
+OmniRoute solves the problem of managing multiple AI provider subscriptions, quotas, and rate limits. It sits between your AI-powered tools (IDE agents, CLI tools) and AI providers, routing requests intelligently through a 4-tier fallback system: Subscription → API Key → Cheap → Free.
+
+**Key value:** One endpoint (`http://localhost:20128/v1`), unlimited models, zero downtime, minimal cost.
+
+**Current version:** 3.5.5
+
+## Tech Stack
+
+- **Runtime:** Node.js >= 18 < 24, ES Modules (`"type": "module"`)
+- **Framework:** Next.js 16 (App Router) with TypeScript 5.9
+- **Database:** SQLite via better-sqlite3 (local, zero-config, 16 migrations)
+- **State management:** Zustand (client), SQLite (server persistence)
+- **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons
+- **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth
+- **Schemas:** Zod v4 for all API / MCP input validation
+- **Background jobs:** Custom token health check scheduler, 24h model auto-sync
+- **Streaming:** Server-Sent Events (SSE) for real-time proxy responses
+- **Proxy engine:** Custom pipeline with format translation, circuit breaker, rate limiting, auto-combo engine
+- **i18n:** next-intl with 30 languages
+- **Desktop:** Electron (cross-platform: Windows, macOS, Linux)
+- **Package:** Published on npm (`omniroute`) and Docker Hub (`diegosouzapw/omniroute`)
+
+## Project Structure
+
+```
+/
+├── src/ # Main application source
+│ ├── app/ # Next.js App Router pages and API routes
+│ │ ├── (dashboard)/ # Dashboard UI pages
+│ │ │ └── dashboard/
+│ │ │ ├── agents/ # ACP Agents dashboard (CLI agent detection + custom agents)
+│ │ │ ├── analytics/ # Usage analytics and charts
+│ │ │ ├── api-manager/ # API key management
+│ │ │ ├── audit/ # Audit logs
+│ │ │ ├── auto-combo/ # Auto-combo engine dashboard
+│ │ │ ├── cache/ # Cache dashboard (semantic cache stats)
+│ │ │ ├── cli-tools/ # CLI tool configuration (Claude Code, Codex, Gemini CLI, etc.)
+│ │ │ ├── combos/ # Model combo management (13 strategies + 4 templates)
+│ │ │ ├── costs/ # Cost tracking per provider/model
+│ │ │ ├── endpoint/ # Unified: Endpoint Proxy, MCP, A2A, API Endpoints tabs
+│ │ │ ├── health/ # System health (uptime, circuit breakers, latency)
+│ │ │ ├── limits/ # Rate limits dashboard
+│ │ │ ├── logs/ # Request, Proxy, Audit, Console logs (tabbed)
+│ │ │ ├── media/ # Image/video/music generation + transcription
+│ │ │ ├── memory/ # Memory system dashboard
+│ │ │ ├── onboarding/ # Onboarding wizard
+│ │ │ ├── playground/ # Model playground (Monaco editor, streaming)
+│ │ │ ├── providers/ # Provider management (OAuth + API key + free)
+│ │ │ ├── search-tools/ # Search tools configuration
+│ │ │ ├── settings/ # Settings tabs (General, Appearance, Security, Routing, Resilience, Advanced)
+│ │ │ ├── skills/ # Skills system dashboard
+│ │ │ ├── translator/ # Format translator + debug tools
+│ │ │ └── usage/ # Usage history
+│ │ ├── api/ # REST API endpoints (51 route directories)
+│ │ │ ├── v1/ # OpenAI-compatible API (chat, completions, models, embeddings,
+│ │ │ │ # images, audio, videos, music, moderations, rerank, search,
+│ │ │ │ # responses, messages, registered-keys, quotas, accounts)
+│ │ │ ├── v1beta/ # Gemini-compatible API
+│ │ │ ├── a2a/ # A2A agent management API
+│ │ │ ├── acp/ # ACP agent management API
+│ │ │ ├── oauth/ # OAuth flows per provider
+│ │ │ ├── providers/ # Provider CRUD and batch testing
+│ │ │ ├── models/ # Dashboard model listing and aliases
+│ │ │ ├── combos/ # Combo CRUD (multi-model fallback chains)
+│ │ │ ├── memory/ # Memory system API
+│ │ │ ├── skills/ # Skills system API
+│ │ │ ├── evals/ # Eval runner API
+│ │ │ ├── mcp/ # MCP HTTP transport API
+│ │ │ ├── search/ # Search provider API
+│ │ │ ├── webhooks/ # Webhook management
+│ │ │ ├── tunnels/ # Cloudflare tunnel management
+│ │ │ └── ... # Other endpoints (usage, logs, health, settings, pricing, etc.)
+│ │ ├── landing/ # Landing page
+│ │ ├── login/ # Login page
+│ │ ├── forgot-password/ # Password recovery
+│ │ ├── status/ # Status page
+│ │ └── docs/ # In-app documentation
+│ ├── domain/ # Domain types and policy engine
+│ │ ├── policyEngine.ts # Central policy engine
+│ │ ├── comboResolver.ts # Combo resolution logic
+│ │ ├── costRules.ts # Cost calculation rules
+│ │ ├── degradation.ts # Graceful degradation
+│ │ ├── fallbackPolicy.ts # Fallback behavior
+│ │ ├── lockoutPolicy.ts # Account lockout logic
+│ │ ├── modelAvailability.ts # Model availability checks
+│ │ ├── providerExpiration.ts # Provider credential expiration
+│ │ ├── quotaCache.ts # Quota caching layer
+│ │ ├── configAudit.ts # Configuration auditing
+│ │ └── responses.ts # Domain response types
+│ ├── i18n/ # Internationalization
+│ │ └── messages/ # 30 language JSON files
+│ ├── lib/ # Core libraries
+│ │ ├── a2a/ # Agent-to-Agent v0.3 protocol server
+│ │ │ ├── skills/ # A2A skills (quotaManagement, smartRouting)
+│ │ │ ├── taskManager.ts # Task lifecycle with TTL cleanup
+│ │ │ └── streaming.ts # SSE streaming for A2A
+│ │ ├── acp/ # Agent Communication Protocol registry and manager
+│ │ ├── compliance/ # Compliance policy engine
+│ │ ├── db/ # SQLite database layer (21 modules + migrations)
+│ │ │ ├── core.ts # Database initialization, connection, schema
+│ │ │ ├── providers.ts # Provider connection CRUD
+│ │ │ ├── models.ts # Model catalog management
+│ │ │ ├── combos.ts # Combo configuration
+│ │ │ ├── apiKeys.ts # API key management
+│ │ │ ├── settings.ts # Settings persistence
+│ │ │ ├── backup.ts # Database backup/restore
+│ │ │ ├── proxies.ts # Proxy registry
+│ │ │ ├── prompts.ts # Prompt templates
+│ │ │ ├── webhooks.ts # Webhook subscriptions
+│ │ │ ├── detailedLogs.ts # Detailed request logging
+│ │ │ ├── domainState.ts # Domain state persistence
+│ │ │ ├── registeredKeys.ts # Registered API keys with quotas
+│ │ │ ├── quotaSnapshots.ts # Quota snapshot history
+│ │ │ ├── modelComboMappings.ts # Model-to-combo mappings
+│ │ │ ├── cliToolState.ts # CLI tool state tracking
+│ │ │ ├── encryption.ts # Data encryption
+│ │ │ ├── readCache.ts # Read-through cache layer
+│ │ │ ├── secrets.ts # Secrets management
+│ │ │ ├── stateReset.ts # State reset utilities
+│ │ │ ├── migrationRunner.ts # Schema migration runner
+│ │ │ └── migrations/ # 16 SQL migration files
+│ │ ├── evals/ # Eval runner and scheduler
+│ │ ├── memory/ # Persistent conversational memory
+│ │ │ ├── extraction.ts # Memory extraction from conversations
+│ │ │ ├── injection.ts # Memory injection into context
+│ │ │ ├── retrieval.ts # Memory retrieval/search
+│ │ │ ├── store.ts # Memory persistence layer
+│ │ │ └── summarization.ts # Memory summarization
+│ │ ├── oauth/ # OAuth providers, services, and utilities
+│ │ │ ├── constants/ # Default OAuth credentials (overridable via env)
+│ │ │ ├── providers/ # Provider-specific OAuth configs
+│ │ │ ├── services/ # Provider-specific token exchange logic
+│ │ │ └── utils/ # PKCE, callback server, token helpers
+│ │ ├── plugins/ # Plugin system
+│ │ ├── skills/ # Extensible skill framework
+│ │ │ ├── registry.ts # Skill registration
+│ │ │ ├── executor.ts # Skill execution engine
+│ │ │ ├── sandbox.ts # Skill sandbox environment
+│ │ │ ├── builtin/ # Built-in skills
+│ │ │ ├── interception.ts # Skill request interception
+│ │ │ └── injection.ts # Skill context injection
+│ │ ├── usage/ # Usage tracking system
+│ │ │ ├── callLogs.ts # Call log persistence
+│ │ │ ├── costCalculator.ts # Cost calculation engine
+│ │ │ └── usageHistory.ts # Usage history queries
+│ │ ├── cloudSync.ts # Cloud sync via Cloudflare Workers
+│ │ ├── cloudflaredTunnel.ts # Cloudflare tunnel management
+│ │ ├── pricingSync.ts # LiteLLM pricing data sync
+│ │ ├── semanticCache.ts # Semantic caching layer
+│ │ ├── tokenHealthCheck.ts # Background OAuth token refresh scheduler
+│ │ ├── webhookDispatcher.ts # Webhook event dispatcher
+│ │ └── localDb.ts # Unified re-export layer for all DB modules
+│ ├── middleware/ # Request middleware
+│ │ └── promptInjectionGuard.ts # Prompt injection detection
+│ ├── mitm/ # MITM proxy capability
+│ │ ├── cert/ # Certificate management
+│ │ ├── dns/ # DNS handling
+│ │ ├── targets/ # Target routing
+│ │ └── manager.ts # MITM proxy manager
+│ ├── shared/ # Shared utilities, components, and constants
+│ │ ├── components/ # Reusable UI components (Card, Badge, Button, Modal, Sidebar, ProviderIcon, etc.)
+│ │ ├── constants/ # Provider definitions (60+), model lists, pricing, routing strategies, MCP scopes
+│ │ ├── contracts/ # Shared API contracts
+│ │ ├── hooks/ # React hooks
+│ │ ├── middleware/ # Shared middleware utilities
+│ │ ├── schemas/ # Shared Zod schemas
+│ │ ├── services/ # Shared services
+│ │ ├── types/ # Shared TypeScript types
+│ │ ├── validation/ # Zod schemas (settings, providers, routes)
+│ │ └── utils/ # Helpers (auth, CORS, error codes, machine ID)
+│ ├── sse/ # SSE proxy pipeline
+│ │ ├── services/ # Auth resolution, format translation, response handling
+│ │ └── middleware/ # Rate limiting, circuit breaker, caching, idempotency
+│ ├── store/ # Zustand client-side stores (theme, providers, etc.)
+│ └── types/ # TypeScript type definitions
+├── open-sse/ # Standalone SSE server (npm workspace)
+│ ├── config/ # Model registries (providerRegistry, embedding, image, audio, video,
+│ │ # music, rerank, moderation, search, CLI fingerprints, Ollama models)
+│ ├── executors/ # Provider-specific request executors (14 executors)
+│ │ ├── base.ts # Base executor with shared logic
+│ │ ├── default.ts # Default OpenAI-compatible executor
+│ │ ├── cursor.ts # Cursor IDE (protobuf + checksum)
+│ │ ├── codex.ts # OpenAI Codex CLI
+│ │ ├── antigravity.ts # Antigravity IDE
+│ │ ├── github.ts # GitHub Copilot
+│ │ ├── gemini-cli.ts # Gemini CLI
+│ │ ├── kiro.ts # Kiro AI
+│ │ ├── qoder.ts # Qoder AI
+│ │ ├── vertex.ts # Vertex AI (Service Account JSON)
+│ │ ├── cloudflare-ai.ts # Cloudflare Workers AI
+│ │ ├── opencode.ts # OpenCode Zen/Go
+│ │ ├── pollinations.ts # Pollinations AI
+│ │ └── puter.ts # Puter AI
+│ ├── handlers/ # Request handlers per API type (11 handlers)
+│ │ ├── chatCore.ts # Main chat completions handler
+│ │ ├── responsesHandler.ts # OpenAI Responses API handler
+│ │ ├── embeddings.ts # Embedding generation
+│ │ ├── imageGeneration.ts # Image generation (DALL-E, FLUX, SD, etc.)
+│ │ ├── videoGeneration.ts # Video generation
+│ │ ├── musicGeneration.ts # Music generation
+│ │ ├── audioSpeech.ts # Text-to-speech
+│ │ ├── audioTranscription.ts # Speech-to-text (Whisper, Deepgram, AssemblyAI)
+│ │ ├── moderations.ts # Content moderation
+│ │ ├── rerank.ts # Reranking API
+│ │ └── search.ts # Web search API
+│ ├── mcp-server/ # Built-in MCP server (25 tools, 3 transports: stdio/SSE/streamable-HTTP)
+│ │ ├── server.ts # MCP server core (tool registration, scope enforcement)
+│ │ ├── tools/ # Tool implementations (advancedTools, memoryTools, skillTools)
+│ │ ├── schemas/ # Zod input schemas (tools, audit, a2a)
+│ │ ├── scopeEnforcement.ts # Scope-based access control (10 scopes)
+│ │ ├── audit.ts # Tool call audit logging
+│ │ ├── runtimeHeartbeat.ts # MCP runtime heartbeat
+│ │ └── httpTransport.ts # HTTP transport handler
+│ ├── services/ # 36+ service modules
+│ │ ├── combo.ts # Core routing engine
+│ │ ├── usage.ts # Usage tracking
+│ │ ├── tokenRefresh.ts # OAuth token refresh
+│ │ ├── rateLimitManager.ts # Rate limit management
+│ │ ├── accountFallback.ts # Multi-account fallback
+│ │ ├── sessionManager.ts # Session management
+│ │ ├── wildcardRouter.ts # Wildcard model routing
+│ │ ├── autoCombo/ # Auto-combo engine (6-factor scoring, bandit exploration)
+│ │ ├── intentClassifier.ts # Request intent classification
+│ │ ├── taskAwareRouter.ts # Task-aware routing
+│ │ ├── thinkingBudget.ts # Thinking budget management
+│ │ ├── contextManager.ts # Context window management
+│ │ ├── modelDeprecation.ts # Model deprecation handling
+│ │ ├── modelFamilyFallback.ts # Intra-family model fallback
+│ │ ├── emergencyFallback.ts # Emergency fallback
+│ │ ├── workflowFSM.ts # Workflow state machine
+│ │ ├── backgroundTaskDetector.ts # Background task detection
+│ │ ├── ipFilter.ts # IP-based access control
+│ │ ├── signatureCache.ts # CLI signature caching
+│ │ ├── volumeDetector.ts # Request volume detection
+│ │ ├── contextHandoff.ts # Context relay handoff generation and injection
+│ │ ├── codexQuotaFetcher.ts # Codex quota fetching for context-relay
+│ │ └── ... # Additional services (14 more modules)
+│ ├── transformer/ # Responses API transformer
+│ │ └── responsesTransformer.ts
+│ ├── translator/ # Format translators (OpenAI ↔ Claude ↔ Gemini ↔ Responses ↔ Ollama ↔ DeepSeek)
+│ │ ├── request/ # Request translators per provider
+│ │ ├── response/ # Response translators per provider
+│ │ ├── helpers/ # Translation helpers
+│ │ └── image/ # Image format translation
+│ └── utils/ # 22 utility modules (stream, TLS, proxy, logging, etc.)
+├── electron/ # Electron desktop app (cross-platform)
+│ ├── main.js # Electron main process
+│ ├── preload.js # Preload script (IPC bridge)
+│ └── assets/ # App icons and assets
+├── tests/ # Test suites
+│ ├── unit/ # 122 unit test files
+│ ├── integration/ # Integration tests
+│ ├── e2e/ # Playwright E2E tests
+│ ├── security/ # Security tests
+│ ├── translator/ # Translator-specific tests
+│ └── load/ # Load tests
+├── docs/ # Documentation
+│ ├── i18n/ # 30-language translated docs
+│ ├── ARCHITECTURE.md # Full architecture documentation
+│ ├── API_REFERENCE.md # API reference
+│ ├── USER_GUIDE.md # User guide
+│ ├── CODEBASE_DOCUMENTATION.md # Codebase overview
+│ ├── CLI-TOOLS.md # CLI tools integration guide
+│ ├── A2A-SERVER.md # A2A agent protocol documentation
+│ ├── AUTO-COMBO.md # Auto-combo engine (6-factor scoring)
+│ ├── MCP-SERVER.md # MCP server (25 tools)
+│ ├── TROUBLESHOOTING.md # Troubleshooting guide
+│ ├── VM_DEPLOYMENT_GUIDE.md # VPS deployment guide
+│ ├── openapi.yaml # OpenAPI specification
+│ └── screenshots/ # Dashboard screenshots
+├── bin/ # CLI entry points (omniroute, reset-password)
+├── scripts/ # Build and utility scripts
+└── .env.example # Environment variable template
+```
+
+## Key Features (v3.5.5)
+
+### Core Proxy
+- **60+ AI providers** with automatic format translation
+- **4 provider categories**: Free (4), OAuth (8), API Key (48+), Custom (OpenAI/Anthropic-compatible)
+- **13 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, context-relay
+- **4-tier fallback**: Subscription → API Key → Cheap → Free
+- **Context Relay strategy**: Session handoff summaries on account rotation for continuity
+- **Auto-combo engine**: Self-healing routing optimization with 6-factor scoring, bandit exploration, progressive cooldown
+- **Semantic caching** with cache hit/miss headers
+- **Idempotency** with configurable dedup window
+- **Circuit breaker** per provider with configurable thresholds
+- **Provider Icons**: 130+ provider logos via `@lobehub/icons` (SVG) with PNG fallback
+- **Model Auto-Sync**: 24h scheduler refreshes model lists for 16 providers
+- **Registered Keys API**: Auto-provision API keys via `POST /api/v1/registered-keys` with quota enforcement
+- **Memory System**: Persistent conversational memory with extraction, injection, retrieval, and summarization
+- **Skills System**: Extensible skill framework with registry, executor, sandbox, built-in and custom skills
+- **Prompt Injection Guard**: Middleware-level prompt injection detection
+- **MITM Proxy**: Certificate management, DNS handling, and target routing
+- **Cloudflare Tunnels**: Managed tunnel creation for remote access
+- **122 unit test files** with comprehensive coverage (55% statements/lines/functions, 60% branches)
+
+### सुरक्षा
+- **CodeQL security**: Fixed 10+ CodeQL alerts (polynomial-redos, insecure-randomness, shell-injection, SSRF, incomplete URLs)
+- **Web Crypto session IDs**: `generateSessionId` uses `crypto.getRandomValues()` instead of `Math.random()`
+- **Route validation**: All API routes validated with Zod v4 schemas + `validateBody()`
+- **omniModel tag sanitization**: Internal `` tags never leak to clients in SSE streams
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint to reduce bot detection
+- **CLI Fingerprint Matching** — Per-provider request signature matching
+- **Prompt injection guard** — Request middleware detection
+- **Provider constants validated at module load** via Zod (`src/shared/validation/providerSchema.ts`)
+- **PII sanitizer** — Sensitive data scrubbing in logs
+
+### Dashboard Pages (23 sections)
+- **Providers** — OAuth, API key, and free provider management with ProviderIcon SVG icons
+- **Combos** — Multi-model combo builder with 4 templates (Free Stack, High Availability, Cost Saver, Balanced) + 13 strategies
+- **Auto-Combo** — Auto-combo engine dashboard with scoring metrics
+- **Analytics** — Token consumption, cost, heatmaps, distributions
+- **Health** — Uptime, memory, latency percentiles, circuit breakers
+- **Logs** — Request, Proxy, Audit, Console (tabbed)
+- **Audit** — Audit trail and compliance logging
+- **Costs** — Cost tracking per provider/model
+- **Limits** — Rate limit monitoring
+- **Cache** — Semantic cache statistics and management
+- **CLI Tools** — One-click configuration for 10+ AI CLI tools
+- **CLI Agents** — Grid of 14+ built-in agents with ProviderIcon and install detection + custom agent registration
+- **Playground** — Test any model with Monaco editor, streaming responses
+- **Media** — Image/video/music generation (DALL-E, FLUX, etc.) + audio transcription (up to 2GB files)
+- **Search Tools** — Search provider configuration and testing
+- **Memory** — Memory system management and visualization
+- **Skills** — Skills framework management and execution
+- **Translator** — Format debugging: playground, chat tester, test bench, live monitor
+- **Settings** — General, Appearance (7 color themes), Security (TLS/CLI fingerprint, IP filter), Routing, Resilience, Advanced
+- **Endpoint** — Unified: Endpoint Proxy, MCP Server, A2A Server, API Endpoints (tabbed)
+- **Onboarding** — Setup wizard for new users
+- **Usage** — Usage history and analytics
+- **API Manager** — API key management with scoped permissions
+
+### Protocol Support
+- **OpenAI-compatible** — `/v1/chat/completions`, `/v1/models`, `/v1/embeddings`, `/v1/images/generations`, `/v1/audio/transcriptions`, `/v1/audio/speech`, `/v1/moderations`, `/v1/rerank`, `/v1/videos/generations`, `/v1/music/generations`
+- **Anthropic** — `/v1/messages`, `/v1/messages/count_tokens`
+- **OpenAI Responses** — `/v1/responses`
+- **Gemini** — `/v1beta/models`, `/v1beta/models/{...path}`
+- **Ollama** — `/v1/api/chat`, `/api/tags`
+- **Search** — `/v1/search` (Perplexity, Serper, Brave, Exa, Tavily)
+- **MCP** — 25-tool MCP server with scope-based auth (3 transports: stdio, SSE, streamable HTTP)
+- **A2A** — Agent-to-Agent v0.3 protocol (JSON-RPC 2.0, smart-routing + quota-management skills)
+- **ACP** — Agent Communication Protocol registry and manager
+
+### MCP Server (25 Tools)
+| Category | Tools |
+|-----------|-------|
+| Core (18) | `get_health`, `list_combos`, `get_combo_metrics`, `switch_combo`, `check_quota`, `route_request`, `cost_report`, `list_models_catalog`, `simulate_route`, `set_budget_guard`, `set_routing_strategy`, `set_resilience_profile`, `test_combo`, `get_provider_metrics`, `best_combo_for_task`, `explain_route`, `get_session_snapshot`, `sync_pricing` |
+| Memory (3) | `memory_search`, `memory_add`, `memory_clear` |
+| Skills (4) | `skills_list`, `skills_enable`, `skills_execute`, `skills_executions` |
+
+**MCP Auth Scopes (10):** `read:health`, `read:combos`, `write:combos`, `read:quota`, `read:usage`, `read:models`, `execute:completions`, `execute:search`, `write:budget`, `write:resilience`
+
+### Provider Categories
+
+**Free Providers (4):** Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI
+
+**OAuth Providers (8):** Claude Code, Antigravity, OpenAI Codex, GitHub Copilot, Cursor IDE, Kimi Coding, Kilo Code, Cline
+
+**API Key Providers (48+):** OpenAI, Anthropic, Gemini (Google AI Studio), DeepSeek, Groq, xAI (Grok), Mistral, Perplexity, Together AI, Fireworks AI, Cerebras, Cohere, NVIDIA NIM, Nebius AI, SiliconFlow, Hyperbolic, HuggingFace, OpenRouter, Vertex AI, Cloudflare Workers AI, Scaleway AI, AI/ML API, Pollinations AI, Puter AI, LongCat AI, Alibaba Cloud (DashScope), Alibaba Intl, Alibaba (AliCode), Kimi, Kimi Coding (API Key), Minimax, Minimax (China), Blackbox AI, Synthetic, Kilo Gateway, Z.AI, GLM Coding, Deepgram, AssemblyAI, ElevenLabs, Cartesia, PlayHT, Inworld, NanoBanana, SD WebUI, ComfyUI, Ollama Cloud, Perplexity Search, Serper Search, Brave Search, Exa Search, Tavily Search, OpenCode Zen, OpenCode Go, Bailian Coding Plan
+
+**Custom Providers:** OpenAI-compatible (`openai-compatible-*`) and Anthropic-compatible (`anthropic-compatible-*`) with custom base URLs
+
+### Internationalization
+- 30 languages for UI (all dashboard pages)
+- 30 translated documentation sets in docs/i18n/
+- Language switcher in documentation
+
+## Key Architectural Decisions
+
+1. **OpenAI-compatible API surface:** All incoming requests follow the OpenAI API format. This makes OmniRoute a drop-in replacement for any tool that supports custom OpenAI endpoints.
+
+2. **Provider abstraction via format translators:** Each AI provider has a translator in `open-sse/translator/` that converts between OpenAI format and the provider's native format transparently.
+
+3. **Connection-based provider model:** Providers are stored as "connections" in SQLite. Each connection has an `id`, `provider`, `authType` (oauth/apikey/free), `isActive` flag, and credentials. Multiple connections per provider for multi-account rotation.
+
+4. **Combo system for fallback:** Users create "combos" — ordered lists of `provider/model` pairs. The proxy tries each in order until one succeeds. Supports 13 strategies including auto-combo with self-healing and context-relay for session continuity.
+
+5. **SSE proxy pipeline:** The proxy pipeline is middleware-based: request → auth resolution → rate limiting → circuit breaker → format translation → upstream call → response translation → SSE streaming back to client.
+
+6. **SQLite for persistence:** All state (providers, combos, logs, settings, API keys, memory, skills) stored in a single SQLite database via 21 domain-specific modules. All DB operations go through `src/lib/db/` modules, never raw SQL in routes.
+
+7. **OAuth with PKCE:** OAuth flows use PKCE for security. Token refresh handled by background job (`tokenHealthCheck.ts`).
+
+8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages.
+
+9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 21 `src/lib/db/` modules with 16 SQL migrations.
+
+10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`.
+
+11. **Memory/Skills cross-cutting systems:** Memory and Skills affect the MCP tools, request pipeline, and A2A skills. Memory provides persistent context across sessions; Skills provide extensible tool execution with sandbox isolation.
+
+12. **Domain policy engine:** `src/domain/` contains policy engine modules (policyEngine, comboResolver, costRules, degradation, fallbackPolicy, lockoutPolicy, modelAvailability, providerExpiration, quotaCache, configAudit) that govern routing decisions independently from the pipeline.
+
+13. **Provider constants validated at load:** All provider definitions validated via Zod schemas at module load time (`src/shared/validation/providerSchema.ts`). Invalid providers fail fast.
+
+## Main Flows
+
+### Proxy Request Flow
+1. Client sends OpenAI-format request to `/v1/chat/completions`
+2. API key validation
+3. Model resolution: direct model or combo lookup
+4. For combos: iterate through models with selected strategy
+5. Auth resolution: get credentials for the target provider
+6. Format translation: OpenAI → provider native format
+7. CLI fingerprint matching (if enabled for provider)
+8. Upstream request with circuit breaker and rate limiting
+9. Response translation: provider → OpenAI format
+10. omniModel tag sanitization (strip internal tags)
+11. SSE streaming back to client
+12. Memory extraction (if memory system enabled)
+13. Usage logging and cost calculation
+
+### OAuth Flow
+1. Dashboard initiates `/api/oauth/[provider]/authorize`
+2. User completes OAuth login in browser
+3. Callback hits `/api/oauth/[provider]/exchange`
+4. Tokens stored as a provider connection in SQLite
+5. Background job refreshes tokens before expiry
+
+## Important Notes for LLMs
+
+1. **Two model endpoints exist:** `/api/models` (dashboard, all models) and `/v1/models` (OpenAI-compatible, active only).
+
+2. **Provider IDs vs aliases:** Providers have both an ID (`claude`, `github`) and a short alias (`cc`, `gh`). Models are referenced as `alias/model-name` (e.g., `cc/claude-opus-4-6`).
+
+3. **The `open-sse/` directory is a separate npm workspace** with its own config, handlers, executors, translators, and services.
+
+4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`.
+
+5. **Database layer:** Operations go through `src/lib/db/` modules (21 domain-specific files). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module.
+
+6. **Tests use Node.js built-in test runner:** 122 unit test files. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`).
+
+7. **MCP and A2A pages are embedded as tabs inside `/dashboard/endpoint`**, not standalone routes.
+
+8. **ACP agents** are in `src/lib/acp/registry.ts` with detection cache. Custom agents stored via settings DB.
+
+9. **Auto-combo engine** in `open-sse/services/autoCombo/` — 6-factor scoring, 4 mode packs, bandit exploration, progressive cooldown.
+
+10. **Docker:** Dockerfile has two targets: `runner-base` and `runner-cli`. `docker-compose.yml` for dev (3 profiles), `docker-compose.prod.yml` for production (port 20130).
+
+11. **Electron desktop app** in `electron/` with main.js and preload.js. Build with `npm run electron:build` (supports Windows, macOS, Linux).
+
+12. **Pricing data** syncs from LiteLLM via `src/lib/pricingSync.ts`. Use `sync_pricing` MCP tool or API endpoint.
+
+13. **Memory system** in `src/lib/memory/` provides extraction, injection, retrieval, summarization, and persistent store. Exposed via MCP memory tools and `/api/memory/ API.
+
+14. **Skills system** in `src/lib/skills/` provides registry, executor, sandbox isolation, built-in skills, custom skill support, request interception, and context injection. Exposed via MCP skill tools and `/api/skills/` API.
+
+15. **Zod v4** is used for all validation. Import from `zod` package. Provider schemas validated at module load time.
+
+16. **Context Relay** strategy (`context-relay`) is split across two layers: `combo.ts` decides if a handoff should be generated after a successful turn; `chat.ts` injects the handoff only after account resolution. Handoff data lives in `context_handoffs` SQLite table. Config: `handoffThreshold`, `handoffModel`, `handoffProviders`.
+
+17. **Proxy enforcement** is now comprehensive: token health checks resolve proxy per connection, provider validation wraps in `runWithProxyContext`, and proxy dispatchers use `undici.fetch()` instead of the Node built-in `fetch()` to avoid dispatcher incompatibilities on Node 22.
+
+18. **Node.js 24+ compatibility**: The login page (`/api/settings/require-login`) detects the Node.js version and sends `nodeVersion`/`nodeCompatible` fields. The login UI renders a warning banner when `nodeCompatible` is false.
+
+## Links
+
+- Repository: https://github.com/diegosouzapw/OmniRoute
+- Website: https://omniroute.online
+- npm: https://www.npmjs.com/package/omniroute
+- Docker Hub: https://hub.docker.com/r/diegosouzapw/omniroute
+- Documentation: See `/docs/` directory
diff --git a/docs/i18n/hu/README.md b/docs/i18n/hu/README.md
index 30edc92590..88e3f44d8e 100644
--- a/docs/i18n/hu/README.md
+++ b/docs/i18n/hu/README.md
@@ -4,11 +4,14 @@
---
+
### Never stop coding. Smart routing to **FREE & low-cost AI models** with automatic fallback.
-_Az univerzális API-proxy – egy végpont, több mint 60 szolgáltató, nulla állásidő. Most**MCP-kiszolgálóval (25 eszköz)**,**A2A protokollal**,**memória-/készségrendszerekkel**és**Electron Desktop alkalmazással**._
+_Your universal API proxy — one endpoint, 60+ providers, zero downtime. Now with **MCP Server (25 tools)**, **A2A Protocol**, **Memory/Skills Systems** & **Electron Desktop App**._
-**Csevegés befejezése • Beágyazások • Képgenerálás • Videó • Zene • Hang • Újrarangsorolás •**Webes keresés**• MCP-szerver • A2A protokoll • 100% TypeScript**---
+**Chat Completions • Embeddings • Image Generation • Video • Music • Audio • Reranking • **Web Search** • MCP Server • A2A Protocol • 100% TypeScript**
+
+---
@@ -39,9 +42,13 @@ _Az univerzális API-proxy – egy végpont, több mint 60 szolgáltató, nulla
[](https://omniroute.online)
[](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-[🌐 Webhely](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Funkciók](#-key-features) • [📖 Dokumentumok](#-dokumentáció) • [💰 Árak](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+[🌐 Website](https://omniroute.online) • [🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Docs](#-documentation) • [💰 Pricing](#-pricing-at-a-glance) • [💬 WhatsApp](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
-🌐**Elérhető:**🇺🇸 [angol](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugália)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Szlovénia](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [filippínó](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)---
+
+
+🌐 **Available in:** 🇺🇸 [English](README.md) | 🇧🇷 [Português (Brasil)](docs/i18n/pt-BR/README.md) | 🇪🇸 [Español](docs/i18n/es/README.md) | 🇫🇷 [Français](docs/i18n/fr/README.md) | 🇮🇹 [Italiano](docs/i18n/it/README.md) | 🇷🇺 [Русский](docs/i18n/ru/README.md) | 🇨🇳 [中文 (简体)](docs/i18n/zh-CN/README.md) | 🇩🇪 [Deutsch](docs/i18n/de/README.md) | 🇮🇳 [हिन्दी](docs/i18n/in/README.md) | 🇹🇭 [ไทย](docs/i18n/th/README.md) | 🇺🇦 [Українська](docs/i18n/uk-UA/README.md) | 🇸🇦 [العربية](docs/i18n/ar/README.md) | 🇯🇵 [日本語](docs/i18n/ja/README.md) | 🇻🇳 [Tiếng Việt](docs/i18n/vi/README.md) | 🇧🇬 [Български](docs/i18n/bg/README.md) | 🇩🇰 [Dansk](docs/i18n/da/README.md) | 🇫🇮 [Suomi](docs/i18n/fi/README.md) | 🇮🇱 [עברית](docs/i18n/he/README.md) | 🇭🇺 [Magyar](docs/i18n/hu/README.md) | 🇮🇩 [Bahasa Indonesia](docs/i18n/id/README.md) | 🇰🇷 [한국어](docs/i18n/ko/README.md) | 🇲🇾 [Bahasa Melayu](docs/i18n/ms/README.md) | 🇳🇱 [Nederlands](docs/i18n/nl/README.md) | 🇳🇴 [Norsk](docs/i18n/no/README.md) | 🇵🇹 [Português (Portugal)](docs/i18n/pt/README.md) | 🇷🇴 [Română](docs/i18n/ro/README.md) | 🇵🇱 [Polski](docs/i18n/pl/README.md) | 🇸🇰 [Slovenčina](docs/i18n/sk/README.md) | 🇸🇪 [Svenska](docs/i18n/sv/README.md) | 🇵🇭 [Filipino](docs/i18n/phi/README.md) | 🇨🇿 [Čeština](docs/i18n/cs/README.md)
+
+---
## 🖼️ Main Dashboard
@@ -53,555 +60,629 @@ _Az univerzális API-proxy – egy végpont, több mint 60 szolgáltató, nulla
## 📸 Dashboard Preview
-
+
+Click to see dashboard screenshots
-Kattintson ide az irányítópult képernyőképeinek megtekintéséhez
+| Page | Screenshot |
+| -------------- | ------------------------------------------------- |
+| **Providers** |  |
+| **Combos** |  |
+| **Analytics** |  |
+| **Health** |  |
+| **Translator** |  |
+| **Settings** |  |
+| **CLI Tools** |  |
+| **Usage Logs** |  |
+| **Endpoints** |  |
-| Oldal | Képernyőkép |
-| --------------------- | -------------------------------------------------- | ---------- |
-| **Szolgáltatók** |  |
-| **Kombók** |  |
-| **Analytics** |  |
-| **Egészség** |  |
-| **Fordító** |  |
-| **Beállítások** |  |
-| **CLI eszközök** |  |
-| **Használati naplók** |  |
-| **Végpontok** |  | |
+
---
### 🤖 Free AI Provider for your favorite coding agents
-_Csatlakoztasson bármilyen mesterséges intelligencia-alapú IDE-t vagy CLI-eszközt az OmniRoute-on keresztül – ingyenes API-átjáró a korlátlan kódoláshoz._
-
-
-
-
-
-
-OpenClaw
-
-⭐ 205 000
- |
-
-
-
-NanoBot
-
-⭐ 20,9K
- |
-
-
-
-PicoClaw
-
-⭐ 14,6 KB
- |
-
-
-
-ZeroClaw
-
-⭐ 9,9 KB
- |
-
-
-
-Vaskarom
-
-⭐ 2,1K
- |
-
-
-
-
-
-OpenCode
-
-⭐ 106K
- |
-
-
-
-Codex CLI
-
-⭐ 60,8K
- |
-
-
-
-Claude Code
-
-⭐ 67,3 K
- |
-
-
-
-Gemini CLI
-
-⭐ 94,7K
- |
-
-
-
-Kilókód
-
-⭐ 15,5 KB
- |
-
+_Connect any AI-powered IDE or CLI tool through OmniRoute — free API gateway for unlimited coding._
+
-📡 Minden ügynök a http://localhost:20128/v1 vagy a http://cloud.omniroute.online/v1 segítségével csatlakozik – egy konfiguráció, korlátlan modellek és kvóta---
+📡 All agents connect via http://localhost:20128/v1 or http://cloud.omniroute.online/v1 — one config, unlimited models and quota
+
+---
## 🤔 Why OmniRoute?
-**Ne pazarolja a pénzt, és ne lépje túl a limiteket:**
+**Stop wasting money and hitting limits:**
--
Az előfizetési kvóta minden hónapban fel nem használt
--
A díjkorlátok megakadályozzák a középső kódolást
--
Drága API-k (20-50 USD/hó szolgáltatónként)
--
Manuális váltás a szolgáltatók között
+-
Subscription quota expires unused every month
+-
Rate limits stop you mid-coding
+-
Expensive APIs ($20-50/month per provider)
+-
Manual switching between providers
-**Az OmniRoute ezt megoldja:**
+**OmniRoute solves this:**
-- ✅**Az előfizetések maximalizálása**- Kövesse nyomon a kvótát, használjon fel minden bitet a visszaállítás előtt
-- ✅**Automatikus tartalék**- Előfizetés → API-kulcs → Olcsó → Ingyenes, nulla állásidő
-- ✅**Több fiók**- Kör-robin a fiókok között szolgáltatónként
-- ✅**Univerzális**- Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, bármilyen CLI eszközzel működik---
+- ✅ **Maximize subscriptions** - Track quota, use every bit before reset
+- ✅ **Auto fallback** - Subscription → API Key → Cheap → Free, zero downtime
+- ✅ **Multi-account** - Round-robin between accounts per provider
+- ✅ **Universal** - Works with Claude Code, Codex, Gemini CLI, Cursor, Cline, OpenClaw, any CLI tool
+
+---
## 📧 Support
-> 💬**Csatlakozzon közösségünkhöz!**[WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) – Kérjen segítséget, ossza meg tippjeit, és maradjon naprakész.
+> 💬 **Join our community!** [WhatsApp Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) — Get help, share tips, and stay updated.
--**Webhely**: [omniroute.online](https://omniroute.online) -**GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute) -**Problémák**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues) -**WhatsApp**: [Közösségi csoport](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t) -**Hozzájárulás**: Tekintse meg a [CONTRIBUTING.md](CONTRIBUTING.md) oldalt, nyisson PR-t, vagy válasszon egy "jó első számot". -**Original Project**: [9router by decolua](https://github.com/decolua/9router)### 🐛 Reporting a Bug?
+- **Website**: [omniroute.online](https://omniroute.online)
+- **GitHub**: [github.com/diegosouzapw/OmniRoute](https://github.com/diegosouzapw/OmniRoute)
+- **Issues**: [github.com/diegosouzapw/OmniRoute/issues](https://github.com/diegosouzapw/OmniRoute/issues)
+- **WhatsApp**: [Community Group](https://chat.whatsapp.com/JI7cDQ1GyaiDHhVBpLxf8b?mode=gi_t)
+- **Contributing**: See [CONTRIBUTING.md](CONTRIBUTING.md), open a PR, or pick a `good first issue`
+- **Original Project**: [9router by decolua](https://github.com/decolua/9router)
-Egy probléma megnyitásakor futtassa a system-info parancsot, és csatolja a generált fájlt:```bash
+### 🐛 Reporting a Bug?
+
+When opening an issue, please run the system-info command and attach the generated file:
+
+```bash
npm run system-info
-
```
-Ez létrehoz egy "system-info.txt" fájlt a Node.js verziójával, az OmniRoute verziójával, az operációs rendszer részleteivel, a telepített CLI-eszközökkel (qoder, gemini, claude, codex, antigravitáció, droid stb.), Docker/PM2 állapottal és rendszercsomagokkal – mindennel, amire szükségünk van a probléma gyors reprodukálásához. Csatolja a fájlt közvetlenül a GitHub-problémához.---
+This generates a `system-info.txt` with your Node.js version, OmniRoute version, OS details, installed CLI tools (qoder, gemini, claude, codex, antigravity, droid, etc.), Docker/PM2 status, and system packages — everything we need to reproduce your issue quickly. Attach the file directly to your GitHub issue.
+
+---
## 🔄 How It Works
```
-
┌─────────────┐
-│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
-│ Tool │
+│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
+│ Tool │
└──────┬──────┘
-│ http://localhost:20128/v1
-↓
+ │ http://localhost:20128/v1
+ ↓
┌─────────────────────────────────────────┐
-│ OmniRoute (Smart Router) │
-│ • Format translation (OpenAI ↔ Claude) │
-│ • Quota tracking + Embeddings + Images │
-│ • Auto token refresh │
+│ OmniRoute (Smart Router) │
+│ • Format translation (OpenAI ↔ Claude) │
+│ • Quota tracking + Embeddings + Images │
+│ • Auto token refresh │
└──────┬──────────────────────────────────┘
-│
-├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
-│ ↓ quota exhausted
-├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
-│ ↓ budget limit
-├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
-│ ↓ budget limit
-└─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
+ │
+ ├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
+ │ ↓ quota exhausted
+ ├─→ [Tier 2: API KEY] DeepSeek, Groq, xAI, Mistral, NVIDIA NIM, etc.
+ │ ↓ budget limit
+ ├─→ [Tier 3: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
+ │ ↓ budget limit
+ └─→ [Tier 4: FREE] Qoder, Qwen, Kiro (unlimited)
Result: Never stop coding, minimal cost
-
-````
+```
---
## 🎯 What OmniRoute Solves — 30 Real Pain Points & Use Cases
->**Minden mesterséges intelligencia-eszközöket használó fejlesztő naponta szembesül ezekkel a problémákkal.**Az OmniRoute úgy készült, hogy ezeket mind megoldja – a költségtúllépésektől a regionális blokkokig, a megszakadt OAuth-folyamatoktól a protokollműveletekig és a vállalati megfigyelhetőségig.
+> **Every developer using AI tools faces these problems daily.** OmniRoute was built to solve them all — from cost overruns to regional blocks, from broken OAuth flows to protocol operations and enterprise observability.
-
-💸 1. "Drága előfizetésért fizetek, de még mindig megzavarnak a korlátok"
+
+💸 1. "I pay for an expensive subscription but still get interrupted by limits"
-A fejlesztők havi 20–200 dollárt fizetnek a Claude Pro, Codex Pro vagy GitHub Copilotért. A kvótának még fizetés esetén is van felső határa – 5 óra használat, heti limitek vagy percdíjkorlátok. Mid-coding session, the provider stops responding and the developer loses flow and productivity.
+Developers pay $20–200/month for Claude Pro, Codex Pro, or GitHub Copilot. Even paying, quota has a ceiling — 5h of usage, weekly limits, or per-minute rate limits. Mid-coding session, the provider stops responding and the developer loses flow and productivity.
-**Hogyan oldja meg az OmniRoute:**
+**How OmniRoute solves it:**
--**Smart 4-Tier Fallback**– Ha az előfizetési kvóta kimerül, automatikusan átirányítja az API-kulcs → Olcsó → Ingyenes, manuális beavatkozás nélkül
--**Szolgáltatói korlátozások követése**– A gyorsítótárazott kvóta pillanatképei szerveroldali ütemezés szerint frissülnek (alapértelmezett `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`), a felhasználói felületen elérhető kézi frissítéssel
--**Több fiók támogatása**- Több fiók szolgáltatónként automatikus körváltással - ha az egyik elfogy, átvált a következőre
--**Egyéni kombinációk**— Testreszabható tartalék láncok 9 kiegyensúlyozási stratégiával (prioritásos, súlyozott, kitöltési sorrendben, körbefutó, P2C, véletlenszerű, legkevésbé használt, költségoptimalizált, szigorúan véletlenszerű)
--**Codex üzleti kvóták**— Üzleti/csapat munkaterület-kvóta figyelése közvetlenül az irányítópulton
+- **Smart 4-Tier Fallback** — If subscription quota runs out, automatically redirects to API Key → Cheap → Free with zero manual intervention
+- **Provider Limits Tracking** — Cached quota snapshots refresh on a server-side schedule (default `PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES=70`) with manual refresh available in the UI
+- **Multi-Account Support** — Multiple accounts per provider with auto round-robin — when one runs out, switches to the next
+- **Custom Combos** — Customizable fallback chains with 13 balancing strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, strict-random, auto, lkgp, context-optimized, **context-relay**)
+- **Codex Business Quotas** — Business/Team workspace quota monitoring directly in the dashboard
-
-🔌 2. "Több szolgáltatót kell használnom, de mindegyik más API-val rendelkezik"
+
-Az OpenAI egy formátumot használ, a Claude (Anthropic) egy másikat, a Gemini pedig egy másikat. Ha egy fejlesztő különböző szolgáltatók modelljeit szeretné tesztelni, vagy tartalékot szeretne közöttük, akkor újra kell konfigurálnia az SDK-kat, módosítania kell a végpontokat, és kezelnie kell az inkompatibilis formátumokat. Az egyéni szolgáltatók (FriendLI, NIM) nem szabványos modellvégpontokkal rendelkeznek.
+
+🔌 2. "I need to use multiple providers but each has a different API"
-**Hogyan oldja meg az OmniRoute:**
+OpenAI uses one format, Claude (Anthropic) uses another, Gemini yet another. If a dev wants to test models from different providers or fallback between them, they need to reconfigure SDKs, change endpoints, deal with incompatible formats. Custom providers (FriendLI, NIM) have non-standard model endpoints.
--**Egységes végpont**- Egyetlen "http://localhost:20128/v1" proxyként szolgál mind a 60+ szolgáltató számára
--**Formátumfordítás**- Automatikus és átlátható: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
--**Response Sanitization**— Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+
--**Szerepek normalizálása**— Átalakítja a "fejlesztő" → "rendszert" a nem OpenAI szolgáltatók számára; `rendszer` → `felhasználó` a GLM/ERNIE-hez
--**Think Tag Extraction**– Kivonja a „” blokkokat olyan modellekből, mint a DeepSeek R1 szabványos „reasoning_content” tartalommá
--**Strukturált kimenet a Gemini számára**— `json_schema` → `responseMimeType`/`responseSchema` automatikus átalakítás
-- A**„stream” alapértelmezése „false”**– Az OpenAI specifikációhoz igazodik, elkerülve a váratlan SSE-t a Python/Rust/Go SDK-kban
+**How OmniRoute solves it:**
-
-🌐 3. „Az AI-szolgáltatóm blokkolja a régiómat/országomat”
+- **Unified Endpoint** — A single `http://localhost:20128/v1` serves as proxy for all 60+ providers
+- **Format Translation** — Automatic and transparent: OpenAI ↔ Claude ↔ Gemini ↔ Responses API
+- **Response Sanitization** — Strips non-standard fields (`x_groq`, `usage_breakdown`, `service_tier`) that break OpenAI SDK v1.83+
+- **Role Normalization** — Converts `developer` → `system` for non-OpenAI providers; `system` → `user` for GLM/ERNIE
+- **Think Tag Extraction** — Extracts `` blocks from models like DeepSeek R1 into standardized `reasoning_content`
+- **Structured Output for Gemini** — `json_schema` → `responseMimeType`/`responseSchema` automatic conversion
+- **`stream` defaults to `false`** — Aligns with OpenAI spec, avoiding unexpected SSE in Python/Rust/Go SDKs
-Az olyan szolgáltatók, mint az OpenAI/Codex, blokkolják a hozzáférést bizonyos földrajzi régiókból. A felhasználók OAuth- és API-kapcsolatok során olyan hibákat kapnak, mint a „nem támogatott_ország_régió_területe”. Ez különösen frusztráló a fejlődő országok fejlesztői számára.
+
-**Hogyan oldja meg az OmniRoute:**
+
+🌐 3. "My AI provider blocks my region/country"
--**3-szintű proxykonfiguráció**– 3 szinten konfigurálható proxy: globális (teljes forgalom), szolgáltatónként (csak egy szolgáltató) és kapcsolatonként/kulcsonként
--**Színes proxy jelvények**- Vizuális jelzők: 🟢 globális proxy, 🟡 szolgáltató proxy, 🔵 kapcsolat proxy, mindig az IP-t mutatja
--**OAuth-tokencsere proxyn keresztül**– Az OAuth-folyamat a proxyn is keresztülmegy, megoldva a `nem támogatott_ország_régió_területét'
--**Kapcsolódási tesztek proxyn keresztül**- A csatlakozási tesztek a konfigurált proxyt használják (nincs többé közvetlen kiiktatás)
--**SOCKS5 támogatás**— Teljes SOCKS5 proxy támogatás a kimenő útválasztáshoz
--**TLS-ujjlenyomat-hamisítás**– Böngészőszerű TLS-ujjlenyomat a "wreq-js"-n keresztül a botészlelés megkerüléséhez
--**🔏 CLI Ujjlenyomat Matching**– A fejlécek és törzsmezők átrendezése, hogy megfeleljenek a natív CLI bináris aláírásoknak, drasztikusan csökkentve a fiók megjelölésének kockázatát. A proxy IP-címe megmarad – egyszerre kapja meg a lopakodó**és**IP-maszkolást
+Providers like OpenAI/Codex block access from certain geographic regions. Users get errors like `unsupported_country_region_territory` during OAuth and API connections. This is especially frustrating for developers from developing countries.
-
-🆓 4. "MI-t akarok használni kódoláshoz, de nincs pénzem"
+**How OmniRoute solves it:**
-Nem mindenki fizethet havi 20–200 dollárt az AI-előfizetésekért. A feltörekvő országok diákjainak, fejlesztőinek, amatőröknek és szabadúszóknak nulla költséggel kell hozzáférniük a minőségi modellekhez.
+- **3-Level Proxy Config** — Configurable proxy at 3 levels: global (all traffic), per-provider (one provider only), and per-connection/key
+- **Color-Coded Proxy Badges** — Visual indicators: 🟢 global proxy, 🟡 provider proxy, 🔵 connection proxy, always showing the IP
+- **OAuth Token Exchange Through Proxy** — OAuth flow also goes through the proxy, solving `unsupported_country_region_territory`
+- **Connection Tests via Proxy** — Connection tests use the configured proxy (no more direct bypass)
+- **SOCKS5 Support** — Full SOCKS5 proxy support for outbound routing
+- **TLS Fingerprint Spoofing** — Browser-like TLS fingerprint via `wreq-js` to bypass bot detection
+- **🔏 CLI Fingerprint Matching** — Reorders headers and body fields to match native CLI binary signatures, drastically reducing account flagging risk. The proxy IP is preserved — you get both stealth **and** IP masking simultaneously
-**Hogyan oldja meg az OmniRoute:**
+
--**Beépített ingyenes szolgáltatók**- Natív támogatás 100%-ban ingyenes szolgáltatókhoz: Qoder (5 korlátlan modell OAuth-on keresztül: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlim-limitus3 modell) qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID ingyen), Gemini CLI (180 000 token/hónap ingyenes)
--**Ollama Cloud**– Felhőben tárolt Ollama-modellek az `api.ollama.com-on, ingyenes "Light usage" szinttel; használja az `ollamacloud/` előtagot
--**Csak ingyenes kombók**— Lánc `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = 0 USD/hó nulla állásidővel
--**NVIDIA NIM ingyenes hozzáférés**– ~40 RPM fejlesztői örökké ingyenes hozzáférés több mint 70 modellhez a build.nvidia.com oldalon (áttérés a kreditekről a tiszta sebességkorlátokra)
--**Költségoptimalizált stratégia**— Útválasztási stratégia, amely automatikusan a legolcsóbb elérhető szolgáltatót választja
+
+🆓 4. "I want to use AI for coding but I have no money"
-
-🔒 5. "Meg kell védenem a mesterséges intelligencia átjárómat a jogosulatlan hozzáféréstől"
+Not everyone can pay $20–200/month for AI subscriptions. Students, devs from emerging countries, hobbyists, and freelancers need access to quality models at zero cost.
-Ha AI átjárót teszünk ki a hálózatnak (LAN, VPS, Docker), a cím birtokában bárki felhasználhatja a fejlesztő tokenjeit/kvótáját. Védelem nélkül az API-k sebezhetőek a visszaélésekkel, azonnali befecskendezéssel és visszaélésekkel szemben.
+**How OmniRoute solves it:**
-**Hogyan oldja meg az OmniRoute:**
+- **Free Tier Providers Built-in** — Native support for 100% free providers: Qoder (5 unlimited models via OAuth: kimi-k2-thinking, qwen3-coder-plus, deepseek-r1, minimax-m2, kimi-k2), Qwen (4 unlimited models: qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next, vision-model), Kiro (Claude + AWS Builder ID for free), Gemini CLI (180K tokens/month free)
+- **Ollama Cloud** — Cloud-hosted Ollama models at `api.ollama.com` with free "Light usage" tier; use `ollamacloud/` prefix
+- **Free-Only Combos** — Chain `gc/gemini-3-flash → if/kimi-k2-thinking → qw/qwen3-coder-plus` = $0/month with zero downtime
+- **NVIDIA NIM Free Access** — ~40 RPM dev-forever free access to 70+ models at build.nvidia.com (transitioning from credits to pure rate limits)
+- **Cost Optimized Strategy** — Routing strategy that automatically chooses the cheapest available provider
--**API-kulcskezelés**– Generálás, rotáció és hatókör szolgáltatónként egy dedikált "/dashboard/api-manager" oldallal
--**Modellszintű engedélyek**– Az API-kulcsok korlátozása adott modellekre ("openai/*", helyettesítő karakteres minták), az Összes engedélyezése/Korlátozása kapcsolóval
--**API Endpoint Protection**– Kulcs szükséges a `/v1/models' számára, és bizonyos szolgáltatók letiltása a listából
--**Auth Guard + CSRF védelem**- Minden műszerfali útvonal "withAuth" köztes szoftverrel + CSRF tokenekkel védett
--**Rate Limiter**— IP-nkénti sebességkorlátozás konfigurálható ablakokkal
--**IP-szűrés**— Engedélyezési lista/blokkolólista a hozzáférés-vezérléshez
--**Prompt Injection Guard**– fertőtlenítés a rosszindulatú felszólítási minták ellen
--**AES-256-GCM titkosítás**- A hitelesítő adatok nyugalmi állapotban titkosítva
+
-
-🛑 6. "A szolgáltatóm leállt, és elvesztettem a kódolási folyamatomat"
+
+🔒 5. "I need to protect my AI gateway from unauthorized access"
-Az AI-szolgáltatók instabillá válhatnak, 5xx-es hibákat adnak vissza, vagy elérhetik az ideiglenes sebességkorlátokat. Ha egy fejlesztő egyetlen szolgáltatótól függ, akkor megszakad. Megszakítók nélkül az ismételt újrapróbálkozások összeomolhatják az alkalmazást.
+When exposing an AI gateway to the network (LAN, VPS, Docker), anyone with the address can consume the developer's tokens/quota. Without protection, APIs are vulnerable to misuse, prompt injection, and abuse.
-**Hogyan oldja meg az OmniRoute:**
+**How OmniRoute solves it:**
--**Megszakító típusonként**- Automatikus nyitás/zárás konfigurálható küszöbértékekkel és lehűtéssel (zárt/nyitott/félig nyitott), modellenkénti hatókör a lépcsőzetes blokkok elkerülése érdekében
--**Exponenciális visszalépés**— Progresszív újrapróbálkozási késések
--**Mennydörgés elleni csorda**- Mutex + szemafor védelem az egyidejű újrapróbálkozási viharok ellen
--**Kombinált tartalék láncok**– Ha az elsődleges szolgáltató meghibásodik, automatikusan, beavatkozás nélkül átesik a láncon
--**Combo Circuit Breaker**– Automatikusan letiltja a hibás szolgáltatókat a kombinált láncon belül
--**Egészségügyi irányítópult**— Üzemidő-figyelés, áramkör-megszakító állapotok, zárolások, gyorsítótár-statisztika, p50/p95/p99 késleltetés
+- **API Key Management** — Generation, rotation, and scoping per provider with a dedicated `/dashboard/api-manager` page
+- **Model-Level Permissions** — Restrict API keys to specific models (`openai/*`, wildcard patterns), with Allow All/Restrict toggle
+- **API Endpoint Protection** — Require a key for `/v1/models` and block specific providers from the listing
+- **Auth Guard + CSRF Protection** — All dashboard routes protected with `withAuth` middleware + CSRF tokens
+- **Rate Limiter** — Per-IP rate limiting with configurable windows
+- **IP Filtering** — Allowlist/blocklist for access control
+- **Prompt Injection Guard** — Sanitization against malicious prompt patterns
+- **AES-256-GCM Encryption** — Credentials encrypted at rest
-
-🔧 7. "Az egyes AI-eszközök konfigurálása fárasztó és ismétlődő."
+
-A fejlesztők Cursort, Claude Code-ot, Codex CLI-t, OpenClaw-ot, Gemini CLI-t, Kilo Code-ot használnak... Minden eszköznek más konfigurációra van szüksége (API végpont, kulcs, modell). Az újrakonfigurálás szolgáltató- vagy modellváltáskor időpocsékolás.
+
+🛑 6. "My provider went down and I lost my coding flow"
-**Hogyan oldja meg az OmniRoute:**
+AI providers can become unstable, return 5xx errors, or hit temporary rate limits. If a dev depends on a single provider, they're interrupted. Without circuit breakers, repeated retries can crash the application.
--**CLI Tools Dashboard**- Dedikált oldal egykattintásos beállítással a Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline számára
--**GitHub Copilot Config Generator**– A "chatLanguageModels.json" fájlt generálja a VS-kódhoz tömeges modellválasztással
--**Bevezető varázsló**– Irányított 4 lépéses beállítás első felhasználók számára
--**Egy végpont, minden modell**- Konfigurálja egyszer a `http://localhost:20128/v1' címet, elérje a 60+ szolgáltatót
+**How OmniRoute solves it:**
-
-🔑 8. „A több szolgáltatótól származó OAuth-tokenek kezelése pokol”
+- **Circuit Breaker per-model** — Auto-open/close with configurable thresholds and cooldown (Closed/Open/Half-Open), scoped per-model to avoid cascading blocks
+- **Exponential Backoff** — Progressive retry delays
+- **Anti-Thundering Herd** — Mutex + semaphore protection against concurrent retry storms
+- **Combo Fallback Chains** — If the primary provider fails, automatically falls through the chain with no intervention
+- **Combo Circuit Breaker** — Auto-disables failing providers within a combo chain
+- **Health Dashboard** — Uptime monitoring, circuit breaker states, lockouts, cache stats, p50/p95/p99 latency
-Claude Code, Codex, Gemini CLI, Copilot – mindegyik az OAuth 2.0-t használja lejáró tokenekkel. A fejlesztőknek folyamatosan újra kell hitelesíteniük, kezelniük kell a „kliens_titka hiányzik”, az „átirányítási_uri_mismatch” és a távoli szerverek hibáival. Az OAuth a LAN/VPS-en különösen problémás.
+
-**Hogyan oldja meg az OmniRoute:**
+
+🔧 7. "Configuring each AI tool is tedious and repetitive"
--**Automatikus tokenfrissítés**- Az OAuth-tokenek a háttérben frissülnek a lejárat előtt
--**OAuth 2.0 (PKCE) beépített**- Automatikus áramlás Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder számára
--**Multi-Account OAuth**- Több fiók szolgáltatónként a JWT/ID token kivonattal
--**OAuth LAN/Távoli javítás**– Privát IP-észlelés az „átirányítási_uri”-hoz + kézi URL mód távoli szerverekhez
--**OAuth Nginx mögött**– A "window.location.origin" fájlt használja a fordított proxy kompatibilitás érdekében
--**Távoli OAuth útmutató**– Lépésről lépésre útmutató a Google Cloud hitelesítő adataihoz VPS/Docker rendszeren
+Developers use Cursor, Claude Code, Codex CLI, OpenClaw, Gemini CLI, Kilo Code... Each tool needs a different config (API endpoint, key, model). Reconfiguring when switching providers or models is a waste of time.
-
-📊 9. "Nem tudom, mennyit költök és hova"
+**How OmniRoute solves it:**
-A fejlesztők több fizetős szolgáltatót használnak, de nincs egységes nézetük a kiadásokról. Minden szolgáltató saját számlázási irányítópulttal rendelkezik, de nincs összevont nézet. A váratlan költségek felhalmozódhatnak.
+- **CLI Tools Dashboard** — Dedicated page with one-click setup for Claude Code, Codex CLI, OpenClaw, Kilo Code, Antigravity, Cline
+- **GitHub Copilot Config Generator** — Generates `chatLanguageModels.json` for VS Code with bulk model selection
+- **Onboarding Wizard** — Guided 4-step setup for first-time users
+- **One endpoint, all models** — Configure `http://localhost:20128/v1` once, access 60+ providers
-**Hogyan oldja meg az OmniRoute:**
+
--**Költségelemzési irányítópult**– Tokenenkénti költségkövetés és költségkeret-kezelés szolgáltatónként
--**Költségkeret-korlátok rétegenként**- Költési felső határ szintenként, amely automatikus visszalépést vált ki
--**Modellenkénti árképzés**- Konfigurálható árak modellenként
--**Használati statisztika API-kulcsonként**— A kérések száma és az utoljára használt időbélyeg kulcsonként
--**Analytics Dashboard**— Statisztikai kártyák, modellhasználati diagram, szolgáltatói táblázat sikerarányokkal és késleltetéssel
+
+🔑 8. "Managing OAuth tokens from multiple providers is hell"
-
-🐛 10. "Nem tudom diagnosztizálni a hibákat és problémákat az AI-hívásoknál"
+Claude Code, Codex, Gemini CLI, Copilot — all use OAuth 2.0 with expiring tokens. Developers need to re-authenticate constantly, deal with `client_secret is missing`, `redirect_uri_mismatch`, and failures on remote servers. OAuth on LAN/VPS is particularly problematic.
-Ha egy hívás meghiúsul, a fejlesztő nem tudja, hogy sebességkorlátozás, lejárt token, rossz formátum vagy szolgáltatói hiba volt-e. Töredezett naplók a különböző terminálokon. Megfigyelhetőség nélkül a hibakeresés próba és hiba.
+**How OmniRoute solves it:**
-**Hogyan oldja meg az OmniRoute:**
+- **Auto Token Refresh** — OAuth tokens refresh in background before expiration
+- **OAuth 2.0 (PKCE) Built-in** — Automatic flow for Claude Code, Codex, Gemini CLI, Copilot, Kiro, Qwen, Qoder
+- **Multi-Account OAuth** — Multiple accounts per provider via JWT/ID token extraction
+- **OAuth LAN/Remote Fix** — Private IP detection for `redirect_uri` + manual URL mode for remote servers
+- **OAuth Behind Nginx** — Uses `window.location.origin` for reverse proxy compatibility
+- **Remote OAuth Guide** — Step-by-step guide for Google Cloud credentials on VPS/Docker
--**Egységes naplók irányítópultja**- 4 lap: Kérelemnaplók, Proxynaplók, Auditnaplók, Konzol
--**Konzolnapló-nézegető**— Valós idejű terminál stílusú megjelenítő színkódolt szintekkel, automatikus görgetés, keresés, szűrés
--**SQLite proxynaplók**– Állandó naplók, amelyek túlélik a szerver újraindítását
--**Translator Playground**– 4 hibakeresési mód: Playground (formátum fordítás), Chat Tester (oda-vissza út), Tesztpad (kötegelt), Élő monitor (valós idejű)
--**Request Telemetria**– p50/p95/p99 késleltetés + X-Request-Id nyomkövetés
--**File-Based Logging with Rotation**— App logs rotate by size, retention days, and archive count; a hívásnapló műtermékei a megőrzési napok és a fájlok száma szerint váltakoznak
--**Rendszerinformációs jelentés**— Az `npm run system-info` létrehozza a `system-info.txt' fájlt a teljes környezettel (Node verzió, OmniRoute verzió, OS, CLI eszközök, Docker/PM2 állapot). Csatolja, amikor az azonnali osztályozással kapcsolatos problémákat jelent be.
+
-
-🏗️ 11. „Az átjáró telepítése és karbantartása bonyolult”
+
+📊 9. "I don't know how much I'm spending or where"
-Az AI-proxy telepítése, konfigurálása és karbantartása különböző környezetekben (helyi, VPS, Docker, felhő) munkaigényes. Az olyan problémák, mint a keménykódolt elérési utak, az „EACCES” a könyvtárakon, a portkonfliktusok és a többplatformos buildek súrlódást okoznak.
+Developers use multiple paid providers but have no unified view of spending. Each provider has its own billing dashboard, but there's no consolidated view. Unexpected costs can pile up.
-**Hogyan oldja meg az OmniRoute:**
+**How OmniRoute solves it:**
--**npm globális telepítés**— `npm install -g omniroute && omniroute` — kész
--**Docker Multi-Platform**– AMD64 + ARM64 natív (Apple Silicon, AWS Graviton, Raspberry Pi)
--**Docker Compose Profiles**– "alap" (nincs CLI-eszközök) és "cli" (Claude Code, Codex, OpenClaw)
--**Electron Desktop App**– Natív alkalmazás Windows/macOS/Linux rendszerhez rendszertálcával, automatikus indítással, offline móddal
--**Split-Port Mode**– API és irányítópult külön portokon haladó forgatókönyvekhez (fordított proxy, konténerhálózat)
--**Cloud Sync**– Szinkronizálás konfigurálása az eszközök között a Cloudflare Workers segítségével
--**DB biztonsági mentések**– Az összes beállítás automatikus biztonsági mentése, visszaállítása, exportálása és importálása, a `DISABLE_SQLITE_AUTO_BACKUP` funkcióval a külsőleg kezelt biztonsági mentésekhez
+- **Cost Analytics Dashboard** — Per-token cost tracking and budget management per provider
+- **Budget Limits per Tier** — Spending ceiling per tier that triggers automatic fallback
+- **Per-Model Pricing Configuration** — Configurable prices per model
+- **Usage Statistics Per API Key** — Request count and last-used timestamp per key
+- **Analytics Dashboard** — Stat cards, model usage chart, provider table with success rates and latency
-
-🌍 12. "A felület csak angol nyelvű, és a csapatom nem beszél angolul"
+
-A nem angol nyelvű országok csapatai, különösen Latin-Amerikában, Ázsiában és Európában, csak angol nyelvű felületekkel küszködnek. A nyelvi akadályok csökkentik az átvételt és növelik a konfigurációs hibákat.
+
+🐛 10. "I can't diagnose errors and problems in AI calls"
-**Hogyan oldja meg az OmniRoute:**
+When a call fails, the dev doesn't know if it was a rate limit, expired token, wrong format, or provider error. Fragmented logs across different terminals. Without observability, debugging is trial-and-error.
--**Irányítópult i18n – 30 nyelv**– Mind az 500+ billentyű lefordítva, beleértve arab, bolgár, dán, német, spanyol, finn, francia, héber, hindi, magyar, indonéz, olasz, japán, koreai, maláj, holland, norvég, lengyel, portugál (PT/BR), román, thai, orosz, szlovák, svéd, filippínó, angol, thai, orosz, kínai, filippínó
--**RTL támogatás**– Jobbról balra haladó arab és héber nyelv támogatása
--**Többnyelvű README-k**— 30 teljes dokumentáció fordítás
--**Nyelvválasztó**— Globe ikon a fejlécben a valós idejű váltáshoz
+**How OmniRoute solves it:**
-
-🔄 13. "Többre van szükségem, mint csevegésre – beágyazásra, képekre, hanganyagra van szükségem"
+- **Unified Logs Dashboard** — 4 tabs: Request Logs, Proxy Logs, Audit Logs, Console
+- **Console Log Viewer** — Real-time terminal-style viewer with color-coded levels, auto-scroll, search, filter
+- **SQLite Proxy Logs** — Persistent logs that survive server restarts
+- **Translator Playground** — 4 debugging modes: Playground (format translation), Chat Tester (round-trip), Test Bench (batch), Live Monitor (real-time)
+- **Request Telemetry** — p50/p95/p99 latency + X-Request-Id tracing
+- **File-Based Logging with Rotation** — App logs rotate by size, retention days, and archive count; call log artifacts rotate by retention days and file count
+- **System Info Report** — `npm run system-info` generates `system-info.txt` with your full environment (Node version, OmniRoute version, OS, CLI tools, Docker/PM2 status). Attach it when reporting issues for instant triage.
-Az AI nem csak a csevegés befejezése. A fejlesztőknek képeket kell generálniuk, hangot kell átírniuk, beágyazást kell létrehozniuk a RAG számára, át kell sorolniuk a dokumentumokat, és moderálniuk kell a tartalmat. Minden API más végponttal és formátummal rendelkezik.
+
-**Hogyan oldja meg az OmniRoute:**
+
+🏗️ 11. "Deploying and maintaining the gateway is complex"
--**Beágyazások**– `/v1/beágyazások' 6 szolgáltatóval és 9+ modellel
--**Képgenerálás**– `/v1/images/generations' 10 szolgáltatóval és 20+ modellel (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
--**Text-to-Video**— `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) és SD WebUI
--**Text-to-Music**— `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
--**Audio átírás**— `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
--**Text-to-Speech**— `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3,**Inworld**,**Cartesia**,**PlayHT**, + meglévő szolgáltatók
--**Moderációk**— `/v1/moderations` — Tartalombiztonsági ellenőrzések
--**Reranging**— `/v1/rerank` — Dokumentumreleváns átsorolás
--**Responses API**- Teljes `/v1/responses` támogatás a Codexhez
+Installing, configuring, and maintaining an AI proxy across different environments (local, VPS, Docker, cloud) is labor-intensive. Problems like hardcoded paths, `EACCES` on directories, port conflicts, and cross-platform builds add friction.
-
-🧪 14. "Nincs módom tesztelni és összehasonlítani a minőséget a különböző modellek között"
+**How OmniRoute solves it:**
-A fejlesztők szeretnék tudni, hogy melyik modell a legjobb az ő használati esetükben – kód, fordítás, érvelés –, de a manuális összehasonlítás lassú. Nincsenek integrált eval eszközök.
+- **npm global install** — `npm install -g omniroute && omniroute` — done
+- **Docker Multi-Platform** — AMD64 + ARM64 native (Apple Silicon, AWS Graviton, Raspberry Pi)
+- **Docker Compose Profiles** — `base` (no CLI tools) and `cli` (with Claude Code, Codex, OpenClaw)
+- **Electron Desktop App** — Native app for Windows/macOS/Linux with system tray, auto-start, offline mode
+- **Split-Port Mode** — API and Dashboard on separate ports for advanced scenarios (reverse proxy, container networking)
+- **Cloud Sync** — Config synchronization across devices via Cloudflare Workers
+- **DB Backups** — Automatic backup, restore, export and import of all settings, with `DISABLE_SQLITE_AUTO_BACKUP` for externally managed backups
-**Hogyan oldja meg az OmniRoute:**
+
--**LLM-értékelések**— Arany készlet tesztelése 10 előre betöltött esettel, beleértve az üdvözlést, a matematikát, a földrajzot, a kódgenerálást, a JSON-megfelelőséget, a fordítást, a leértékelést, a biztonsági megtagadást
--**4 egyezési stratégia**– "pontos", "tartalmazza", "regex", "egyéni" (JS függvény)
--**Translator Playground Test Bench**- Kötegelt tesztelés több bemenettel és várható kimenettel, szolgáltatók közötti összehasonlítás
--**Csevegés tesztelő**- Teljes körút vizuális válaszmegjelenítéssel
--**Élő monitor**– Valós idejű adatfolyam a proxyn keresztül folyó összes kérésről
+
+🌍 12. "The interface is English-only and my team doesn't speak English"
-
-📈 15. "A teljesítmény elvesztése nélkül kell skáláznom"
+Teams in non-English-speaking countries, especially in Latin America, Asia, and Europe, struggle with English-only interfaces. Language barriers reduce adoption and increase configuration errors.
-A kérelmek mennyiségének növekedésével ugyanazok a kérdések gyorsítótárazás nélkül duplikált költségeket generálnak. Idempotencia nélkül a duplikált hulladékfeldolgozási kérelmek. A szolgáltatónkénti díjkorlátokat be kell tartani.
+**How OmniRoute solves it:**
-**Hogyan oldja meg az OmniRoute:**
+- **Dashboard i18n — 30 Languages** — All 500+ keys translated including Arabic, Bulgarian, Danish, German, Spanish, Finnish, French, Hebrew, Hindi, Hungarian, Indonesian, Italian, Japanese, Korean, Malay, Dutch, Norwegian, Polish, Portuguese (PT/BR), Romanian, Russian, Slovak, Swedish, Thai, Ukrainian, Vietnamese, Chinese, Filipino, English
+- **RTL Support** — Right-to-left support for Arabic and Hebrew
+- **Multi-Language READMEs** — 30 complete documentation translations
+- **Language Selector** — Globe icon in header for real-time switching
--**Szemantikus gyorsítótár**– A kétszintű gyorsítótár (aláírás + szemantikai) csökkenti a költségeket és a késleltetést
--**Idempotency kérése**– 5 másodperces deduplikációs ablak azonos kérések esetén
--**Drátakorlát észlelése**– Szolgáltatónkénti RPM, minimális rés és maximális egyidejű követés
--**Szerkeszthető sebességkorlátok**- Konfigurálható alapértékek a Beállítások → Kitartással ellenálló képesség menüpontban
--**API Key Validation Cache**– 3-szintű gyorsítótár az éles teljesítményhez
--**Egészségügyi irányítópult telemetriával**— p50/p95/p99 késleltetés, gyorsítótár statisztika, üzemidő
+
-
-🤖 16. "Globálisan szeretném irányítani a modell viselkedését"
+
+🔄 13. "I need more than chat — I need embeddings, images, audio"
-Azok a fejlesztők, akik minden választ egy adott nyelven, egy adott hangnemben szeretnének, vagy korlátozni szeretnék az érvelési tokeneket. Ennek konfigurálása minden eszközben/kérelemben nem praktikus.
+AI isn't just chat completion. Devs need to generate images, transcribe audio, create embeddings for RAG, rerank documents, and moderate content. Each API has a different endpoint and format.
-**Hogyan oldja meg az OmniRoute:**
+**How OmniRoute solves it:**
--**Rendszerprompt Injection**— Globális prompt minden kérelemre vonatkozik
--**A költségkeret átgondolásának ellenőrzése**– Indoklási token-kiosztás ellenőrzése kérésenként (áthaladó, automatikus, egyéni, adaptív)
--**9 Routing Strategies**– Globális stratégiák, amelyek meghatározzák a kérések elosztását
--**Wildcard Router**– a „szolgáltató/*” minták dinamikusan továbbítanak bármely szolgáltatóhoz
--**Kombinációs engedélyezés/letiltás váltás**- A kombók váltása közvetlenül az irányítópultról
--**Provider Toggle**— Egy szolgáltató összes kapcsolatának engedélyezése/letiltása egyetlen kattintással
--**Letiltott szolgáltatók**- Adott szolgáltatók kizárása a `/v1/models' listáról
+- **Embeddings** — `/v1/embeddings` with 6 providers and 9+ models
+- **Image Generation** — `/v1/images/generations` with 10 providers and 20+ models (OpenAI, xAI, Together, Fireworks, Nebius, Hyperbolic, NanoBanana, Antigravity, SD WebUI, ComfyUI)
+- **Text-to-Video** — `/v1/videos/generations` — ComfyUI (AnimateDiff, SVD) and SD WebUI
+- **Text-to-Music** — `/v1/music/generations` — ComfyUI (Stable Audio Open, MusicGen)
+- **Audio Transcription** — `/v1/audio/transcriptions` — Whisper + Nvidia NIM, HuggingFace, Qwen3
+- **Text-to-Speech** — `/v1/audio/speech` — ElevenLabs, Nvidia NIM, HuggingFace, Coqui, Tortoise, Qwen3, **Inworld**, **Cartesia**, **PlayHT**, + existing providers
+- **Moderations** — `/v1/moderations` — Content safety checks
+- **Reranking** — `/v1/rerank` — Document relevance reranking
+- **Responses API** — Full `/v1/responses` support for Codex
-
-🧰 17. "Szükségem van az MCP-eszközökre, mint első osztályú termékképességekre"
+
-Sok mesterséges intelligencia-átjáró csak rejtett megvalósítási részletként teszi közzé az MCP-t. A csapatoknak látható, kezelhető műveleti rétegre van szükségük.
+
+🧪 14. "I have no way to test and compare quality across models"
-**Hogyan oldja meg az OmniRoute:**
+Developers want to know which model is best for their use case — code, translation, reasoning — but comparing manually is slow. No integrated eval tools exist.
-- Az MCP megjelenik az irányítópult navigációs és végponti protokoll lapján
-- Dedikált MCP-kezelési oldal folyamatokkal, eszközökkel, hatókörökkel és audittal
-- Beépített gyorsindítás az "omniroute --mcp" és a kliens beépítéséhez
+**How OmniRoute solves it:**
-
-🧠 18. "A2A hangszerelésre van szükségem szinkronizálással + adatfolyam feladatútvonalak"
+- **LLM Evaluations** — Golden set testing with 10 pre-loaded cases covering greetings, math, geography, code generation, JSON compliance, translation, markdown, safety refusal
+- **4 Match Strategies** — `exact`, `contains`, `regex`, `custom` (JS function)
+- **Translator Playground Test Bench** — Batch testing with multiple inputs and expected outputs, cross-provider comparison
+- **Chat Tester** — Full round-trip with visual response rendering
+- **Live Monitor** — Real-time stream of all requests flowing through the proxy
-Az ügynöki munkafolyamatokhoz közvetlen válaszokra és hosszú távú, streamelt végrehajtásra van szükség életciklus-vezérléssel.
+
-**Hogyan oldja meg az OmniRoute:**
+
+📈 15. "I need to scale without losing performance"
-- A2A JSON-RPC végpont ("POST /a2a") "message/send" és "message/stream" paraméterekkel
-- SSE streaming terminál állapot terjesztéssel
-- Feladatéletciklus API-k a "tasks/get" és a "tasks/cancel" számára
+As request volume grows, without caching the same questions generate duplicate costs. Without idempotency, duplicate requests waste processing. Per-provider rate limits must be respected.
-
-🛰️ 19. "Valódi MCP-folyamat-állapotra van szükségem, nem kitalált állapotra"
+**How OmniRoute solves it:**
-Az operatív csapatoknak tudniuk kell, hogy az MCP valóban életben van-e, nem csak azt, hogy egy API elérhető-e.
+- **Semantic Cache** — Two-tier cache (signature + semantic) reduces cost and latency
+- **Request Idempotency** — 5s deduplication window for identical requests
+- **Rate Limit Detection** — Per-provider RPM, min gap, and max concurrent tracking
+- **Editable Rate Limits** — Configurable defaults in Settings → Resilience with persistence
+- **API Key Validation Cache** — 3-tier cache for production performance
+- **Health Dashboard with Telemetry** — p50/p95/p99 latency, cache stats, uptime
-**Hogyan oldja meg az OmniRoute:**
+
-- Futásidejű szívverés fájl PID-vel, időbélyegekkel, szállítással, szerszámszámmal és hatókör móddal
-- MCP állapot API, amely kombinálja a szívverést + a legutóbbi tevékenységet
-- UI állapotkártyák a folyamat/üzemidő/szívverés frissességéhez
+
+🤖 16. "I want to control model behavior globally"
-
-📋 20. "Kivizsgálható MCP-eszköz-végrehajtásra van szükségem"
+Developers who want all responses in a specific language, with a specific tone, or want to limit reasoning tokens. Configuring this in every tool/request is impractical.
-Amikor az eszközök módosítják a konfigurációt vagy működési műveleteket indítanak el, a csapatoknak kriminalisztikai nyomon követhetőségre van szükségük.
+**How OmniRoute solves it:**
-**Hogyan oldja meg az OmniRoute:**
+- **System Prompt Injection** — Global prompt applied to all requests
+- **Thinking Budget Validation** — Reasoning token allocation control per request (passthrough, auto, custom, adaptive)
+- **9 Routing Strategies** — Global strategies that determine how requests are distributed
+- **Wildcard Router** — `provider/*` patterns route dynamically to any provider
+- **Combo Enable/Disable Toggle** — Toggle combos directly from the dashboard
+- **Provider Toggle** — Enable/disable all connections for a provider with one click
+- **Blocked Providers** — Exclude specific providers from `/v1/models` listing
-- SQLite-alapú auditnaplózás MCP-eszközhívásokhoz
-- Szűrések eszköz, siker/kudarc, API-kulcs és oldalszámozás szerint
-- Irányítópult audit táblázat + statisztikai végpontok az automatizáláshoz
+
-
-🔐 21. "Integrációnként hatókörű MCP-engedélyekre van szükségem"
+
+🧰 17. "I need MCP tools as first-class product capabilities"
-A különböző ügyfeleknek a legkevesebb jogosultsággal kell rendelkezniük az eszközkategóriákhoz.
+Many AI gateways expose MCP only as a hidden implementation detail. Teams need a visible, manageable operation layer.
-**Hogyan oldja meg az OmniRoute:**
+**How OmniRoute solves it:**
-- 10 szemcsés MCP hatókör az ellenőrzött szerszámhozzáféréshez
-- Hatályérvényesítés és láthatóság az MCP-kezelő felületen
-- Biztonságos alaphelyzet az üzemi szerszámokhoz
+- MCP appears in the dashboard navigation and endpoint protocol tab
+- Dedicated MCP management page with process, tools, scopes, and audit
+- Built-in quick-start for `omniroute --mcp` and client onboarding
-
-⚙️ 22. "Üzemeltetési vezérlőkre van szükségem átcsoportosítás nélkül"
+
-A csapatoknak gyors futásidejű változtatásokra van szükségük incidensek vagy költségesemények során.
+
+🧠 18. "I need A2A orchestration with sync + stream task paths"
-**Hogyan oldja meg az OmniRoute:**
+Agent workflows need both direct replies and long-running streamed execution with lifecycle control.
-- A kombinált aktiválás váltása közvetlenül az MCP műszerfaláról
-- Rugalmassági profilok alkalmazása előre meghatározott házirend-csomagokból
-- Állítsa vissza a megszakító állapotát ugyanarról a kezelőpanelről
+**How OmniRoute solves it:**
-
-🔄 23. "Szükségem van az élő A2A feladat életciklusának láthatóságára és törlésére"
+- A2A JSON-RPC endpoint (`POST /a2a`) with `message/send` and `message/stream`
+- SSE streaming with terminal state propagation
+- Task lifecycle APIs for `tasks/get` and `tasks/cancel`
-Az életciklus láthatósága nélkül a feladat-incidensek nehezen osztályozhatók.
+
-**Hogyan oldja meg az OmniRoute:**
+
+🛰️ 19. "I need real MCP process health, not guessed status"
-- Feladatok listázása/szűrés állapot/készség szerint oldalszámozással
-- A feladatok metaadatainak, eseményeinek és műtermékeinek részletezése
-- Feladat törlési végpont és felhasználói felület művelet megerősítéssel
+Operational teams need to know if MCP is actually alive, not just whether an API is reachable.
-
-🌊 24. "Aktív adatfolyam-metrikákra van szükségem A2A terheléshez"
+**How OmniRoute solves it:**
-A streamelési munkafolyamatok működési betekintést igényelnek a párhuzamosság és az élő kapcsolatok terén.
+- Runtime heartbeat file with PID, timestamps, transport, tool count, and scope mode
+- MCP status API combining heartbeat + recent activity
+- UI status cards for process/uptime/heartbeat freshness
-**Hogyan oldja meg az OmniRoute:**
+
-- Az A2A állapotba integrált aktív folyamszámlálók
-- Utolsó feladat időbélyegzője és állapotonkénti száma
-- A2A műszerfalkártyák a valós idejű műveletek figyeléséhez
+
+📋 20. "I need auditable MCP tool execution"
-
-🪪 25. "Szabványos ügynökfelderítésre van szükségem az ügyfelek számára"
+When tools mutate config or trigger ops actions, teams need forensic traceability.
-A külső klienseknek és hangszerelőknek géppel olvasható metaadatokra van szükségük a bevezetéshez.
+**How OmniRoute solves it:**
-**Hogyan oldja meg az OmniRoute:**
+- SQLite-backed audit logging for MCP tool calls
+- Filters by tool, success/failure, API key, and pagination
+- Dashboard audit table + stats endpoints for automation
-- Az ügynökkártya elérhető a `/.well-known/agent.json' címen
-- A menedzsment felületen látható képességek és készségek
-- Az A2A állapot API felfedezési metaadatokat tartalmaz az automatizáláshoz
+
-
-🧭 26. "Protokoll felfedezhetőségre van szükségem a termék felhasználói élményében"
+
+🔐 21. "I need scoped MCP permissions per integration"
-Ha a felhasználók nem fedezik fel a protokollfelületeket, az elfogadás és a támogatás minősége csökken.
+Different clients should have least-privilege access to tool categories.
-**Hogyan oldja meg az OmniRoute:**
+**How OmniRoute solves it:**
-- Összevont**Végpontok**oldal a proxy, MCP, A2A és API végpontok lapjaival
-- Inline szolgáltatás állapotát váltja (Online/Offline) MCP és A2A esetén
-- Hivatkozások az áttekintésből a dedikált kezelőlapokhoz
+- 10 granular MCP scopes for controlled tool access
+- Scope enforcement and visibility in MCP management UI
+- Safe default posture for operational tooling
-
-🧪 27. "Végponttól végpontig terjedő protokoll-érvényesítésre van szükségem valódi ügyfelekkel"
+
-A próbatesztek nem elegendőek a protokoll-kompatibilitás ellenőrzéséhez a kiadás előtt.
+
+⚙️ 22. "I need operational controls without redeploying"
-**Hogyan oldja meg az OmniRoute:**
+Teams need quick runtime changes during incidents or cost events.
-- E2E csomag, amely elindítja az alkalmazást, és valódi MCP SDK kliens szállítást használ
-- Az A2A kliens teszteli az áramlások felfedezését, küldését, streamingjét, lekérését és megszakítását
-- Az állítások keresztellenőrzése az MCP audit és az A2A feladatok API-jával szemben
+**How OmniRoute solves it:**
-
-📡 28. "Egységes megfigyelhetőségre van szükségem az összes felületen"
+- Switch combo activation directly from MCP dashboard
+- Apply resilience profiles from pre-defined policy packs
+- Reset circuit breaker state from the same operations panel
-A megfigyelhetőség protokoll szerinti felosztása vakfoltokat és hosszabb MTTR-t hoz létre.
+
-**Hogyan oldja meg az OmniRoute:**
+
+🔄 23. "I need live A2A task lifecycle visibility and cancellation"
-- Egységes irányítópultok/naplók/analytics egy termékben
-- Egészség + audit + kérés telemetria OpenAI, MCP és A2A rétegeken keresztül
-- Működési API-k az állapothoz és az automatizáláshoz
+Without lifecycle visibility, task incidents become hard to triage.
-
-💼 29. "Egy futási időre van szükségem a proxy + eszközök + ügynök hangszereléshez"
+**How OmniRoute solves it:**
-Számos külön szolgáltatás futtatása növeli a működési költségeket és a hibamódokat.
+- Task listing/filtering by state/skill with pagination
+- Drill-down on task metadata, events, and artifacts
+- Task cancellation endpoint and UI action with confirmation
-**Hogyan oldja meg az OmniRoute:**
+
-- OpenAI-kompatibilis proxy, MCP szerver és A2A szerver egy veremben
-- Megosztott hitelesítés, rugalmasság, adattárolás és megfigyelhetőség
-- Konzisztens politikai modell az összes interakciós felületen
+
+🌊 24. "I need active stream metrics for A2A load"
-
-🚀 30. "Ügynöki munkafolyamatokat ragasztókód szétterülése nélkül kell szállítanom"
+Streaming workflows require operational insight into concurrency and live connections.
-A csapatok veszítenek sebességükből, amikor több ad-hoc szolgáltatást és szkriptet illesztenek össze.
+**How OmniRoute solves it:**
-**Hogyan oldja meg az OmniRoute:**
+- Active stream counters integrated into A2A status
+- Last task timestamp and per-state counts
+- A2A dashboard cards for real-time ops monitoring
-- Egységes végpont stratégia az ügyfelek és ügynökök számára
-- Beépített protokollkezelő felhasználói felületek és füstellenőrzési útvonalak
-- Gyártásra kész alapok (biztonság, naplózás, rugalmasság, biztonsági mentés)
+
+
+
+🪪 25. "I need standard agent discovery for clients"
+
+External clients and orchestrators need machine-readable metadata for onboarding.
+
+**How OmniRoute solves it:**
+
+- Agent Card exposed at `/.well-known/agent.json`
+- Capabilities and skills shown in management UI
+- A2A status API includes discovery metadata for automation
+
+
+
+
+🧭 26. "I need protocol discoverability in the product UX"
+
+If users cannot discover protocol surfaces, adoption and support quality drop.
+
+**How OmniRoute solves it:**
+
+- Consolidated **Endpoints** page with tabs for Proxy, MCP, A2A, and API Endpoints
+- Inline service status toggles (Online/Offline) for MCP and A2A
+- Links from overview to dedicated management tabs
+
+
+
+
+🧪 27. "I need end-to-end protocol validation with real clients"
+
+Mock tests are not enough to validate protocol compatibility before release.
+
+**How OmniRoute solves it:**
+
+- E2E suite that boots app and uses real MCP SDK client transport
+- A2A client tests for discovery, send, stream, get, and cancel flows
+- Cross-check assertions against MCP audit and A2A tasks APIs
+
+
+
+
+📡 28. "I need unified observability across all interfaces"
+
+Splitting observability by protocol creates blind spots and longer MTTR.
+
+**How OmniRoute solves it:**
+
+- Unified dashboards/logs/analytics in one product
+- Health + audit + request telemetry across OpenAI, MCP, and A2A layers
+- Operational APIs for status and automation
+
+
+
+
+💼 29. "I need one runtime for proxy + tools + agent orchestration"
+
+Running many separate services increases operational cost and failure modes.
+
+**How OmniRoute solves it:**
+
+- OpenAI-compatible proxy, MCP server, and A2A server in one stack
+- Shared auth, resilience, data store, and observability
+- Consistent policy model across all interaction surfaces
+
+
+
+
+🚀 30. "I need to ship agentic workflows without glue-code sprawl"
+
+Teams lose velocity when stitching multiple ad-hoc services and scripts.
+
+**How OmniRoute solves it:**
+
+- Unified endpoint strategy for clients and agents
+- Built-in protocol management UIs and smoke validation paths
+- Production-ready foundations (security, logging, resilience, backup)
+
+
### Example Playbooks (Integrated Use Cases)
-**A játékkönyv: Maximalizálja a fizetett előfizetést + olcsó biztonsági mentés**```txt
+**Playbook A: Maximize paid subscription + cheap backup**
+
+```txt
Combo: "maximize-claude"
1. cc/claude-opus-4-6
2. glm/glm-4.7
@@ -609,21 +690,23 @@ Combo: "maximize-claude"
Monthly cost: $20 + small backup spend
Outcome: higher quality, near-zero interruption
-````
+```
-**Playbook B: Zéró költségű kódolási verem**```txt
+**Playbook B: Zero-cost coding stack**
+
+```txt
Combo: "free-forever"
-
-1. gc/gemini-3-flash
-2. if/kimi-k2-thinking
-3. qw/qwen3-coder-plus
+ 1. gc/gemini-3-flash
+ 2. if/kimi-k2-thinking
+ 3. qw/qwen3-coder-plus
Monthly cost: $0
Outcome: stable free coding workflow
+```
-````
+**Playbook C: 24/7 always-on fallback chain**
-**Playbook C: 24/7 mindig bekapcsolt tartalék lánc**```txt
+```txt
Combo: "always-on"
1. cc/claude-opus-4-6
2. cx/gpt-5.2-codex
@@ -632,122 +715,134 @@ Combo: "always-on"
5. if/kimi-k2-thinking
Outcome: deep fallback depth for deadline-critical workloads
-````
+```
-**D játékkönyv: Az ügynök MCP + A2A-val működik**```txt
+**Playbook D: Agent ops with MCP + A2A**
-1. Start MCP transport (`omniroute --mcp`) for tool-driven operations
-2. Run A2A tasks via `message/send` and `message/stream`
-3. Observe via /dashboard/endpoint (MCP and A2A tabs)
-4. Toggle services via inline status controls
-
-````
+```txt
+1) Start MCP transport (`omniroute --mcp`) for tool-driven operations
+2) Run A2A tasks via `message/send` and `message/stream`
+3) Observe via /dashboard/endpoint (MCP and A2A tabs)
+4) Toggle services via inline status controls
+```
---
## 🆓 Start Free — Zero Configuration Cost
-> Állítsa be az AI-kódolást percek alatt**0 USD/hó**áron. Csatlakoztassa ezeket az ingyenes fiókokat, és használja a beépített**Free Stack**kombinációt.
+> Setup AI coding in minutes at **$0/month**. Connect these free accounts and use the built-in **Free Stack** combo.
-| lépés | Akció | Szolgáltatók feloldva |
-| ---- | --------------------------------------------------- | ------------------------------------------------------------------ |
-| 1 | Csatlakozás**Kiro**(AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 —**korlátlan**|
-| 2 | Csatlakozás**Qoder**(Google OAuth) | kimi-k2-gondolkodás, qwen3-coder-plus, deepseek-r1... —**korlátlan**|
-| 3 | Csatlakoztassa a**Qwen**(eszközkód) | qwen3-coder-plus, qwen3-coder-flash... —**korlátlan**|
-| 4 | Csatlakozás**Gemini CLI**(Google OAuth) | gemini-3-flash, gemini-2.5-pro –**180K/hó ingyenes**|
-| 5 | `/dashboard/combos` →**Ingyenes köteg ($0)**sablon | Körbe-körbe minden ingyenes szolgáltató automatikusan |
+| Step | Action | Providers Unlocked |
+| ---- | -------------------------------------------------- | ------------------------------------------------------------------ |
+| 1 | Connect **Kiro** (AWS Builder ID OAuth) | Claude Sonnet 4.5, Haiku 4.5 — **unlimited** |
+| 2 | Connect **Qoder** (Google OAuth) | kimi-k2-thinking, qwen3-coder-plus, deepseek-r1... — **unlimited** |
+| 3 | Connect **Qwen** (Device Code) | qwen3-coder-plus, qwen3-coder-flash... — **unlimited** |
+| 4 | Connect **Gemini CLI** (Google OAuth) | gemini-3-flash, gemini-2.5-pro — **180K/mo free** |
+| 5 | `/dashboard/combos` → **Free Stack ($0)** template | Round-robin all free providers automatically |
-**Mutasson bármely IDE/CLI-t a következőre:**`http://localhost:20128/v1` · API-kulcs: `any-string` · Kész.
+**Point any IDE/CLI to:** `http://localhost:20128/v1` · API Key: `any-string` · Done.
->**Opcionális extra lefedettség (szintén ingyenes):**Groq API kulcs (30 RPM ingyenes), NVIDIA NIM (40 RPM ingyenes, 70+ modell), Cerebras (1 millió tok/nap), LongCat API kulcs (50 millió token/nap!), Cloudflare Workers AI (10 000 neuron/nap, 50+ modell).## Gyors kezdés
+> **Optional extra coverage (also free):** Groq API key (30 RPM free), NVIDIA NIM (40 RPM free, 70+ models), Cerebras (1M tok/day), LongCat API key (50M tokens/day!), Cloudflare Workers AI (10K Neurons/day, 50+ models).
+
+## Gyors kezdés
### 1) Install and run
```bash
npm install -g omniroute
omniroute
-````
+```
-> **pnpm felhasználók:**Telepítés után futtassa a `pnpm approve-builds -g` parancsot, hogy engedélyezze a `better-sqlite3` és a `@swc/core` által igényelt natív build szkripteket:
+> **pnpm users:** Run `pnpm approve-builds -g` after install to enable native build scripts required by `better-sqlite3` and `@swc/core`:
>
> ```bash
> pnpm install -g omniroute
-> pnpm approve-builds -g # Válassza ki az összes csomagot → jóváhagyja
+> pnpm approve-builds -g # Select all packages → approve
> omniroute
> ```
-Az irányítópult a „http://localhost:20128” címen nyílik meg, az API alap URL-címe pedig „http://localhost:20128/v1”.
+Dashboard opens at `http://localhost:20128` and API base URL is `http://localhost:20128/v1`.
-| Parancs | Leírás |
-| ----------------------- | ----------------------------------------------------------------------- |
-| `omniroute` | Szerver indítása (`PORT=20128`, API és irányítópult ugyanazon a porton) |
-| `omniroute --port 3000` | A kanonikus/API port beállítása 3000 |
-| `omniroute --mcp` | MCP-kiszolgáló indítása (stdio szállítás) |
-| `omniroute --no-open` | Ne nyissa meg automatikusan a böngészőt |
-| `omniroute --help` | Segítség megjelenítése |
+| Command | Description |
+| ----------------------- | ----------------------------------------------------------- |
+| `omniroute` | Start server (`PORT=20128`, API and dashboard on same port) |
+| `omniroute --port 3000` | Set canonical/API port to 3000 |
+| `omniroute --mcp` | Start MCP server (stdio transport) |
+| `omniroute --no-open` | Don't auto-open browser |
+| `omniroute --help` | Show help |
-Opcionális osztott portos mód:```bash
+Optional split-port mode:
+
+```bash
PORT=20128 DASHBOARD_PORT=20129 omniroute
-
-# API: http://localhost:20128/v1
-
+# API: http://localhost:20128/v1
# Dashboard: http://localhost:20129
-
-````
+```
### Long-Running Streaming Timeouts
-A legtöbb telepítéshez csak a következőkre van szüksége:
+For most deployments, you only need:
-| Változó | Alapértelmezett | Cél |
-| ------------------------- | ------------------------------ | -------------------------------------------------------------- -------------------------------------------------------------- |
-| `REQUEST_TIMEOUT_MS` | "600000" | Megosztott alapvonal az upstream lekéréshez, a rejtett Undici-időtúllépésekhez, a TLS-ujjlenyomat-kérésekhez és az API-híd kérés/proxy időtúllépéséhez |
-| `STREAM_IDLE_TIMEOUT_MS` | örökli a `REQUEST_TIMEOUT_MS' | Maximális hézag a streaming darabok között, mielőtt az OmniRoute megszakítja az SSE adatfolyamot |
+| Variable | Default | Purpose |
+| ------------------------ | ----------------------------- | --------------------------------------------------------------------------------------------------------------------------- |
+| `REQUEST_TIMEOUT_MS` | `600000` | Shared baseline for upstream fetch, hidden Undici timeouts, TLS fingerprint requests, and API bridge request/proxy timeouts |
+| `STREAM_IDLE_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Maximum gap between streaming chunks before OmniRoute aborts the SSE stream |
-A visszamenőleges kompatibilitás megmarad: a meglévő „FETCH_TIMEOUT_MS”, „API_BRIDGE_PROXY_TIMEOUT_MS” és más rétegenkénti időtúllépési változók továbbra is működnek, és felülírják a megosztott alapvonalat.
+Backward compatibility is preserved: existing `FETCH_TIMEOUT_MS`, `API_BRIDGE_PROXY_TIMEOUT_MS`, and other per-layer timeout vars still work and override the shared baseline.
-Speciális felülírások állnak rendelkezésre, ha finomabb vezérlésre van szüksége:| Változó | Alapértelmezett | Cél |
-| ----------------------------------------- | ------------------------------------------- | --------------------------------------------------------------------- |
-| `FETCH_TIMEOUT_MS` | örökli a `REQUEST_TIMEOUT_MS' | A fő lekérés megszakítási jele által használt teljes felfelé irányuló kérés időtúllépése |
-| `FETCH_HEADERS_TIMEOUT_MS` | örökli a `FETCH_TIMEOUT_MS` | Undici időkorlát az upstream válaszfejlécek fogadására |
-| `FETCH_BODY_TIMEOUT_MS` | örökli a `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) |
-| `FETCH_CONNECT_TIMEOUT_MS` | "30000" | Undici TCP csatlakozási időtúllépés |
-| `FETCH_KEEPALIVE_TIMEOUT_MS` | "4000" | Undici tétlen életben tartási aljzat időtúllépése |
-| `TLS_CLIENT_TIMEOUT_MS` | örökli a `FETCH_TIMEOUT_MS` | Időtúllépés a `wreq-js` | segítségével küldött TLS-ujjlenyomat-kéréseknél
-| `API_BRIDGE_PROXY_TIMEOUT_MS` | örökli a „REQUEST_TIMEOUT_MS” vagy „30000” | Időtúllépés a „/v1” proxy API-portról az irányítópult-portra való továbbítására |
-| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | "max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)" | Bejövő kérés időtúllépése az API-hídszerveren |
-| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | "60000" | Bejövő fejléc időtúllépése az API-hídszerveren |
-| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | "5000" | Életben tartás időtúllépés az API-hídszerveren |
-| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | "0" | Socket inaktivitási időtúllépése az API-hídszerveren ("0" letiltja) |
+Advanced overrides are available if you need finer control:
-Ha az OmniRoute alkalmazást az Nginx, Caddy, Cloudflare vagy más fordított proxy mögött futtatja, győződjön meg arról, hogy a proxy
-az időtúllépések is magasabbak, mint az OmniRoute adatfolyam/lekérési időkorlátok.### 2) Connect providers and create your API key
+| Variable | Default | Purpose |
+| ---------------------------------------- | ------------------------------------------ | -------------------------------------------------------------------- |
+| `FETCH_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` | Total upstream request timeout used by the main fetch abort signal |
+| `FETCH_HEADERS_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit for receiving upstream response headers |
+| `FETCH_BODY_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Undici time limit between upstream body chunks (`0` disables it) |
+| `FETCH_CONNECT_TIMEOUT_MS` | `30000` | Undici TCP connect timeout |
+| `FETCH_KEEPALIVE_TIMEOUT_MS` | `4000` | Undici idle keep-alive socket timeout |
+| `TLS_CLIENT_TIMEOUT_MS` | inherits `FETCH_TIMEOUT_MS` | Timeout for TLS fingerprint requests made through `wreq-js` |
+| `API_BRIDGE_PROXY_TIMEOUT_MS` | inherits `REQUEST_TIMEOUT_MS` or `30000` | Timeout for `/v1` proxy forwarding from API port to dashboard port |
+| `API_BRIDGE_SERVER_REQUEST_TIMEOUT_MS` | `max(API_BRIDGE_PROXY_TIMEOUT_MS, 300000)` | Incoming request timeout on the API bridge server |
+| `API_BRIDGE_SERVER_HEADERS_TIMEOUT_MS` | `60000` | Incoming header timeout on the API bridge server |
+| `API_BRIDGE_SERVER_KEEPALIVE_TIMEOUT_MS` | `5000` | Keep-alive timeout on the API bridge server |
+| `API_BRIDGE_SERVER_SOCKET_TIMEOUT_MS` | `0` | Socket inactivity timeout on the API bridge server (`0` disables it) |
-1. Nyissa meg az Irányítópult → „Szolgáltatók” menüpontot, és csatlakoztasson legalább egy szolgáltatót (OAuth- vagy API-kulcs).
-2. Nyissa meg a Dashboard → `Végpontok` menüpontot, és hozzon létre egy API-kulcsot.
-3. (Opcionális) Nyissa meg az Irányítópult → Kombók menüpontot, és állítsa be a tartalék láncot.### 3) Point your coding tool to OmniRoute
+If you run OmniRoute behind Nginx, Caddy, Cloudflare, or another reverse proxy, make sure the proxy
+timeouts are also higher than your OmniRoute stream/fetch timeouts.
+
+### 2) Connect providers and create your API key
+
+1. Open Dashboard → `Providers` and connect at least one provider (OAuth or API key).
+2. Open Dashboard → `Endpoints` and create an API key.
+3. (Optional) Open Dashboard → `Combos` and set your fallback chain.
+
+### 3) Point your coding tool to OmniRoute
```txt
Base URL: http://localhost:20128/v1
API Key: [copy from Endpoint page]
Model: if/kimi-k2-thinking (or any provider/model prefix)
-````
+```
-Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode és OpenAI-kompatibilis SDK-kkal működik.### 4) Enable and validate protocols (v2.0)
+Works with Claude Code, Codex CLI, Gemini CLI, Cursor, Cline, OpenClaw, OpenCode, and OpenAI-compatible SDKs.
-**MCP (szerszámvezérelt műveletekhez):**```bash
+### 4) Enable and validate protocols (v2.0)
+
+**MCP (for tool-driven operations):**
+
+```bash
omniroute --mcp
+```
-````
-
-Ezután csatlakoztassa MCP-kliensét `stdio'-n keresztül, és tesztelje az olyan eszközöket, mint:
+Then connect your MCP client over `stdio` and test tools like:
- `omniroute_get_health`
- `omniroute_list_combos`
-**A2A (ügynök-ügynök munkafolyamatokhoz):**```bash
+**A2A (for agent-to-agent workflows):**
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
```bash
curl -X POST http://localhost:20128/a2a \
@@ -761,7 +856,9 @@ curl -X POST http://localhost:20128/a2a \
npm run test:protocols:e2e
```
-Ez a csomag érvényesíti a valódi MCP- és A2A-kliensfolyamokat egy futó alkalmazással szemben.### Alternative: run from source
+This suite validates real MCP and A2A client flows against a running app.
+
+### Alternative: run from source
```bash
cp .env.example .env
@@ -769,14 +866,13 @@ npm install
PORT=20128 DASHBOARD_PORT=20129 NEXT_PUBLIC_BASE_URL=http://localhost:20129 npm run dev
```
-
+
+Void Linux (`xbps-src` template)
-Érvénytelen Linux (`xbps-src` sablon)
-
-Void Linux felhasználók számára natív csomagot készíthet az `xbps-src` használatával. Mentse ezt a blokkot `srcpkgs/omniroute/template' néven:```bash
+For Void Linux users, you can build a native package using `xbps-src`. Save this block as `srcpkgs/omniroute/template`:
+```bash
# Template file for 'omniroute'
-
pkgname=omniroute
version=3.4.1
revision=1
@@ -788,7 +884,7 @@ license="MIT"
homepage="https://github.com/diegosouzapw/OmniRoute"
distfiles="https://github.com/diegosouzapw/OmniRoute/archive/refs/tags/v${version}.tar.gz"
checksum=009400afee90a9f32599d8fe734145cfd84098140b7287990183dde45ae2245b
-system_accounts="\_omniroute"
+system_accounts="_omniroute"
omniroute_homedir="/var/lib/omniroute"
export NODE_ENV=production
export npm_config_engine_strict=false
@@ -796,71 +892,70 @@ export npm_config_loglevel=error
export npm_config_fund=false
export npm_config_audit=false
-do_build() { # Determine target CPU arch for node-gyp
-local \_gyp_arch
-case "$XBPS_TARGET_MACHINE" in
-aarch64*) \_gyp_arch=arm64 ;;
-armv7*|armv6*) \_gyp_arch=arm ;;
-i686*) \_gyp_arch=ia32 ;;
-\*) \_gyp_arch=x64 ;;
-esac
+do_build() {
+ # Determine target CPU arch for node-gyp
+ local _gyp_arch
+ case "$XBPS_TARGET_MACHINE" in
+ aarch64*) _gyp_arch=arm64 ;;
+ armv7*|armv6*) _gyp_arch=arm ;;
+ i686*) _gyp_arch=ia32 ;;
+ *) _gyp_arch=x64 ;;
+ esac
- # 1) Install all deps – skip scripts (no network in do_build, native modules
- # compiled separately below; better-sqlite3 is serverExternalPackage so
- # Next.js does not execute it during next build)
- NODE_ENV=development npm ci --ignore-scripts
+ # 1) Install all deps – skip scripts (no network in do_build, native modules
+ # compiled separately below; better-sqlite3 is serverExternalPackage so
+ # Next.js does not execute it during next build)
+ NODE_ENV=development npm ci --ignore-scripts
- # 2) Build the Next.js standalone bundle
- npm run build
+ # 2) Build the Next.js standalone bundle
+ npm run build
- # 3) Copy static assets into standalone
- cp -r .next/static .next/standalone/.next/static
- [ -d public ] && cp -r public .next/standalone/public || true
+ # 3) Copy static assets into standalone
+ cp -r .next/static .next/standalone/.next/static
+ [ -d public ] && cp -r public .next/standalone/public || true
- # 4) Compile better-sqlite3 native binding for the target architecture.
- # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
- # without npm altering them.
- local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
- (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
+ # 4) Compile better-sqlite3 native binding for the target architecture.
+ # Use node-gyp directly so CC/CXX from xbps-src cross-toolchain are used
+ # without npm altering them.
+ local _node_gyp=/usr/lib/node_modules/npm/node_modules/node-gyp/bin/node-gyp.js
+ (cd node_modules/better-sqlite3 && node "$_node_gyp" rebuild --arch="$_gyp_arch")
- # 5) Place the compiled binding into the standalone bundle
- local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
- mkdir -p "$_bs3_release"
- cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
+ # 5) Place the compiled binding into the standalone bundle
+ local _bs3_release=.next/standalone/node_modules/better-sqlite3/build/Release
+ mkdir -p "$_bs3_release"
+ cp node_modules/better-sqlite3/build/Release/better_sqlite3.node "$_bs3_release/"
- # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
- # so sharp is not used at runtime; x64 .so files would break aarch64 strip
- rm -rf .next/standalone/node_modules/@img
-
- # 7) Copy pino runtime deps omitted by Next.js static analysis:
- # pino-abstract-transport – required by pino's worker thread
- # split2 – dep of pino-abstract-transport
- # process-warning – dep of pino itself
- for _mod in pino-abstract-transport split2 process-warning; do
- cp -r "node_modules/$_mod" .next/standalone/node_modules/
- done
+ # 6) Remove arch-specific sharp bundles – upstream sets images.unoptimized=true
+ # so sharp is not used at runtime; x64 .so files would break aarch64 strip
+ rm -rf .next/standalone/node_modules/@img
+ # 7) Copy pino runtime deps omitted by Next.js static analysis:
+ # pino-abstract-transport – required by pino's worker thread
+ # split2 – dep of pino-abstract-transport
+ # process-warning – dep of pino itself
+ for _mod in pino-abstract-transport split2 process-warning; do
+ cp -r "node_modules/$_mod" .next/standalone/node_modules/
+ done
}
do_check() {
-npm run test:unit
+ npm run test:unit
}
do_install() {
-vmkdir usr/lib/omniroute/.next
+ vmkdir usr/lib/omniroute/.next
- vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
+ vcopy .next/standalone/. usr/lib/omniroute/.next/standalone
- # Prevent removal of empty Next.js app router dirs by the post-install hook
- for _d in \
- .next/standalone/.next/server/app/dashboard \
- .next/standalone/.next/server/app/dashboard/settings \
- .next/standalone/.next/server/app/dashboard/providers; do
- touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
- done
-
- cat > "${WRKDIR}/omniroute" <<'EOF'
+ # Prevent removal of empty Next.js app router dirs by the post-install hook
+ for _d in \
+ .next/standalone/.next/server/app/dashboard \
+ .next/standalone/.next/server/app/dashboard/settings \
+ .next/standalone/.next/server/app/dashboard/providers; do
+ touch "${DESTDIR}/usr/lib/omniroute/${_d}/.keep"
+ done
+ cat > "${WRKDIR}/omniroute" <<'EOF'
#!/bin/sh
export PORT="${PORT:-20128}"
export DATA_DIR="${DATA_DIR:-${XDG_DATA_HOME:-${HOME}/.local/share}/omniroute}"
@@ -872,10 +967,9 @@ EOF
}
post_install() {
-vlicense LICENSE
+ vlicense LICENSE
}
-
-````
+```
@@ -883,9 +977,11 @@ vlicense LICENSE
## 🐳 Docker
-Az OmniRoute nyilvános Docker-képként érhető el a [Docker Hubon](https://hub.docker.com/r/diegosouzapw/omniroute).
+OmniRoute is available as a public Docker image on [Docker Hub](https://hub.docker.com/r/diegosouzapw/omniroute).
-**Gyors futás:**```bash
+**Quick run:**
+
+```bash
docker run -d \
--name omniroute \
--restart unless-stopped \
@@ -893,85 +989,96 @@ docker run -d \
-p 20128:20128 \
-v omniroute-data:/app/data \
diegosouzapw/omniroute:latest
-````
+```
-**Környezetfájllal:**```bash
+**With environment file:**
+```bash
# Copy and edit .env first
-
cp .env.example .env
docker run -d \
- --name omniroute \
- --restart unless-stopped \
- --stop-timeout 40 \
- --env-file .env \
- -p 20128:20128 \
- -v omniroute-data:/app/data \
- diegosouzapw/omniroute:latest
+ --name omniroute \
+ --restart unless-stopped \
+ --stop-timeout 40 \
+ --env-file .env \
+ -p 20128:20128 \
+ -v omniroute-data:/app/data \
+ diegosouzapw/omniroute:latest
+```
-````
+**Using Docker Compose:**
-**A Docker Compose használata:**```bash
+```bash
# Base profile (no CLI tools)
docker compose --profile base up -d
# CLI profile (Claude Code, Codex, OpenClaw built-in)
docker compose --profile cli up -d
-````
+```
-A Docker-telepítések irányítópult-támogatása mostantól magában foglal egy egykattintásos**Cloudflare Quick Tunnel**-t az "Irányítópult → Végpontok" oldalon. Az első engedélyezés csak szükség esetén tölti le a „cloudflared” funkciót, ideiglenes alagutat indít a jelenlegi „/v1” végponthoz, és megjeleníti a generált „https://\*.trycloudflare.com/v1” URL-t közvetlenül a normál nyilvános URL alatt.
+Dashboard support for Docker deployments now includes a one-click **Cloudflare Quick Tunnel** on `Dashboard → Endpoints`. The first enable downloads `cloudflared` only when needed, starts a temporary tunnel to your current `/v1` endpoint, and shows the generated `https://*.trycloudflare.com/v1` URL directly below your normal public URL.
-Megjegyzések:
+Notes:
-- A Quick Tunnel URL-ek ideiglenesek, és minden újraindítás után megváltoznak.
-- A gyors alagutak nem állnak vissza automatikusan az OmniRoute vagy a tároló újraindítása után. Ha szükséges, engedélyezze őket újra az irányítópulton.
-- A felügyelt telepítés jelenleg támogatja a Linuxot, a macOS-t és a Windowst „x64” / „arm64” rendszeren.
-- A felügyelt gyorsalagutak alapértelmezés szerint a HTTP/2 átvitelt használják, hogy elkerüljék a zajos QUIC UDP puffer figyelmeztetéseket a korlátozott tárolókörnyezetekben. Állítsa be a "CLOUDFLARED_PROTOCOL=quic" vagy az "auto" értéket, ha más átvitelt szeretne.
-- A Docker képek a rendszer CA-gyökereit csomagolják, és átadják a felügyelt "cloudflared"-nek, amely elkerüli a TLS-megbízhatósági hibákat, amikor az alagút a tárolón belül bootstradik.
-- Az SQLite WAL módban fut. Engedélyezni kell a `docker stop' befejezését, hogy az OmniRoute vissza tudja irányítani a legutóbbi változtatásokat a `storage.sqlite' fájlba.
-- A kötegelt Compose-fájlok már beállítottak egy 40 másodperces türelmi időt. Ha közvetlenül futtatja a képet, tartsa be a "--stop-timeout 40" értéket (vagy hasonlót), hogy a kézi leállítások ne szakítsák meg a leállítási tisztítást.
-- Állítsa be a `CLOUDFLARED_BIN=/absolute/path/to/cloudflared' értéket, ha azt szeretné, hogy az OmniRoute egy létező binárist használjon a letöltés helyett.
+- Quick Tunnel URLs are temporary and change after every restart.
+- Quick Tunnels are not auto-restored after an OmniRoute or container restart. Re-enable them from the dashboard when needed.
+- Managed install currently supports Linux, macOS, and Windows on `x64` / `arm64`.
+- Managed Quick Tunnels default to HTTP/2 transport to avoid noisy QUIC UDP buffer warnings in constrained container environments. Set `CLOUDFLARED_PROTOCOL=quic` or `auto` if you want a different transport.
+- Docker images bundle system CA roots and pass them to managed `cloudflared`, which avoids TLS trust failures when the tunnel bootstraps inside the container.
+- SQLite runs in WAL mode. `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`.
+- The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40` (or similar) so manual stops do not cut off shutdown cleanup.
+- Set `CLOUDFLARED_BIN=/absolute/path/to/cloudflared` if you want OmniRoute to use an existing binary instead of downloading one.
-**A Docker Compose with Caddy (HTTPS Auto-TLS) használata:**
+**Using Docker Compose with Caddy (HTTPS Auto-TLS):**
-Az OmniRoute biztonságosan elérhető a Caddy automatikus SSL-kiépítésével. Győződjön meg arról, hogy a domain DNS-rekordja a szerver IP-címére mutat.```yaml
+OmniRoute can be securely exposed using Caddy's automatic SSL provisioning. Ensure your domain's DNS A record points to your server's IP.
+
+```yaml
services:
-omniroute:
-image: diegosouzapw/omniroute:latest
-container_name: omniroute
-restart: unless-stopped
-volumes: - omniroute-data:/app/data
-environment: - PORT=20128 - NEXT_PUBLIC_BASE_URL=https://your-domain.com
+ omniroute:
+ image: diegosouzapw/omniroute:latest
+ container_name: omniroute
+ restart: unless-stopped
+ volumes:
+ - omniroute-data:/app/data
+ environment:
+ - PORT=20128
+ - NEXT_PUBLIC_BASE_URL=https://your-domain.com
-caddy:
-image: caddy:latest
-container_name: caddy
-restart: unless-stopped
-ports: - "80:80" - "443:443"
-command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
+ caddy:
+ image: caddy:latest
+ container_name: caddy
+ restart: unless-stopped
+ ports:
+ - "80:80"
+ - "443:443"
+ command: caddy reverse-proxy --from https://your-domain.com --to http://omniroute:20128
volumes:
-omniroute-data:
+ omniroute-data:
+```
-````
+| Image | Tag | Size | Description |
+| ------------------------ | -------- | ------ | --------------------- |
+| `diegosouzapw/omniroute` | `latest` | ~250MB | Latest stable release |
+| `diegosouzapw/omniroute` | `1.0.3` | ~250MB | Current version |
-| Kép | Címke | Méret | Leírás |
-| ------------------------- | -------- | ------ | ---------------------- |
-| "diegosouzapw/omniroute" | "legújabb" | ~250 MB | Legújabb stabil kiadás |
-| "diegosouzapw/omniroute" | "1.0.3" | ~250 MB | Jelenlegi verzió |---
+---
## 🖥️ Desktop App — Offline & Always-On
-> 🆕**ÚJ!**Az OmniRoute már elérhető**natív asztali alkalmazásként**Windows, macOS és Linux rendszeren.
+> 🆕 **NEW!** OmniRoute is now available as a **native desktop application** for Windows, macOS, and Linux.
-Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. Az elektronalapú alkalmazás a következőket tartalmazza:
+Run OmniRoute as a standalone desktop app — no terminal, no browser, no internet required for local models. The Electron-based app includes:
-- 🖥️**Natív ablak**- Dedikált alkalmazásablak rendszertálca-integrációval
-- 🔄**Automatikus indítás**- Indítsa el az OmniRoute alkalmazást a rendszerbe való bejelentkezéskor
-- 🔔**Natív értesítések**- Értesítést kaphat a kvóta kimerüléséről vagy a szolgáltatói problémákról
-- ⚡**Egykattintásos telepítés**- NSIS (Windows), DMG (macOS), AppImage (Linux)
-- 🌐**Offline mód**- Teljesen offline módban működik a mellékelt szerverrel### Gyors kezdés
+- 🖥️ **Native Window** — Dedicated app window with system tray integration
+- 🔄 **Auto-Start** — Launch OmniRoute on system login
+- 🔔 **Native Notifications** — Get alerts for quota exhaustion or provider issues
+- ⚡ **One-Click Install** — NSIS (Windows), DMG (macOS), AppImage (Linux)
+- 🌐 **Offline Mode** — Works fully offline with bundled server
+
+### Gyors kezdés
```bash
# Development mode
@@ -982,308 +1089,370 @@ npm run electron:build # Current platform
npm run electron:build:win # Windows (.exe)
npm run electron:build:mac # macOS (.dmg) — x64 & arm64
npm run electron:build:linux # Linux (.AppImage)
-````
+```
### System Tray
-Ha minimalizálja, az OmniRoute a tálcán él, gyors műveletekkel:
+When minimized, OmniRoute lives in your system tray with quick actions:
-- Nyissa meg a műszerfalat
-- Szerver port módosítása
-- Lépjen ki az alkalmazásból
+- Open dashboard
+- Change server port
+- Quit application
-📖 Teljes dokumentáció: [`electron/README.md`](electron/README.md)---
+📖 Full documentation: [`electron/README.md`](electron/README.md)
+
+---
## 💰 Pricing at a Glance
-| Tier | Szolgáltató | Költség | Kvóta visszaállítása | Legjobb a |
-| ----------------- | ----------------------------- | ---------------------------------- | ---------------------- | --------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
-| **💳 ELŐFIZETÉS** | Claude Code (Pro) | 20 USD/hó | 5 óra + heti | Már előfizetett |
-| | Codex (Plus/Pro) | 20-200 USD/hó | 5 óra + heti | OpenAI felhasználók |
-| | Gemini CLI | **INGYENES** | 180 000/hó + 1 000/nap | Mindenki! |
-| | GitHub másodpilóta | 10-19 USD/hó | Havi | GitHub felhasználók |
-| **🔑 API KEY** | NVIDIA NIM | **INGYENES**(végre fejlesztő) | ~40 RPM | 70+ nyitott modell |
-| | Cerebrák | **INGYENES**(1 millió tok/nap) | 60K TPM / 30 RPM | A világ leggyorsabb |
-| | Groq | **INGYENES**(30 RPM) | 14,4K RPD | Ultragyors Llama/Gemma |
-| | DeepSeek V3.2 | 0,27 USD/1,10 USD/1 millió | Nincs | Legjobb ár/minőség érvelés |
-| | xAI Grok-4 Fast | **0,20 USD/0,50 USD/1 millió**🆕 | Nincs | Leggyorsabb + szerszámhívás, ultralow |
-| | xAI Grok-4 (standard) | 0,20 USD/1,50 USD/1M 🆕 | Nincs | Oktatás zászlóshajója az xAI-tól |
-| | Mistral | Ingyenes próbaverzió + fizetett | Ár korlátozott | Európai AI |
-| | OpenRouter | Felhasználásonkénti fizetés | Nincs | 100+ modell aggr. |
-| **💰 OLCSÓ** | GLM-5 (a Z.AI-n keresztül) 🆕 | 0,5 USD/1M | Naponta 10:00 | 128K teljesítmény, legújabb zászlóshajó |
-| | GLM-4.7 | 0,6 USD/1M | Naponta 10:00 | Költségvetési biztonsági mentés |
-| | MiniMax M2.5 🆕 | 0,3 USD/1 millió bemenet | 5 órás gurulás | Érvelés + ügynöki feladatok |
-| | MiniMax M2.1 | 0,2 USD/1M | 5 órás gurulás | Legolcsóbb lehetőség |
-| | Kimi K2.5 (Moonshot API) 🆕 | Felhasználásonkénti fizetés | Nincs | Közvetlen Moonshot API hozzáférés |
-| | Kimi K2 | 9 USD/hó lakás | 10 millió token/hó | Előrelátható költség |
-| **🆓 INGYENES** | Qoder | **0 USD** | Korlátlan | 5 modell korlátlan |
-| | Qwen | **0 USD** | Korlátlan | 4 modell korlátlan |
-| | Kiro | **0 USD** | Korlátlan | Claude Sonnet/Haiku (AWS Builder) |
-| | LongCat Flash-Lite 🆕 | **0 USD**(50 millió tok/nap 🔥) | 1 RPS | A legnagyobb ingyenes kvóta a Földön |
-| | Beporzások AI 🆕 | **0 USD**(nincs szükség kulcsra) | 1 rekv/15mp | GPT-5, Claude, DeepSeek, Llama 4 |
-| | Cloudflare Workers AI 🆕 | **0 USD**(10 000 neuron/nap) | ~150 ill./nap | 50+ modell, globális élvonal |
-| | Scaleway AI 🆕 | **0 USD**(összesen 1 millió token) | Ár korlátozott | EU/GDPR, Qwen3 235B, Llama 70B | > 🆕**Új modellek hozzáadva (2026. március):**Grok-4 Fast család 0,20 USD/0,50 USD/M áron (1143 ms-os benchmark – 30%-kal gyorsabb, mint a Gemini 2.5 Flash), GLM-5 Z.AI-n keresztül 128K kimenettel, MiniMax M2.5 Vc3-on keresztül, KiepSeedk 2.5-ös okfejtéssel. Moonshot közvetlen API. |
+| Tier | Provider | Cost | Quota Reset | Best For |
+| ------------------- | --------------------------- | ------------------------- | ---------------- | --------------------------------- |
+| **💳 SUBSCRIPTION** | Claude Code (Pro) | $20/mo | 5h + weekly | Already subscribed |
+| | Codex (Plus/Pro) | $20-200/mo | 5h + weekly | OpenAI users |
+| | Gemini CLI | **FREE** | 180K/mo + 1K/day | Everyone! |
+| | GitHub Copilot | $10-19/mo | Monthly | GitHub users |
+| **🔑 API KEY** | NVIDIA NIM | **FREE** (dev forever) | ~40 RPM | 70+ open models |
+| | Cerebras | **FREE** (1M tok/day) | 60K TPM / 30 RPM | World's fastest |
+| | Groq | **FREE** (30 RPM) | 14.4K RPD | Ultra-fast Llama/Gemma |
+| | DeepSeek V3.2 | $0.27/$1.10 per 1M | None | Best price/quality reasoning |
+| | xAI Grok-4 Fast | **$0.20/$0.50 per 1M** 🆕 | None | Fastest + tool calling, ultralow |
+| | xAI Grok-4 (standard) | $0.20/$1.50 per 1M 🆕 | None | Reasoning flagship from xAI |
+| | Mistral | Free trial + paid | Rate limited | European AI |
+| | OpenRouter | Pay-per-use | None | 100+ models aggr. |
+| **💰 CHEAP** | GLM-5 (via Z.AI) 🆕 | $0.5/1M | Daily 10AM | 128K output, newest flagship |
+| | GLM-4.7 | $0.6/1M | Daily 10AM | Budget backup |
+| | MiniMax M2.5 🆕 | $0.3/1M input | 5-hour rolling | Reasoning + agentic tasks |
+| | MiniMax M2.1 | $0.2/1M | 5-hour rolling | Cheapest option |
+| | Kimi K2.5 (Moonshot API) 🆕 | Pay-per-use | None | Direct Moonshot API access |
+| | Kimi K2 | $9/mo flat | 10M tokens/mo | Predictable cost |
+| **🆓 FREE** | Qoder | **$0** | Unlimited | 5 models unlimited |
+| | Qwen | **$0** | Unlimited | 4 models unlimited |
+| | Kiro | **$0** | Unlimited | Claude Sonnet/Haiku (AWS Builder) |
+| | LongCat Flash-Lite 🆕 | **$0** (50M tok/day 🔥) | 1 RPS | Largest free quota on Earth |
+| | Pollinations AI 🆕 | **$0** (no key needed) | 1 req/15s | GPT-5, Claude, DeepSeek, Llama 4 |
+| | Cloudflare Workers AI 🆕 | **$0** (10K Neurons/day) | ~150 resp/day | 50+ models, global edge |
+| | Scaleway AI 🆕 | **$0** (1M tokens total) | Rate limited | EU/GDPR, Qwen3 235B, Llama 70B |
-**💡 0 dolláros kombinált halom – a teljes ingyenes beállítás:**```
+> 🆕 **New models added (Mar 2026):** Grok-4 Fast family at $0.20/$0.50/M (benchmarked at 1143ms — 30% faster than Gemini 2.5 Flash), GLM-5 via Z.AI with 128K output, MiniMax M2.5 reasoning, DeepSeek V3.2 updated pricing, Kimi K2.5 via Moonshot direct API.
+**💡 $0 Combo Stack — The Complete Free Setup:**
+
+```
# 🆓 Ultimate Free Stack 2026 — 11 Providers, $0 Forever
+Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
+Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
+Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
+Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
+NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+```
-Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
-Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
-LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
-Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
-Qwen (qw/) → qwen3-coder-plus, qwen3-coder-flash, qwen3-coder-next UNLIMITED
-Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free API key
-Cloudflare AI (cf/) → Llama 70B, Gemma 3, Mistral — 10K Neurons/day
-Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
-Groq (groq/) → Llama/Gemma ultra-fast — 14.4K req/day
-NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
-Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+**Zero cost. Never stops coding.** Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever.
-````
-
-**Zéró költség. Never stops coding.**Configure this as one OmniRoute combo and all fallbacks happen automatically — no manual switching ever.---
+---
---
## 🆓 Free Models — What You Actually Get
-> Az alábbi modellek**100%-ban ingyenesek, hitelkártya nélkül**. Az OmniRoute automatikus útvonalakat indít közöttük, ha egy kvóta kifogy – kombinálja őket egy feltörhetetlen 0 dolláros kombinációért.### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
+> All models below are **100% free with zero credit card required**. OmniRoute auto-routes between them when one quota runs out — combine them all for an unbreakable $0 combo.
-| Modell | Előtag | Limit | Rate Limit |
-| -------------------- | ------ | ------------- | ---------------------- |
-| `claude-szonett-4,5` | `kr/` |**Korlátlan**| Nincs bejelentett napi felső határ |
-| `claude-haiku-4,5` | `kr/` |**Korlátlan**| Nincs bejelentett napi felső határ |
-| `claude-opus-4,6` | `kr/` |**Korlátlan**| Legújabb Opus via Kiro |### 🟢 QODER MODELS (Free PAT via qodercli)
+### 🔵 CLAUDE MODELS (via Kiro — AWS Builder ID)
-| Modell | Előtag | Limit | Rate Limit |
-| ------------------- | ------ | ------------- | ---------------- |
-| `kimi-k2-gondolkodás` | "ha/" |**Korlátlan**| Nincs bejelentett felső határ |
-| "qwen3-coder-plus" | "ha/" |**Korlátlan**| Nincs bejelentett felső határ |
-| `deepseek-r1` | "ha/" |**Korlátlan**| Nincs bejelentett felső határ |
-| `minimax-m2,1` | "ha/" |**Korlátlan**| Nincs bejelentett felső határ |
-| "kimi-k2" | "ha/" |**Korlátlan**| Nincs bejelentett felső határ |
+| Model | Prefix | Limit | Rate Limit |
+| ------------------- | ------ | ------------- | --------------------- |
+| `claude-sonnet-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-haiku-4.5` | `kr/` | **Unlimited** | No reported daily cap |
+| `claude-opus-4.6` | `kr/` | **Unlimited** | Latest Opus via Kiro |
-> Javasolt csatlakozási mód:**Személyes hozzáférési token + `qodercli`**. A böngésző OAuth
-> kísérleti és alapértelmezés szerint le van tiltva, hacsak nincsenek beállítva a `QODER_OAUTH_*` környezeti változók.### 🟡 QWEN MODELS (Device Code Auth)
+### 🟢 QODER MODELS (Free PAT via qodercli)
-| Modell | Előtag | Limit | Rate Limit |
-| -------------------- | ------ | ------------- | -------------------- |
-| "qwen3-coder-plus" | `qw/` |**Korlátlan**| Nincs bejelentett felső határ |
-| `qwen3-coder-flash` | `qw/` |**Korlátlan**| Nincs bejelentett felső határ |
-| `qwen3-coder-next` | `qw/` |**Korlátlan**| Nincs bejelentett felső határ |
-| "látás-modell" | `qw/` |**Korlátlan**| Multimodális (képek) |### 🟣 GEMINI CLI (Google OAuth)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------ | ------ | ------------- | --------------- |
+| `kimi-k2-thinking` | `if/` | **Unlimited** | No reported cap |
+| `qwen3-coder-plus` | `if/` | **Unlimited** | No reported cap |
+| `deepseek-r1` | `if/` | **Unlimited** | No reported cap |
+| `minimax-m2.1` | `if/` | **Unlimited** | No reported cap |
+| `kimi-k2` | `if/` | **Unlimited** | No reported cap |
-| Modell | Előtag | Limit | Rate Limit |
-| ------------------------- | ------ | ---------------------------- | ------------- |
-| `gemini-3-flash-preview` | `gc/` |**180 000 tok/hó**+ 1 000/nap | Havi visszaállítás |
-| "gemini-2.5-pro" | `gc/` | 180 000/hó (megosztott medence) | Kiváló minőségű |### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
+> Recommended connection method: **Personal Access Token + `qodercli`**. Browser OAuth is
+> experimental and disabled by default unless `QODER_OAUTH_*` environment variables are configured.
-| Tier | Napi limit | Rate Limit | Megjegyzések |
-| ---------- | ------------ | ----------- | ------------------------------------------------------- |
-| Ingyenes (fejlesztő) | Nincs token cap |**~40 RPM**| 70+ modell; átállás a tiszta díjhatárokra 2025 közepén |
+### 🟡 QWEN MODELS (Device Code Auth)
-Népszerű ingyenes modellek: "moonshotai/kimi-k2.5" (Kimi K2.5), "z-ai/glm4.7" (GLM 4.7), "deepseek-ai/deepseek-v3.2" (DeepSeek V3.2), "nvidia/llama-3.3-70b-deepseek"### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------- | ------ | ------------- | ------------------- |
+| `qwen3-coder-plus` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-flash` | `qw/` | **Unlimited** | No reported cap |
+| `qwen3-coder-next` | `qw/` | **Unlimited** | No reported cap |
+| `vision-model` | `qw/` | **Unlimited** | Multimodal (images) |
-| Tier | Napi limit | Rate Limit | Megjegyzések |
-| ---- | ------------------ | ----------------- | -------------------------------------------- |
-| Ingyenes |**1 millió token/nap**| 60K TPM / 30 RPM | A világ leggyorsabb LLM-következtetése; naponta visszaállítja |
+### 🟣 GEMINI CLI (Google OAuth)
-Ingyenesen elérhető: "llama-3.3-70b", "llama-3.1-8b", "deepseek-r1-distill-llama-70b"### 🔴 GROQ (Free API Key — console.groq.com)
+| Model | Prefix | Limit | Rate Limit |
+| ------------------------ | ------ | --------------------------- | ------------- |
+| `gemini-3-flash-preview` | `gc/` | **180K tok/month** + 1K/day | Monthly reset |
+| `gemini-2.5-pro` | `gc/` | 180K/month (shared pool) | High quality |
-| Tier | Napi limit | Rate Limit | Megjegyzések |
-| ---- | ------------- | ----------------- | ------------------------------------------ |
-| Ingyenes |**14,4K RPD**| 30 ford./perc modellenként | Nincs hitelkártya; 429 limiten, nem terhelik |
+### ⚫ NVIDIA NIM (Free API Key — build.nvidia.com)
-Ingyenesen elérhető: "láma-3.3-70b-veratile", "gemma2-9b-it", "mixtral-8x7b", "whisper-large-v3"### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---------- | ------------ | ----------- | ------------------------------------------------------ |
+| Free (Dev) | No token cap | **~40 RPM** | 70+ models; transitioning to pure rate limits mid-2025 |
-| Modell | Előtag | Napi ingyenes kvóta | Megjegyzések |
-| ------------------------------ | ------ | ------------------ | ------------------------ |
-| "LongCat-Flash-Lite" | "lc/" |**50 millió token**💥 | A valaha volt legnagyobb ingyenes kvóta |
-| `LongCat-Flash-Chat` | "lc/" | 500 000 token | Többfordulós csevegés |
-| "LongCat-Flash-Thinking" | "lc/" | 500 000 token | Érvelés / CoT |
-| `LongCat-Flash-Thinking-2601` | "lc/" | 500 000 token | 2026. januári verzió |
-| `LongCat-Flash-Omni-2603` | "lc/" | 500 000 token | Multimodális |
+Popular free models: `moonshotai/kimi-k2.5` (Kimi K2.5), `z-ai/glm4.7` (GLM 4.7), `deepseek-ai/deepseek-v3.2` (DeepSeek V3.2), `nvidia/llama-3.3-70b-instruct`, `deepseek/deepseek-r1`
-> 100%-ban ingyenes nyilvános bétaverzióban. Regisztráljon a [longcat.chat](https://longcat.chat) oldalon e-mailben vagy telefonon. Napi alaphelyzetbe állítás 00:00 UTC.### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+### ⚪ CEREBRAS (Free API Key — inference.cerebras.ai)
-| Modell | Előtag | Rate Limit | Szolgáltató mögött |
-| ---------- | ------ | ---------- | ------------------- |
-| "openai" | `pol/` | 1 rekv/15mp | GPT-5 |
-| `claude` | `pol/` | 1 rekv/15mp | Antropikus Claude |
-| "gemini" | `pol/` | 1 rekv/15mp | Google Gemini |
-| `mélyre törekszik` | `pol/` | 1 rekv/15mp | DeepSeek V3 |
-| `láma` | `pol/` | 1 rekv/15mp | Meta Llama 4 Scout |
-| "mistral" | `pol/` | 1 rekv/15mp | Mistral AI |
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---- | ----------------- | ---------------- | ------------------------------------------- |
+| Free | **1M tokens/day** | 60K TPM / 30 RPM | World's fastest LLM inference; resets daily |
-> ✨**Zéró súrlódás:**Nincs regisztráció, nincs API-kulcs. Adja hozzá a Pollinations szolgáltatót egy üres kulcsmezővel, és azonnal működik.### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+Available free: `llama-3.3-70b`, `llama-3.1-8b`, `deepseek-r1-distill-llama-70b`
-| Tier | Napi neuronok | Egyenértékű használat | Megjegyzések |
-| ---- | ------------- | ---------------------------------------- | ------------------------ |
-| Ingyenes |**10 000**| ~150 LLM ill / 500s hang / 15K beágyazás | Globális élvonal, 50+ modell |
+### 🔴 GROQ (Free API Key — console.groq.com)
-Népszerű ingyenes modellek: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (ingyenes hang!), `@cf/qwen/qwen2.5-coder-15b-coder-
+| Tier | Daily Limit | Rate Limit | Notes |
+| ---- | ------------- | ---------------- | ----------------------------------------- |
+| Free | **14.4K RPD** | 30 RPM per model | No credit card; 429 on limit, not charged |
-> API-token + fiókazonosító szükséges a [dash.cloudflare.com] webhelyről (https://dash.cloudflare.com). Tárolja fiókazonosítóját a szolgáltató beállításaiban.### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+Available free: `llama-3.3-70b-versatile`, `gemma2-9b-it`, `mixtral-8x7b`, `whisper-large-v3`
-| Tier | Ingyenes kvóta | Helyszín | Megjegyzések |
-| ---- | ------------- | ------------ | ------------------------------------ |
-| Ingyenes |**1M token**| 🇫🇷 Párizs, EU | Nincs szükség hitelkártyára a korlátokon belül |
+### 🔴 LONGCAT AI (Free API Key — longcat.chat) 🆕
-Ingyenesen elérhető: "qwen3-235b-a22b-instruct-2507" (Qwen3 235B!), "llama-3.1-70b-instruct", "mistral-small-3.2-24b-instruct-2506", "deepseek-v3-032"
+| Model | Prefix | Daily Free Quota | Notes |
+| ----------------------------- | ------ | ----------------- | ----------------------- |
+| `LongCat-Flash-Lite` | `lc/` | **50M tokens** 💥 | Largest free quota ever |
+| `LongCat-Flash-Chat` | `lc/` | 500K tokens | Multi-turn chat |
+| `LongCat-Flash-Thinking` | `lc/` | 500K tokens | Reasoning / CoT |
+| `LongCat-Flash-Thinking-2601` | `lc/` | 500K tokens | Jan 2026 version |
+| `LongCat-Flash-Omni-2603` | `lc/` | 500K tokens | Multimodal |
-> EU/GDPR-kompatibilis. Szerezze be az API-kulcsot a [console.scaleway.com](https://console.scaleway.com) címen.
+> 100% free while in public beta. Sign up at [longcat.chat](https://longcat.chat) with email or phone. Resets daily 00:00 UTC.
->**💡 The Ultimate Free Stack (11 szolgáltató, 0 USD örökké):**
+### 🟢 POLLINATIONS AI (No API Key Required) 🆕
+
+| Model | Prefix | Rate Limit | Provider Behind |
+| ---------- | ------ | ---------- | ------------------ |
+| `openai` | `pol/` | 1 req/15s | GPT-5 |
+| `claude` | `pol/` | 1 req/15s | Anthropic Claude |
+| `gemini` | `pol/` | 1 req/15s | Google Gemini |
+| `deepseek` | `pol/` | 1 req/15s | DeepSeek V3 |
+| `llama` | `pol/` | 1 req/15s | Meta Llama 4 Scout |
+| `mistral` | `pol/` | 1 req/15s | Mistral AI |
+
+> ✨ **Zero friction:** No signup, no API key. Add the Pollinations provider with an empty key field and it works immediately.
+
+### 🟠 CLOUDFLARE WORKERS AI (Free API Key — cloudflare.com) 🆕
+
+| Tier | Daily Neurons | Equivalent Usage | Notes |
+| ---- | ------------- | --------------------------------------- | ----------------------- |
+| Free | **10,000** | ~150 LLM resp / 500s audio / 15K embeds | Global edge, 50+ models |
+
+Popular free models: `@cf/meta/llama-3.3-70b-instruct`, `@cf/google/gemma-3-12b-it`, `@cf/openai/whisper-large-v3-turbo` (free audio!), `@cf/qwen/qwen2.5-coder-15b-instruct`
+
+> Requires API Token + Account ID from [dash.cloudflare.com](https://dash.cloudflare.com). Store Account ID in provider settings.
+
+### 🟣 SCALEWAY AI (1M Free Tokens — scaleway.com) 🆕
+
+| Tier | Free Quota | Location | Notes |
+| ---- | ------------- | ------------ | ----------------------------------- |
+| Free | **1M tokens** | 🇫🇷 Paris, EU | No credit card needed within limits |
+
+Available free: `qwen3-235b-a22b-instruct-2507` (Qwen3 235B!), `llama-3.1-70b-instruct`, `mistral-small-3.2-24b-instruct-2506`, `deepseek-v3-0324`
+
+> EU/GDPR compliant. Get API key at [console.scaleway.com](https://console.scaleway.com).
+
+> **💡 The Ultimate Free Stack (11 Providers, $0 Forever):**
>
> ```
-> Kiro (kr/) → Claude Sonnet/Haiku KORLÁTALAN
-> Qoder (if/) → kimi-k2-gondolkodás, qwen3-coder-plus, deepseek-r1 UNLIMITED
-> LongCat Lite (lc/) → LongCat-Flash-Lite – 50 millió token/nap 🔥
-> Beporzások (pol/) → GPT-5, Claude, DeepSeek, Llama 4 – nincs szükség kulcsra
-> Qwen (qw/) → qwen3 kódoló modellek KORLÁTALAN
-> Gemini (gemini/) → Gemini 2.5 Flash – 1500 rekv/nap ingyenes
-> Cloudflare AI (vö./) → 50+ modell – 10 000 neuron/nap
-> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1 millió ingyenes token (EU)
-> Groq (groq/) → Llama/Gemma – 14,4 ezer rekv/nap ultragyors
-> NVIDIA NIM (nvidia/) → 70+ nyitott modell – 40 RPM örökké
-> Cerebrák (cerebras/) → Llama/Qwen a világ leggyorsabb – 1 millió tok/nap
-> ```## 🎙️ Free Transcription Combo
+> Kiro (kr/) → Claude Sonnet/Haiku UNLIMITED
+> Qoder (if/) → kimi-k2-thinking, qwen3-coder-plus, deepseek-r1 UNLIMITED
+> LongCat Lite (lc/) → LongCat-Flash-Lite — 50M tokens/day 🔥
+> Pollinations (pol/) → GPT-5, Claude, DeepSeek, Llama 4 — no key needed
+> Qwen (qw/) → qwen3-coder models UNLIMITED
+> Gemini (gemini/) → Gemini 2.5 Flash — 1,500 req/day free
+> Cloudflare AI (cf/) → 50+ models — 10K Neurons/day
+> Scaleway (scw/) → Qwen3 235B, Llama 70B — 1M free tokens (EU)
+> Groq (groq/) → Llama/Gemma — 14.4K req/day ultra-fast
+> NVIDIA NIM (nvidia/) → 70+ open models — 40 RPM forever
+> Cerebras (cerebras/) → Llama/Qwen world-fastest — 1M tok/day
+> ```
-> Bármilyen hang/videó átírása**0 USD-ért**– Deepgram vezet 200 USD ingyenes, AssemblyAI 50 USD tartalék, Groq Whisper korlátlan vészhelyzeti tartalékként.
+## 🎙️ Free Transcription Combo
-| Szolgáltató | Ingyenes kreditek | Legjobb modell | Rate Limit |
-| ------------------ | ----------------------- | --------------------------------------------- | ----------------------------- |
-| 🟢**Deepgram**|**200 USD ingyenes**(regisztráció) | `nova-3` — a legjobb pontosság, több mint 30 nyelv | Nincs RPM-korlát az ingyenes krediteknél |
-| 🔵**AssemblyAI**|**50 USD ingyenes**(regisztráció) | "univerzális-3-pro" – fejezetek, hangulat, személyazonosításra alkalmas adatok | Nincs RPM-korlát az ingyenes krediteknél |
-| 🔴**Groq**|**Örökre ingyenes**| `whisper-large-v3` — OpenAI Whisper | 30 RPM (korlátozott sebesség) |
+> Transcribe any audio/video for **$0** — Deepgram leads with $200 free, AssemblyAI $50 fallback, Groq Whisper as unlimited emergency backup.
-**Javasolt kombináció a `/dashboard/combos'-ban:**```
+| Provider | Free Credits | Best Model | Rate Limit |
+| ----------------- | ---------------------- | -------------------------------------------- | ---------------------------- |
+| 🟢 **Deepgram** | **$200 free** (signup) | `nova-3` — best accuracy, 30+ languages | No RPM limit on free credits |
+| 🔵 **AssemblyAI** | **$50 free** (signup) | `universal-3-pro` — chapters, sentiment, PII | No RPM limit on free credits |
+| 🔴 **Groq** | **Free forever** | `whisper-large-v3` — OpenAI Whisper | 30 RPM (rate limited) |
+
+**Suggested combo in `/dashboard/combos`:**
+
+```
Name: free-transcription
Strategy: Priority
Nodes:
[1] deepgram/nova-3 → uses $200 free first
[2] assemblyai/universal-3-pro → fallback when Deepgram credits run out
[3] groq/whisper-large-v3 → free forever, emergency fallback
-````
+```
-Ezután a `/dashboard/media` →**Átírás**lapon: töltsön fel bármilyen audio- vagy videofájlt → válassza ki a kombinált végpontot → kérje le az átírást a támogatott formátumokban.## 💡 Key Features
+Then in `/dashboard/media` → **Transcription** tab: upload any audio or video file → select your combo endpoint → get transcription in supported formats.
-Az OmniRoute v2.0 működési platformként készült, nem csak közvetítő proxyként.### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
+## 💡 Key Features
-| Funkció | Mit csinál |
-| --------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------ |
-| ⚡**Grok-4 Fast Family** | xAI modellek 0,20 USD/0,50 USD/M áron – 1143 ms benchmark (30%-kal gyorsabb, mint a Gemini 2.5 Flash) |
-| 🧠**GLM-5 a Z.AI-n keresztül** | 128 000 kimeneti kontextus, 0,5 USD/1 millió – a GLM család legújabb zászlóshajója |
-| 🔮**MiniMax M2.5** | Érvelés + ügynöki feladatok 0,30 USD/1M áron – jelentős fejlesztés az M2.1-hez képest |
-| 🎯**ToolCalling Flag modellenként** | Modellenkénti `toolCalling: igaz/hamis` a rendszerleíró adatbázisban – Az AutoCombo kihagyja az eszközzel nem rendelkező modelleket |
-| 🌍**Többnyelvű szándékfelismerés** | PT/ZH/ES/AR kulcsszavak az AutoCombo pontozásban – jobb modellválasztás nem angol nyelvű tartalomhoz |
-| 📊**Benchmark-vezérelt tartalékok** | Valódi p95 késés az élő kérések hírcsatornáiból, kombinált pontozásból – Az AutoCombo tanul a tényleges adatokból |
-| 🔁**Duplikáció visszavonásának kérése** | Tartalom-kivonat alapú dedup ablak – többügynök biztonságos, megakadályozza az ismétlődő terheléseket |
-| 🔌**Pluggable RouterStrategy** | Bővíthető "RouterStrategy" interfész – egyéni útválasztási logika hozzáadása pluginként | ### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP |
+OmniRoute v3.5 is built as an operational platform, not just a relay proxy.
-| Funkció | Mit csinál |
-| ------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |
-| 🎮**Játszótér modell** | Irányítópult oldal bármely modell közvetlen teszteléséhez – szolgáltató/modell/végpont választó, Monaco Editor, adatfolyam, megszakítás, időzítés |
-| 🔏**CLI ujjlenyomat-egyeztetés** | Szolgáltatónkénti fejléc/törzs rendezés a natív CLI-aláírásoknak megfelelően – váltson szolgáltatónként a Beállítások > Biztonság menüpontban.**A proxy IP-címe megmarad** |
-| 🤝**ACP-támogatás (Agent Client Protocol)** | CLI ügynök felderítés (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 további), folyamat spawner, "/api/acp/agents" végpont |
-| 🤖**ACP Agents Dashboard** | Hibakeresés › Ügynökök oldal – 14 ügynökből álló rács telepítési állapottal, verzióval, egyéni ügynök űrlappal bármely CLI-eszközhöz. Az**OpenCode**felhasználók egy „Opencode.json letöltése” gombot kapnak, amely automatikusan létrehoz egy használatra kész konfigurációt az összes elérhető modellhez. |
-| 🔧**Egyéni modell `apiFormat` Routing** | Egyéni modellek `apiFormat: "responses"` segítségével most már megfelelően irányítják a Responses API fordítóhoz |
-| 🏢**Codex Workspace Isolation** | E-mailenként több Codex-munkaterület – az OAuth megfelelően választja el a kapcsolatokat a munkaterület-azonosító |
-| 🔄**Elektronikus automatikus frissítés** | Az asztali alkalmazás ellenőrzi a frissítéseket + automatikus telepítés újraindításkor | ### 🤖 Agent & Protocol Operations (v2.0) |
+### 🆕 New — v3.5.5 Highlights (Apr 2026)
-| Funkció | Mit csinál |
-| ---------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- |
-| 🔧**MCP szerver (25 eszköz)** | IDE/agent eszközök 3 átvitelen keresztül: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 mag + 3 memória + 4 ügyességi eszköz |
-| 🤝**A2A szerver (JSON-RPC + SSE)** | Ügynök-ügynök feladatvégrehajtás szinkronizálási és adatfolyam-folyamokkal |
-| 🧭**Konszolidált végpontok oldal** | Lapos kezelőoldal Endpoint Proxy, MCP, A2A és API Endpoints lapokkal |
-| 🎚️**Szolgáltatás engedélyezése/letiltása kapcsolók** | BE/KI kapcsolók az MCP-hez és az A2A-hoz a beállítások fennmaradásával (alapértelmezett: KI) |
-| 🛰️**MCP Runtime Heartbeat** | Valós folyamatállapot (pid, üzemidő, szívverés kora, szállítás, hatókör mód) |
-| 📋**MCP Audit Trail** | Szűrhető auditnaplók sikerrel/sikertelenséggel és kulcshozzárendeléssel |
-| 🔐**MCP Scope Enforcement** | 10 részletes hatókörű engedély az ellenőrzött szerszám-hozzáféréshez |
-| 📡**A2A Task Lifecycle Management** | Feladatok listázása/szűrése, események/műtermékek ellenőrzése, futó feladatok megszakítása |
-| 📋**Agent Card Discovery** | `/.well-known/agent.json` az ügyfél automatikus felfedezéséhez |
-| 🧪**E2E protokoll tesztkábel** | Valódi MCP SDK + A2A kliens a "test:protocols:e2e" |
-| ⚙️**Működési vezérlők** | Váltókombó, rugalmassági profilok alkalmazása, megszakítók alaphelyzetbe állítása egyetlen vezérlőfelületről | ### 🧠 Routing & Intelligence |
+| Feature | What It Does |
+| ------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| 🔗 **Context Relay Strategy** | New combo strategy that preserves session continuity via structured handoff summaries when accounts rotate mid-conversation |
+| 🛡️ **Proxy Hardening** | Token health check, API key validation, and undici dispatcher all honor proxy config — no more bypass in restricted envs |
+| ⚠️ **Node.js 24 Login Warning** | Login page proactively detects incompatible Node.js versions and shows a clear warning banner with instructions |
+| 📎 **Gemini PDF Attachments** | PDF files attached in chat messages are now correctly routed to Gemini via `inline_data` and generic base64 detection |
+| 🔒 **CodeQL Security Hardening** | Resolved SSRF, insecure randomness, polynomial ReDoS, and incomplete URL sanitization alerts |
-| Funkció | Mit csinál |
-| --------------------------------------------- | --------------------------------------------------------------------------------------- | ----------------------- |
-| 🎯**Intelligens 4-szintű tartalék** | Automatikus útvonal: Előfizetés → API-kulcs → Olcsó → Ingyenes |
-| 📊**Valós idejű kvótakövetés** | Élő tokenszám + visszaszámlálás visszaállítása szolgáltatónként |
-| 🔄**Formátum fordítás** | OpenAI ↔ Claude ↔ Gemini ↔ Válaszok sémabiztos konverziókkal |
-| 👥**Többfiókos támogatás** | Több fiók szolgáltatónként intelligens kiválasztással |
-| 🔄**Automatikus token frissítés** | Az OAuth-tokenek automatikusan frissülnek |
-| 🎨**Egyéni kombók** | 9 kiegyensúlyozási stratégia + tartalék láncvezérlés |
-| 🌐**Wildcard Router** | `szolgáltató/*` dinamikus útválasztás |
-| 🧠**A költségvetési szabályozás átgondolása** | Áthaladási, automatikus, egyéni és adaptív érvelési korlátok |
-| 🔀**Modell álnevek** | Beépített + egyedi modell alias és migrációs biztonság |
-| ⚡**Háttérromlás** | Alacsony prioritású háttérfeladatok irányítása olcsóbb modellek felé |
-| 🧪**Feladattudatos intelligens útválasztás** | Modell automatikus kiválasztása tartalomtípus szerint (kódolás/látás/elemzés/összegzés) |
-| 🔄**A2A ügynök munkafolyamatok** | Determinisztikus FSM hangszerelő állapotfüggő többlépcsős ügynök-végrehajtáshoz |
-| 🔀**Adaptív útválasztás** | Dinamikus stratégia felülbírálása a token mennyisége és a prompt bonyolultsága alapján |
-| 🎲**Szolgáltatói sokszínűség** | Shannon entrópia pontozás kiegyenlítő automatikus kombinált forgalomelosztás |
-| 💬**Rendszer azonnali befecskendezés** | Következetesen alkalmazott globális viselkedésszabályozás |
-| 📄**Responses API-kompatibilitás** | Full `/v1/responses` support for Codex and advanced agentic workflows | ### 🎵 Multi-Modal APIs |
+### 🆕 New — ClawRouter-Inspired Improvements (Mar 2026)
-| Funkció | Mit csinál |
-| -------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- |
-| 🖼️**Képgenerálás** | `/v1/images/generations' felhővel és helyi háttérprogramokkal |
-| 📐**Beágyazás** | `/v1/embeddings' a kereséshez és a RAG-folyamatokhoz |
-| 🎤**Audio átírás** | "/v1/audio/transcriptions" – 7 szolgáltató (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), automatikus nyelvérzékelés, MP4/MP3/WAV támogatás |
-| 🔊**Szövegfelolvasó** | "/v1/audio/speech" – 10 szolgáltató (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) helyes hibaüzenetekkel |
-| 🎬**Videogeneráció** | `/v1/videos/generations` (ComfyUI + SD WebUI munkafolyamatok) |
-| 🎵**Zenegeneráció** | `/v1/music/generations' (ComfyUI munkafolyamatok) |
-| 🛡️**Moderálás** | "/v1/moderations" biztonsági ellenőrzések |
-| 🔀**Átsorolás** | "/v1/rerank" a relevanciapontozáshoz |
-| 🔍**Internetes keresés**🆕 | "/v1/search" – 5 szolgáltató (Serper, Brave, Perplexity, Exa, Tavily), 6500+ ingyenes/hó, automatikus feladatátvétel, gyorsítótár | ### 🛡️ Resilience, Security & Governance |
+| Feature | What It Does |
+| ------------------------------------ | ------------------------------------------------------------------------------------------- |
+| ⚡ **Grok-4 Fast Family** | xAI models at $0.20/$0.50/M — benchmarked 1143ms (30% faster than Gemini 2.5 Flash) |
+| 🧠 **GLM-5 via Z.AI** | 128K output context, $0.5/1M — newest flagship from the GLM family |
+| 🔮 **MiniMax M2.5** | Reasoning + agentic tasks at $0.30/1M — significant upgrade from M2.1 |
+| 🎯 **toolCalling Flag per Model** | Per-model `toolCalling: true/false` in registry — AutoCombo skips non-tool-capable models |
+| 🌍 **Multilingual Intent Detection** | PT/ZH/ES/AR keywords in AutoCombo scoring — better model selection for non-English content |
+| 📊 **Benchmark-Driven Fallbacks** | Real p95 latency from live requests feeds combo scoring — AutoCombo learns from actual data |
+| 🔁 **Request Deduplication** | Content-hash based dedup window — multi-agent safe, prevents duplicate charges |
+| 🔌 **Pluggable RouterStrategy** | Extensible `RouterStrategy` interface — add custom routing logic as plugins |
-| Funkció | Mit csinál |
-| ---------------------------------------------- | ---------------------------------------------------------------------------------------------------------------- | -------------------------------- |
-| 🔌**Megszakítók** | Modellenkénti kioldás/helyreállítás küszöbérték-vezérlőkkel |
-| 🎯**Végpont-tudatos modellek** | Az egyéni modellek deklarálják a támogatott végpontokat + API formátumot |
-| 🛡️**Menydörgésellenes csorda** | Mutex + szemafor védelem az újrapróbálkozás/rate eseményeknél |
-| 🧠**Szemantikai + aláírás gyorsítótár** | Költség/késleltetés csökkentése két gyorsítótár-réteggel |
-| ⚡**Idempotencia kérése** | Megkettőzött védőablak |
-| 🔒**TLS ujjlenyomat-hamisítás** | Böngészőszerű TLS-ujjlenyomat –**csökkenti a botfelismerést és a fiók megjelölését** |
-| 🔏**CLI ujjlenyomat-egyeztetés** | Megfelel a natív CLI-kérés aláírásainak —**csökkenti a kitiltási kockázatot, miközben megőrzi a proxy IP-címét** |
-| 🌐**IP-szűrés** | Engedélyezési lista/blokkolista vezérlés a nyílt telepítésekhez |
-| 📊**Szerkeszthető díjkorlátok** | Konfigurálható globális/szolgáltatói szintű korlátozások tartósan |
-| 📉**Kecses leépülés** | Többrétegű képességek tartalékai az alapvető átjáróműveletek védelmére |
-| 📜**Config Audit Trail** | Diff-alapú változáskövetés, amely megakadályozza a működési eltolódást egyszerű visszaállításokkal |
-| ⏳**Provider Health Sync** | Proaktív jogkivonat lejárati figyelése, amely riasztásokat vált ki az engedélyezési hibák előtt |
-| 🚪**A kitiltott fiókok automatikus letiltása** | A működő megszakító automatikusan lezárja a véglegesen blokkolt tokenszámlákat |
-| 🔑**API-kulcskezelés + hatókör** | A kulcsok biztonságos kiadása/forgatása és a modell/szolgáltató vezérlői |
-| 👁️**Hatáskörű API-kulcs felfedése**🆕 | Az API-kulcsok visszaállításának engedélyezése a következőn keresztül: `ALLOW_API_KEY_REVEAL` |
-| 🛡️**Védett `/modellek`** | Opcionális hitelesítési kapu és szolgáltatói elrejtés a modellkatalógushoz | ### 📊 Observability & Analytics |
+### 🚀 Previous v2.0.9+ — Playground, CLI Fingerprints & ACP
-| Funkció | Mit csinál |
-| ----------------------------------- | ---------------------------------------------------------------------------- | ---------------------------- |
-| 📝**Kérés + proxynaplózás** | Teljes kérés/válasz és proxynaplózás |
-| 📉**Folyamatos részletes naplók**🆕 | Tisztán rekonstruálja az SSE hasznos adatfolyamokat a felhasználói felületbe |
-| 📋**Unified Logs Dashboard** | Kérelem, proxy, audit és konzolnézet egy oldalon |
-| 🔍**Telemetria kérése** | p50/p95/p99 késleltetés és nyomkövetési kérelem |
-| 🏥**Egészségügyi irányítópult** | Üzemidő, megszakítási állapotok, zárolások, gyorsítótár statisztika |
-| 💰**Költségkövetés** | Költségvetési vezérlők és modellenkénti árképzés láthatósága |
-| 📈**Analytics vizualizációk** | Modell/szolgáltató használati betekintések és trendnézetek |
-| 🧪**Értékelési keret** | Arany készlet tesztelése konfigurálható meccsstratégiákkal |
-| 📡**Élő diagnosztika**🆕 | Szemantikus gyorsítótár bypass a pontos kombinált élő teszteléshez | ### ☁️ Deployment & Platform |
+| Feature | What It Does |
+| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🎮 **Model Playground** | Dashboard page to test any model directly — provider/model/endpoint selectors, Monaco Editor, streaming, abort, timing |
+| 🔏 **CLI Fingerprint Matching** | Per-provider header/body ordering to match native CLI signatures — toggle per provider in Settings > Security. **Your proxy IP is preserved** |
+| 🤝 **ACP Support (Agent Client Protocol)** | CLI agent discovery (Codex, Claude, Goose, Gemini CLI, OpenClaw + 9 more), process spawner, `/api/acp/agents` endpoint |
+| 🤖 **ACP Agents Dashboard** | Debug › Agents page — grid of 14 agents with install status, version, custom agent form for any CLI tool. **OpenCode** users get a "Download opencode.json" button that auto-generates a ready-to-use config with all available models. |
+| 🔧 **Custom Model `apiFormat` Routing** | Custom models with `apiFormat: "responses"` now correctly route to the Responses API translator |
+| 🏢 **Codex Workspace Isolation** | Multiple Codex workspaces per email — OAuth correctly separates connections by workspace ID |
+| 🔄 **Electron Auto-Update** | Desktop app checks for updates + auto-install on restart |
-| Funkció | Mit csinál |
-| ------------------------------------- | --------------------------------------------------------------------------------------- | --------------------- |
-| 🌐**Deploy Anywhere** | Localhost, VPS, Docker, Cloud környezetek |
-| 🚇**Cloudflare Tunnel**🆕 | Egykattintásos Quick Tunnel integráció az irányítópultról |
-| 🔑**API kulcsmodell szűrése** | Natív /v1/models válasz a hozzárendelt hordozókörnyezeti szerepkörökön keresztül szűrve |
-| ⚡**Smart Cache Bypass** | Konfigurálható TTL-heurisztika és kényszerített visszatöltési vezérlők |
-| 🔄**Biztonsági mentés/visszaállítás** | Export/import és katasztrófa utáni helyreállítási folyamatok |
-| 🧙**Bevezető varázsló** | Első futtatás irányított beállítás |
-| 🔧**CLI Tools Dashboard** | Egykattintásos beállítás a népszerű kódolóeszközökhöz |
-| 🎮**Játszótér modell** | Teszteljen bármely szolgáltatót/modellt/végpontot az irányítópultról |
-| 🔏**CLI ujjlenyomat kapcsoló** | Szolgáltatónkénti ujjlenyomat-egyeztetés a Beállítások > Biztonság |
-| 🌐**i18n (30 nyelv)** | Teljes irányítópult + dokumentumok nyelvi támogatása RTL lefedettséggel |
-| 🧹**Minden modell törlése** | Egykattintásos modelllista törlése a szolgáltató adatai között |
-| 👁️**Sidebar Controls**🆕 | Összetevők és integrációk elrejtése a Megjelenés beállításaiból |
-| 📋**Kiadássablonok** | Szabványos GitHub-sablonok hibákhoz és szolgáltatásokhoz |
-| 📂**Egyéni adattár** | `DATA_DIR` felülírása a tárolási helyhez | ### Feature Deep Dive |
+### 🤖 Agent & Protocol Operations (v2.0)
+
+| Feature | What It Does |
+| ------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
+| 🔧 **MCP Server (25 tools)** | IDE/agent tools via 3 transports: stdio, SSE (`/api/mcp/sse`), Streamable HTTP (`/api/mcp/stream`). 18 core + 3 memory + 4 skill tools |
+| 🤝 **A2A Server (JSON-RPC + SSE)** | Agent-to-agent task execution with sync and streaming flows |
+| 🧭 **Consolidated Endpoints Page** | Tabbed management page with Endpoint Proxy, MCP, A2A, and API Endpoints tabs |
+| 🎚️ **Service Enable/Disable Toggles** | ON/OFF switches for MCP and A2A with settings persistence (default: OFF) |
+| 🛰️ **MCP Runtime Heartbeat** | Real process status (pid, uptime, heartbeat age, transport, scope mode) |
+| 📋 **MCP Audit Trail** | Filterable audit logs with success/failure and key attribution |
+| 🔐 **MCP Scope Enforcement** | 10 granular scope permissions for controlled tool access |
+| 📡 **A2A Task Lifecycle Management** | List/filter tasks, inspect events/artifacts, cancel running tasks |
+| 📋 **Agent Card Discovery** | `/.well-known/agent.json` for client auto-discovery |
+| 🧪 **Protocol E2E Test Harness** | Real MCP SDK + A2A client flows in `test:protocols:e2e` |
+| ⚙️ **Operational Controls** | Switch combo, apply resilience profiles, reset breakers from one control surface |
+
+### 🧠 Routing & Intelligence
+
+| Feature | What It Does |
+| ---------------------------------- | ------------------------------------------------------------------------ |
+| 🎯 **Smart 4-Tier Fallback** | Auto-route: Subscription → API Key → Cheap → Free |
+| 📊 **Real-Time Quota Tracking** | Live token count + reset countdown per provider |
+| 🔄 **Format Translation** | OpenAI ↔ Claude ↔ Gemini ↔ Responses with schema-safe conversions |
+| 👥 **Multi-Account Support** | Multiple accounts per provider with intelligent selection |
+| 🔄 **Auto Token Refresh** | OAuth tokens refresh automatically with retry |
+| 🎨 **Custom Combos** | 13 balancing strategies + fallback chain control |
+| 🔗 **Context Relay** | Session continuity handoffs when account rotation happens mid-session |
+| 🌐 **Wildcard Router** | `provider/*` dynamic routing |
+| 🧠 **Thinking Budget Controls** | Passthrough, auto, custom, and adaptive reasoning limits |
+| 🔀 **Model Aliases** | Built-in + custom model aliasing and migration safety |
+| ⚡ **Background Degradation** | Route low-priority background tasks to cheaper models |
+| 🧪 **Task-Aware Smart Routing** | Auto-select model by content type (coding/vision/analysis/summarization) |
+| 🔄 **A2A Agent Workflows** | Deterministic FSM orchestrator for stateful multi-step agent executions |
+| 🔀 **Adaptive Routing** | Dynamic strategy override based on token volume and prompt complexity |
+| 🎲 **Provider Diversity** | Shannon entropy scoring balancing auto-combo traffic distribution |
+| 💬 **System Prompt Injection** | Global behavior controls applied consistently |
+| 📄 **Responses API Compatibility** | Full `/v1/responses` support for Codex and advanced agentic workflows |
+
+### 🎵 Multi-Modal APIs
+
+| Feature | What It Does |
+| -------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| 🖼️ **Image Generation** | `/v1/images/generations` with cloud and local backends |
+| 📐 **Embeddings** | `/v1/embeddings` for search and RAG pipelines |
+| 🎤 **Audio Transcription** | `/v1/audio/transcriptions` — 7 providers (Deepgram Nova 3, AssemblyAI, Groq Whisper, HuggingFace, ElevenLabs, OpenAI, Azure), auto-language detection, MP4/MP3/WAV support |
+| 🔊 **Text-to-Speech** | `/v1/audio/speech` — 10 providers (ElevenLabs, OpenAI, Deepgram, Cartesia, PlayHT, HuggingFace, Nvidia NIM, Inworld, Coqui, Tortoise) with correct error messages |
+| 🎬 **Video Generation** | `/v1/videos/generations` (ComfyUI + SD WebUI workflows) |
+| 🎵 **Music Generation** | `/v1/music/generations` (ComfyUI workflows) |
+| 🛡️ **Moderations** | `/v1/moderations` safety checks |
+| 🔀 **Reranking** | `/v1/rerank` for relevance scoring |
+| 🔍 **Web Search** 🆕 | `/v1/search` — 5 providers (Serper, Brave, Perplexity, Exa, Tavily), 6,500+ free/month, auto-failover, cache |
+
+### 🛡️ Resilience, Security & Governance
+
+| Feature | What It Does |
+| ----------------------------------- | -------------------------------------------------------------------------------------- |
+| 🔌 **Circuit Breakers** | Per-model trip/recover with threshold controls |
+| 🎯 **Endpoint-Aware Models** | Custom models declare supported endpoints + API format |
+| 🛡️ **Anti-Thundering Herd** | Mutex + semaphore protections on retry/rate events |
+| 🧠 **Semantic + Signature Cache** | Cost/latency reduction with two cache layers |
+| ⚡ **Request Idempotency** | Duplicate protection window |
+| 🔒 **TLS Fingerprint Spoofing** | Browser-like TLS fingerprint — **reduces bot detection and account flagging** |
+| 🔏 **CLI Fingerprint Matching** | Matches native CLI request signatures — **reduces ban risk while preserving proxy IP** |
+| 🌐 **IP Filtering** | Allowlist/blocklist control for exposed deployments |
+| 📊 **Editable Rate Limits** | Configurable global/provider-level limits with persistence |
+| 📉 **Graceful Degradation** | Multi-layer capability fallbacks protecting core gateway operations |
+| 📜 **Config Audit Trail** | Diff-based change tracking preventing operational drift with simple rollbacks |
+| ⏳ **Provider Health Sync** | Proactive token expiration monitoring triggering alerts before authorization failures |
+| 🚪 **Auto-Disable Banned Accounts** | Operational circuit breaker sealing permanently blocked token accounts automatically |
+| 🔑 **API Key Management + Scoping** | Secure key issuance/rotation and model/provider controls |
+| 👁️ **Scoped API Key Reveal** 🆕 | Opt-in recovery of API keys via `ALLOW_API_KEY_REVEAL` |
+| 🛡️ **Protected `/models`** | Optional auth gating and provider hiding for model catalog |
+
+### 📊 Observability & Analytics
+
+| Feature | What It Does |
+| -------------------------------- | ----------------------------------------------------- |
+| 📝 **Request + Proxy Logging** | Full request/response and proxy logging |
+| 📉 **Streamed Detailed Logs** 🆕 | Reconstructs SSE payload streams cleanly into the UI |
+| 📋 **Unified Logs Dashboard** | Request, proxy, audit, and console views in one page |
+| 🔍 **Request Telemetry** | p50/p95/p99 latency and request tracing |
+| 🏥 **Health Dashboard** | Uptime, breaker states, lockouts, cache stats |
+| 💰 **Cost Tracking** | Budget controls and per-model pricing visibility |
+| 📈 **Analytics Visualizations** | Model/provider usage insights and trend views |
+| 🧪 **Evaluation Framework** | Golden set testing with configurable match strategies |
+| 📡 **Live Diagnostics** 🆕 | Semantic cache bypass for accurate combo live testing |
+
+### ☁️ Deployment & Platform
+
+| Feature | What It Does |
+| ------------------------------ | --------------------------------------------------------------------- |
+| 🌐 **Deploy Anywhere** | Localhost, VPS, Docker, Cloud environments |
+| 🚇 **Cloudflare Tunnel** 🆕 | One-click Quick Tunnel integration from the dashboard |
+| 🔑 **API Key Model Filtering** | Native /v1/models response filtered via assigned Bearer context roles |
+| ⚡ **Smart Cache Bypass** | Configurable TTL heuristics and forced refetch controls |
+| 🔄 **Backup/Restore** | Export/import and disaster recovery flows |
+| 🧙 **Onboarding Wizard** | First-run guided setup |
+| 🔧 **CLI Tools Dashboard** | One-click setup for popular coding tools |
+| 🎮 **Model Playground** | Test any provider/model/endpoint from the dashboard |
+| 🔏 **CLI Fingerprint Toggle** | Per-provider fingerprint matching in Settings > Security |
+| 🌐 **i18n (30 languages)** | Full dashboard + docs language support with RTL coverage |
+| 🧹 **Clear All Models** | One-click model list clearing in provider details |
+| 👁️ **Sidebar Controls** 🆕 | Hide components and integrations from Appearance Settings |
+| 📋 **Issue Templates** | Standardized GitHub templates for bugs and features |
+| 📂 **Custom Data Directory** | `DATA_DIR` override for storage location |
+
+### Feature Deep Dive
#### Smart fallback with practical cost control
@@ -1295,105 +1464,132 @@ Combo: "my-coding-stack"
4. if/kimi-k2-thinking
```
-Ha a kvóta, arány vagy állapot meghiúsul, az OmniRoute kézi váltás nélkül automatikusan a következő jelöltre lép.#### Protocol management that is visible and operable
+When quota, rate, or health fails, OmniRoute automatically moves to the next candidate without manual switching.
-- Az MCP + A2A felfedezhető a felhasználói felületen és a dokumentumokban (nem rejtett)
-- A protokollállapot API-k élő működési adatokat tesznek közzé (`/api/mcp/*`, `/api/a2a/*`)
-- Az irányítópultok tartalmaznak műveleteket a 2. napi műveletekhez (kombinált kapcsolók, megszakítók alaphelyzetbe állítása, feladat törlése)#### Translator + validation workflow
+#### Protocol management that is visible and operable
-A Fordító terület a következőket tartalmazza:
+- MCP + A2A are discoverable in UI and docs (not hidden)
+- Protocol status APIs expose live operational data (`/api/mcp/*`, `/api/a2a/*`)
+- Dashboards include actions for day-2 ops (combo toggles, breaker resets, task cancellation)
--**Játszótér**: kérjen átalakítási ellenőrzéseket -**Csevegés tesztelő**: teljes kérés/válasz oda-vissza út -**Tesztpad**: több eset egy menetben -**Élő monitor**: valós idejű forgalmi nézet
+#### Translator + validation workflow
-Plusz a protokoll érvényesítése valós kliensekkel az "npm run test:protocols:e2e" segítségével.
+The Translator area includes:
-> 📖**[MCP Server README](open-sse/mcp-server/README.md)**— Eszközreferencia, IDE-konfigurációk és példák kliensekre
+- **Playground**: request transformation checks
+- **Chat Tester**: full request/response round-trip
+- **Test Bench**: multiple cases in one run
+- **Live Monitor**: real-time traffic view
+
+Plus protocol validation with real clients via `npm run test:protocols:e2e`.
+
+> 📖 **[MCP Server README](open-sse/mcp-server/README.md)** — Tool reference, IDE configs, and client examples
>
-> 📖**[A2A kiszolgáló README](src/lib/a2a/README.md)**– Készségek, JSON-RPC metódusok, adatfolyamok és feladatok életciklusa## 🧪 Evaluations (Evals)
+> 📖 **[A2A Server README](src/lib/a2a/README.md)** — Skills, JSON-RPC methods, streaming, and task lifecycle
-Az OmniRoute egy beépített kiértékelő keretrendszert tartalmaz, amellyel az LLM válaszminőségét egy aranykészlettel összehasonlítva tesztelheti. Az irányítópulton az**Analytics → Evals**menüpontban érheti el.### Built-in Golden Set
+## 🧪 Evaluations (Evals)
-Az előre feltöltött "OmniRoute Golden Set" teszteseteket tartalmaz:
+OmniRoute includes a built-in evaluation framework to test LLM response quality against a golden set. Access it via **Analytics → Evals** in the dashboard.
-- Üdvözlet, matematika, földrajz, kódgenerálás
-- JSON formátum megfelelőség, fordítás, leértékelés generálása
-- Biztonsági elutasítás (káros tartalom), számlálás, logikai logika### Evaluation Strategies
+### Built-in Golden Set
-| Stratégia | Leírás | Példa |
-| ----------- | ------------------------------------------------------------------------------------------------- | --------------------------------- | --- |
-| "pontos" | A kimenetnek pontosan meg kell egyeznie | "4" |
-| `tartalmaz` | A kimenetnek tartalmaznia kell részkarakterláncot (a kis- és nagybetűk nem különböznek egymástól) | "Párizs" |
-| "regex" | A kimenetnek meg kell egyeznie a regex mintával | "1.*2.*3" |
-| "egyedi" | Az egyéni JS függvény igaz/hamis | `(kimenet) => output.length > 10` | --- |
+The pre-loaded "OmniRoute Golden Set" contains test cases for:
+
+- Greetings, math, geography, code generation
+- JSON format compliance, translation, markdown generation
+- Safety refusal (harmful content), counting, boolean logic
+
+### Evaluation Strategies
+
+| Strategy | Description | Example |
+| ---------- | ------------------------------------------------ | -------------------------------- |
+| `exact` | Output must match exactly | `"4"` |
+| `contains` | Output must contain substring (case-insensitive) | `"Paris"` |
+| `regex` | Output must match regex pattern | `"1.*2.*3"` |
+| `custom` | Custom JS function returns true/false | `(output) => output.length > 10` |
+
+---
## 📖 Setup Guide
### Protocol Setup (MCP + A2A)
-
+
+🧩 MCP Setup (Model Context Protocol)
-🧩 MCP beállítása (Model Context Protocol)
+Start MCP transport in stdio mode:
-Indítsa el az MCP-átvitelt stdio módban:```bash
+```bash
omniroute --mcp
+```
-````
+Recommended validation flow:
-Javasolt érvényesítési folyamat:
+1. Connect your MCP client over stdio.
+2. Run `omniroute_get_health`.
+3. Run `omniroute_list_combos`.
+4. Open `/dashboard/mcp` to confirm heartbeat, activity, and audit.
-1. Csatlakoztassa az MCP-klienst az stdio-n keresztül.
-2. Futtassa az „omniroute_get_health” parancsot.
-3. Futtassa az `omniroute_list_combos` parancsot.
-4. Nyissa meg a „/dashboard/mcp” mappát a szívverés, a tevékenység és az ellenőrzés megerősítéséhez.
-
-Hasznos API-k az automatizáláshoz:
+Useful APIs for automation:
- `GET /api/mcp/status`
- `GET /api/mcp/tools`
- `GET /api/mcp/audit`
-- "GET /api/mcp/audit/stats".
+- `GET /api/mcp/audit/stats`
-
-🤝 A2A beállítás (Agent2Agent)
+
-Fedezze fel az ügynököt:```bash
+
+🤝 A2A Setup (Agent2Agent)
+
+Discover the agent:
+
+```bash
curl http://localhost:20128/.well-known/agent.json
-````
+```
-Feladat küldése:```bash
+Send a task:
+
+```bash
curl -X POST http://localhost:20128/a2a \
- -H 'content-type: application/json' \
- -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+ -H 'content-type: application/json' \
+ -d '{"jsonrpc":"2.0","id":"setup-a2a","method":"message/send","params":{"skill":"quota-management","messages":[{"role":"user","content":"Summarize quota status."}]}}'
+```
-````
-
-Életciklus kezelése:
+Manage lifecycle:
- `GET /api/a2a/status`
- `GET /api/a2a/tasks`
- `GET /api/a2a/tasks/:id`
- `POST /api/a2a/tasks/:id/cancel`
-Működési UI:
+Operational UI:
-- `/dashboard/a2a` a feladat/állapot/folyam megfigyelhetőségéhez és füstműveletekhez
+- `/dashboard/a2a` for task/state/stream observability and smoke actions
-
-🧪 Végpontok közötti protokollellenőrzés
+
-Érvényesítse mindkét protokollt valódi ügyfelekkel:```bash
+
+🧪 End-to-end protocol validation
+
+Validate both protocols with real clients:
+
+```bash
npm run test:protocols:e2e
-````
+```
-Ez igazolja:
+This verifies:
-- MCP SDK kliens csatlakozás/lista/hívás
-- A2A felfedezés/küldés/stream/get/cancel
-- Az MCP audit és A2A feladatkezelő API-k adatainak keresztellenőrzése
+- MCP SDK client connect/list/call
+- A2A discovery/send/stream/get/cancel
+- Cross-check data in MCP audit and A2A task management APIs
-
+
-💳 Előfizetéses szolgáltatók### Claude Code (Pro/Max)
+
+💳 Subscription Providers
+
+### Claude Code (Pro/Max)
```bash
Dashboard → Providers → Connect Claude Code
@@ -1406,7 +1602,9 @@ Models:
cc/claude-haiku-4-5-20251001
```
-**Profi tipp:**Használja az Opust összetett feladatokhoz, a Sonnet pedig a sebességhez. Az OmniRoute nyomkövetési kvóta modellenként!### OpenAI Codex (Plus/Pro)
+**Pro Tip:** Use Opus for complex tasks, Sonnet for speed. OmniRoute tracks quota per model!
+
+### OpenAI Codex (Plus/Pro)
```bash
Dashboard → Providers → Connect Codex
@@ -1420,20 +1618,22 @@ Models:
#### Codex Account Limit Management (5h + Weekly)
-Mostantól minden Codex-fiók rendelkezik házirend-kapcsolókkal az "Irányítópult -> Szolgáltatók" részben:
+Each Codex account now has policy toggles in `Dashboard -> Providers`:
-- "5h" (BE/KI): az 5 órás ablak küszöbszabályának érvényesítése.
-- `Heti` (BE/KI): érvényesíti a heti ablak küszöbértékét.
-- Küszöbbeli viselkedés: ha egy engedélyezett ablak eléri a >=90%-os használatot, a fiók kimarad.
-- Forgatási viselkedés: Az OmniRoute automatikusan a következő jogosult Codex-fiókhoz irányít.
-- Visszaállítási viselkedés: amikor a szolgáltató „resetAt” ideje letelik, a fiók automatikusan újra jogosulttá válik.
+- `5h` (ON/OFF): enforce the 5-hour window threshold policy.
+- `Weekly` (ON/OFF): enforce the weekly window threshold policy.
+- Threshold behavior: when an enabled window reaches >=90% usage, that account is skipped.
+- Rotation behavior: OmniRoute routes to the next eligible Codex account automatically.
+- Reset behavior: when the provider `resetAt` time passes, the account becomes eligible again automatically.
-Forgatókönyvek:
+Scenarios:
-- `5h ON` + `Heti BE`: a fiók kimarad, ha valamelyik ablak eléri a küszöbértéket.
-- `5h OFF` + `Heti BE`: csak heti használat blokkolhatja a fiókot.
-- `5h BE` + `Heti KI`: csak 5 órás használat blokkolhatja a fiókot.
-- `resetAt` sikeres: a fiók automatikusan újraindul a forgatásba (nincs kézi újraengedélyezés).### Gemini CLI (FREE 180K/month!)
+- `5h ON` + `Weekly ON`: account is skipped when either window reaches threshold.
+- `5h OFF` + `Weekly ON`: only weekly usage can block the account.
+- `5h ON` + `Weekly OFF`: only 5-hour usage can block the account.
+- `resetAt` passed: account re-enters rotation automatically (no manual re-enable).
+
+### Gemini CLI (FREE 180K/month!)
```bash
Dashboard → Providers → Connect Gemini CLI
@@ -1445,7 +1645,9 @@ Models:
gc/gemini-2.5-pro
```
-**Legjobb érték:**Hatalmas ingyenes szint! Használja ezt a fizetett szintek előtt.### GitHub Copilot
+**Best Value:** Huge free tier! Use this before paid tiers.
+
+### GitHub Copilot
```bash
Dashboard → Providers → Connect GitHub
@@ -1460,74 +1662,91 @@ Models:
-
+
+🔑 API Key Providers
-🔑 API-kulcsszolgáltatók
### NVIDIA NIM (FREE developer access — 70+ models)
+### NVIDIA NIM (FREE developer access — 70+ models)
-1. Regisztráljon: [build.nvidia.com](https://build.nvidia.com)
-2. Ingyenes API-kulcs beszerzése (1000 következtetési kredit)
-3. Irányítópult → Szolgáltató hozzáadása → NVIDIA NIM:
- - API-kulcs: "nvapi-your-key".
+1. Sign up: [build.nvidia.com](https://build.nvidia.com)
+2. Get free API key (1000 inference credits included)
+3. Dashboard → Add Provider → NVIDIA NIM:
+ - API Key: `nvapi-your-key`
-**Modelek:**"nvidia/llama-3.3-70b-instruct", "nvidia/mistral-7b-instruct" és több mint 50 további
+**Models:** `nvidia/llama-3.3-70b-instruct`, `nvidia/mistral-7b-instruct`, and 50+ more
-**Profi tipp:**OpenAI-kompatibilis API – zökkenőmentesen működik az OmniRoute formátumfordításával!### DeepSeek
+**Pro Tip:** OpenAI-compatible API — works seamlessly with OmniRoute's format translation!
-1. Regisztráljon: [platform.deepseek.com](https://platform.deepseek.com)
-2. Szerezze be az API-kulcsot
-3. Irányítópult → Szolgáltató hozzáadása → DeepSeek
+### DeepSeek
-**Modellek:**"deepseek/deepseek-chat", "deepseek/deepseek-coder"### Groq (Free Tier Available!)
+1. Sign up: [platform.deepseek.com](https://platform.deepseek.com)
+2. Get API key
+3. Dashboard → Add Provider → DeepSeek
-1. Regisztráljon: [console.groq.com](https://console.groq.com)
-2. API-kulcs beszerzése (ingyenes szint tartalmazza)
-3. Irányítópult → Szolgáltató hozzáadása → Groq
+**Models:** `deepseek/deepseek-chat`, `deepseek/deepseek-coder`
-**Modellek:**"groq/llama-3.3-70b", "groq/mixtral-8x7b"
+### Groq (Free Tier Available!)
-**Profi tipp:**Ultragyors következtetés – a legjobb valós idejű kódoláshoz!### OpenRouter (100+ Models)
+1. Sign up: [console.groq.com](https://console.groq.com)
+2. Get API key (free tier included)
+3. Dashboard → Add Provider → Groq
-1. Regisztráljon: [openrouter.ai](https://openrouter.ai)
-2. Szerezze be az API-kulcsot
-3. Irányítópult → Szolgáltató hozzáadása → OpenRouter
+**Models:** `groq/llama-3.3-70b`, `groq/mixtral-8x7b`
-**Modellek:**Hozzáférés több mint 100 modellhez az összes főbb szolgáltatótól egyetlen API-kulccsal.
+**Pro Tip:** Ultra-fast inference — best for real-time coding!
-**Az irányítópult viselkedése:**Az OpenRouter modellek kezelése az**Elérhető modellek**oldalon történik. A kézi hozzáadása, importálása és automatikus szinkronizálása ugyanazt a listát frissíti.
+### OpenRouter (100+ Models)
-
+1. Sign up: [openrouter.ai](https://openrouter.ai)
+2. Get API key
+3. Dashboard → Add Provider → OpenRouter
-💰 Olcsó szolgáltatók (tartalék)### GLM-4.7 (Daily reset, $0.6/1M)
+**Models:** Access 100+ models from all major providers through a single API key.
-1. Regisztráljon: [Zhipu AI](https://open.bigmodel.cn/)
-2. Szerezze be az API-kulcsot a Coding Plan-ból
-3. Irányítópult → API-kulcs hozzáadása:
- - Szolgáltató: "glm".
- - API-kulcs: "a-kulcs".
+**Dashboard behavior:** OpenRouter models are managed from **Available Models**. Manual add, import, and auto-sync all update the same list.
-**Használja:**`glm/glm-4.7`
+
-**Profi tipp:**A kódolási terv 3-szoros kvótát kínál 1/7 költséggel! Visszaállítás naponta 10:00.### MiniMax M2.1 (5h reset, $0.20/1M)
+
+💰 Cheap Providers (Backup)
-1. Regisztráljon: [MiniMax](https://www.minimax.io/)
-2. Szerezze be az API-kulcsot
-3. Irányítópult → API-kulcs hozzáadása
+### GLM-4.7 (Daily reset, $0.6/1M)
-**Használja:**`minimax/MiniMax-M2.1`
+1. Sign up: [Zhipu AI](https://open.bigmodel.cn/)
+2. Get API key from Coding Plan
+3. Dashboard → Add API Key:
+ - Provider: `glm`
+ - API Key: `your-key`
-**Profi tipp:**A legolcsóbb lehetőség hosszú kontextushoz (1 millió token)!### Kimi K2 ($9/month flat)
+**Use:** `glm/glm-4.7`
-1. Feliratkozás: [Moonshot AI](https://platform.moonshot.ai/)
-2. Szerezze be az API-kulcsot
-3. Irányítópult → API-kulcs hozzáadása
+**Pro Tip:** Coding Plan offers 3× quota at 1/7 cost! Reset daily 10:00 AM.
-**Használd:**`kimi/kimi-latest`
+### MiniMax M2.1 (5h reset, $0.20/1M)
-**Profi tipp:**Fix 9 USD/hó 10 millió token esetén = 0,90 USD/1 millió tényleges költség!
+1. Sign up: [MiniMax](https://www.minimax.io/)
+2. Get API key
+3. Dashboard → Add API Key
-
+**Use:** `minimax/MiniMax-M2.1`
-🆓 INGYENES szolgáltatók (vészhelyzeti biztonsági mentés)### Qoder (5 FREE models via OAuth)
+**Pro Tip:** Cheapest option for long context (1M tokens)!
+
+### Kimi K2 ($9/month flat)
+
+1. Subscribe: [Moonshot AI](https://platform.moonshot.ai/)
+2. Get API key
+3. Dashboard → Add API Key
+
+**Use:** `kimi/kimi-latest`
+
+**Pro Tip:** Fixed $9/month for 10M tokens = $0.90/1M effective cost!
+
+
+
+
+🆓 FREE Providers (Emergency Backup)
+
+### Qoder (5 FREE models via OAuth)
```bash
Dashboard → Connect Qoder
@@ -1568,9 +1787,10 @@ Models:
-
+
+🎨 Create Combos
-🎨 Hozzon létre kombókat
### Example 1: Maximize Subscription → Cheap Backup
+### Example 1: Maximize Subscription → Cheap Backup
```
Dashboard → Combos → Create New
@@ -1598,9 +1818,10 @@ Cost: $0 forever!
-
+
+🔧 CLI Integration
-🔧 CLI-integráció
### Cursor IDE
+### Cursor IDE
```
Settings → Models → Advanced:
@@ -1611,7 +1832,9 @@ Settings → Models → Advanced:
### Claude Code
-Használja az irányítópult**CLI Tools**oldalát az egykattintásos konfiguráláshoz, vagy szerkessze manuálisan a `~/.claude/settings.json` fájlt.### Codex CLI
+Use the **CLI Tools** page in the dashboard for one-click configuration, or edit `~/.claude/settings.json` manually.
+
+### Codex CLI
```bash
export OPENAI_BASE_URL="http://localhost:20128"
@@ -1622,12 +1845,15 @@ codex "your prompt"
### OpenClaw
-**1. lehetőség – Irányítópult (ajánlott):**```
+**Option 1 — Dashboard (recommended):**
+
+```
Dashboard → CLI Tools → OpenClaw → Select Model → Apply
+```
-````
+**Option 2 — Manual:** Edit `~/.openclaw/openclaw.json`:
-**2. lehetőség – Kézi:**Az `~/.openclaw/openclaw.json` szerkesztése:```json
+```json
{
"models": {
"providers": {
@@ -1639,9 +1865,11 @@ Dashboard → CLI Tools → OpenClaw → Select Model → Apply
}
}
}
-````
+```
-> **Megjegyzés:**Az OpenClaw csak a helyi OmniRoute-tal működik. Az IPv6-feloldási problémák elkerülése érdekében használja a "127.0.0.1" értéket a "localhost" helyett.### Cline / Continue / RooCode
+> **Note:** OpenClaw only works with local OmniRoute. Use `127.0.0.1` instead of `localhost` to avoid IPv6 resolution issues.
+
+### Cline / Continue / RooCode
```
Settings → API Configuration:
@@ -1653,15 +1881,17 @@ Settings → API Configuration:
### OpenCode
-**1. lépés:**Az OmniRoute hozzáadása egyéni szolgáltatóként:```bash
+**Step 1:** Add OmniRoute as a custom provider:
+
+```bash
opencode
/connect
-
# Select "Other" → Enter ID: "omniroute" → Enter your OmniRoute API key
+```
-````
+**Step 2:** Create/edit `opencode.json` in your project root:
-**2. lépés:**Hozza létre/szerkesztse az `opencode.json` fájlt a projekt gyökérjében:```json
+```json
{
"$schema": "https://opencode.ai/config.json",
"provider": {
@@ -1679,117 +1909,130 @@ opencode
}
}
}
-````
+```
-**3. lépés:**Válassza ki a modellt az OpenCode-ban:```bash
+**Step 3:** Select the model in OpenCode:
+
+```bash
/models
-
# Select any OmniRoute model from the list
+```
-````
+> **Tip:** Add any model available in your OmniRoute `/v1/models` endpoint to the `models` section. Use the format `provider/model-id` from your OmniRoute dashboard.
->**Tipp:**Adjon hozzá bármely, az OmniRoute `/v1/models' végpontjában elérhető modellt a `modellek' szakaszhoz. Használja a „szolgáltató/modellazonosító” formátumot az OmniRoute irányítópultján.
+
---
## Hibaelhárítás
-
-Kattintson ide a hibaelhárítási útmutató kibontásához
+
+Click to expand troubleshooting guide
-**"A nyelvi modell nem adott üzenetet"**
+**"Language model did not provide messages"**
-- A szolgáltatói kvóta kimerült → Ellenőrizze az irányítópult kvótakövetőjét
-- Megoldás: Használjon kombinált tartalékot, vagy váltson olcsóbb szintre
+- Provider quota exhausted → Check dashboard quota tracker
+- Solution: Use combo fallback or switch to cheaper tier
-**Drátakorlát**
+**Rate limiting**
-- Előfizetési kvóta lejárt → Tartalék a GLM/MiniMax-hoz
-- Kombó hozzáadása: "cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking"
+- Subscription quota out → Fallback to GLM/MiniMax
+- Add combo: `cc/claude-opus-4-6 → glm/glm-4.7 → if/kimi-k2-thinking`
-**OAuth token lejárt**
+**OAuth token expired**
-- Az OmniRoute automatikusan frissíti
-- Ha a problémák továbbra is fennállnak: Irányítópult → Szolgáltató → Újracsatlakozás
+- Auto-refreshed by OmniRoute
+- If issues persist: Dashboard → Provider → Reconnect
-**Magas költségek**
+**High costs**
-- Ellenőrizze a használati statisztikákat az Irányítópult → Költségek menüpontban
-- Állítsa át az elsődleges modellt GLM/MiniMax-ra
-- Használjon ingyenes réteget (Gemini CLI, Qoder) a nem kritikus feladatokhoz
+- Check usage stats in Dashboard → Costs
+- Switch primary model to GLM/MiniMax
+- Use free tier (Gemini CLI, Qoder) for non-critical tasks
-**Az irányítópult/API portok hibásak**
+**Dashboard/API ports are wrong**
-- A "PORT" a kanonikus alapport (és alapértelmezés szerint API-port)
-- Az „API_PORT” csak az OpenAI-kompatibilis API figyelőt írja felül
-- A „DASHBOARD_PORT” csak az irányítópultot/Next.js figyelőt írja felül
-- Állítsa be a „NEXT_PUBLIC_BASE_URL” címet irányítópultjára/nyilvános URL-címére (OAuth-visszahívásokhoz)
+- `PORT` is the canonical base port (and API port by default)
+- `API_PORT` overrides only OpenAI-compatible API listener
+- `DASHBOARD_PORT` overrides only dashboard/Next.js listener
+- Set `NEXT_PUBLIC_BASE_URL` to your dashboard/public URL (for OAuth callbacks)
-**Felhő szinkronizálási hibák**
+**Cloud sync errors**
-- Ellenőrizze, hogy a "BASE_URL" a futó példányra mutat
-- Ellenőrizze, hogy a „CLOUD_URL” a várható felhő-végpontra mutat
-- Tartsa a `NEXT_PUBLIC_*` értékeket a szerveroldali értékekkel összhangban
+- Verify `BASE_URL` points to your running instance
+- Verify `CLOUD_URL` points to your expected cloud endpoint
+- Keep `NEXT_PUBLIC_*` values aligned with server-side values
-**Az első bejelentkezés nem működik**
+**First login not working**
-- Ellenőrizze az "INITIAL_PASSWORD" értéket a ".env" fájlban
-- Ha nincs beállítva, a tartalék jelszó `123456`
+- Check `INITIAL_PASSWORD` in `.env`
+- If unset, fallback password is `123456`
-**Nincs kérésnapló**
+**No request logs**
-- A kérés melléktermékei kérésenként egy JSON-fájlként íródnak a `DATA_DIR/call_logs/` mappába
-- Engedélyezze a folyamat rögzítését az Irányítópult → Naplók → Kérelemnaplók menüpontból, ha részletes, szakaszonkénti hasznos adatokra van szüksége
-- Állítsa be az "APP_LOG_TO_FILE=true" értéket, ha az alkalmazáskonzolnaplókat is szeretné a "logs/application/app.log" fájlban
-- Szükség szerint állítsa be a `APP_LOG_MAX_FILE_SIZE', 'APP_LOG_RETENTION_DAYS', 'APP_LOG_MAX_FILES' és 'CALL_LOG_MAX_ENTRIES'
+- Request artifacts are written to `DATA_DIR/call_logs/` as one JSON file per request
+- Enable pipeline capture from Dashboard → Logs → Request Logs if you need detailed per-stage payloads
+- Set `APP_LOG_TO_FILE=true` if you also want application console logs in `logs/application/app.log`
+- Adjust `APP_LOG_MAX_FILE_SIZE`, `APP_LOG_RETENTION_DAYS`, `APP_LOG_MAX_FILES`, and `CALL_LOG_MAX_ENTRIES` as needed
-**A csatlakozási teszt „Érvénytelen” üzenetet mutat az OpenAI-kompatibilis szolgáltatók esetében**
+**Connection test shows "Invalid" for OpenAI-compatible providers**
-- Sok szolgáltató nem tesz közzé „/models” végpontot
-- Az OmniRoute v1.0.6+ tartalmazza a tartalék érvényesítést a csevegés befejezésén keresztül
-- Győződjön meg arról, hogy az alap URL tartalmazza a „/v1” utótagot### 🔐 OAuth on a Remote Server
+- Many providers don't expose a `/models` endpoint
+- OmniRoute v1.0.6+ includes fallback validation via chat completions
+- Ensure base URL includes `/v1` suffix
+
+### 🔐 OAuth on a Remote Server
->**⚠️ Fontos az OmniRoute-ot VPS-en, Dockeren vagy bármely távoli szerveren futtató felhasználók számára**#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
+> **⚠️ Important for users running OmniRoute on a VPS, Docker, or any remote server**
-Az**Antigravity**és**Gemini CLI**szolgáltatók a**Google OAuth 2.0**szolgáltatást használják. A Google megköveteli, hogy az OAuth-folyamatban szereplő „redirect_uri” pontosan egyezzen az alkalmazás Google Cloud Console-jában előregisztrált URI-k egyikével.
+#### Why does Antigravity / Gemini CLI OAuth fail on remote servers?
-Az OmniRoute csomagban található OAuth hitelesítő adatok**csak a „localhost” számára vannak regisztrálva**. Amikor egy távoli szerveren éri el az OmniRoute-ot (pl. `https://omniroute.myserver.com`), a Google a következőkkel utasítja el a hitelesítést:```
+The **Antigravity** and **Gemini CLI** providers use **Google OAuth 2.0**. Google requires the `redirect_uri` in the OAuth flow to exactly match one of the pre-registered URIs in the app's Google Cloud Console.
+
+The OAuth credentials bundled in OmniRoute are registered **for `localhost` only**. When you access OmniRoute on a remote server (e.g. `https://omniroute.myserver.com`), Google rejects the authentication with:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solution: Configure your own OAuth credentials
-Létre kell hoznia egy**OAuth 2.0 ügyfél-azonosítót**a Google Cloud Console-ban a szerver URI-jával.#### Step-by-step
+You need to create an **OAuth 2.0 Client ID** in Google Cloud Console with your server's URI.
-**1. Nyissa meg a Google Cloud Console-t**
+#### Step-by-step
-Keresse fel a következőt: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
+**1. Open Google Cloud Console**
-**2. Új OAuth 2.0 ügyfél-azonosító létrehozása**
+Go to: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-- Kattintson a**„+ Hitelesítési adatok létrehozása”**→**„OAuth-ügyfélazonosító”**elemre.
-- Alkalmazás típusa:**"Web alkalmazás"**
-- Név: bármi, ami tetszik (pl. "OmniRoute Remote")
+**2. Create a new OAuth 2.0 Client ID**
-**3. Engedélyezett átirányítási URI-k hozzáadása**
+- Click **"+ Create Credentials"** → **"OAuth client ID"**
+- Application type: **"Web application"**
+- Name: anything you like (e.g. `OmniRoute Remote`)
-Az**"Engedélyezett átirányítási URI-k"**mezőbe írja be:```
+**3. Add Authorized Redirect URIs**
+
+In the **"Authorized redirect URIs"** field, add:
+
+```
https://your-server.com/callback
+```
-````
+> Replace `your-server.com` with your server's domain or IP (include the port if needed, e.g. `http://45.33.32.156:20128/callback`).
-> Cserélje ki a "your-server.com" címet a szerver domainjére vagy IP-címére (ha szükséges, adja meg a portot, pl. "http://45.33.32.156:20128/callback").
+**4. Save and copy the credentials**
-**4. Mentse és másolja a hitelesítő adatokat**
+After creating, Google will show the **Client ID** and **Client Secret**.
-A létrehozás után a Google megjeleníti az**Client ID**és**Client Secret**kódot.
+**5. Set environment variables**
-**5. Környezeti változók beállítása**
+In your `.env` (or Docker environment variables):
-Az „.env” (vagy a Docker környezeti változókban):```bash
+```bash
# For Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
@@ -1798,77 +2041,88 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_OAUTH_CLIENT_ID=your-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-your-secret
-````
+```
-**6. Az OmniRoute újraindítása**```bash
+**6. Restart OmniRoute**
+```bash
# npm:
-
npm run dev
# Docker:
-
docker restart omniroute
+```
-````
+**7. Try connecting again**
-**7. Próbáljon újra csatlakozni**
+Dashboard → Providers → Antigravity (or Gemini CLI) → OAuth
-Irányítópult → Szolgáltatók → Antigravity (vagy Gemini CLI) → OAuth
+Google will now redirect correctly to `https://your-server.com/callback`.
-A Google most megfelelően átirányítja a `https://your-server.com/callback` címre.---
+---
#### Temporary workaround (without custom credentials)
-Ha most nem szeretné beállítani a saját hitelesítő adatait, továbbra is használhatja a**manuális URL-folyamatot**:
+If you don't want to set up your own credentials right now, you can still use the **manual URL flow**:
-1. Az OmniRoute megnyitja a Google engedélyezési URL-címét
+1. OmniRoute opens the Google authorization URL
2. After authorizing, Google tries to redirect to `localhost` (which fails on the remote server)
-3.**Másolja ki a teljes URL-t**a böngésző címsorából (még akkor is, ha az oldal nem töltődik be)
-4. Illessze be az URL-t az OmniRoute csatlakozási módban látható mezőbe
-5. Kattintson a**"Csatlakozás"**gombra.
+3. **Copy the full URL** from your browser's address bar (even if the page doesn't load)
+4. Paste that URL into the field shown in the OmniRoute connection modal
+5. Click **"Connect"**
-> Ez azért működik, mert az URL-ben szereplő engedélyezési kód attól függetlenül érvényes, hogy az átirányítási oldal betöltődött-e.---
+> This works because the authorization code in the URL is valid regardless of whether the redirect page loaded.
-
-🇧🇷 Versão em Português
#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+---
-Az**Antigravitáció**és a**Gemini CLI**usam**Google OAuth 2.0**hitelesítése. O Google exige que a `redirect_uri` usada no fluxo OAuth seja**exatamente**uma das URI-k pre-cadastradas no Google Cloud Console do aplicativo.
+
+🇧🇷 Versão em Português
-As credenciais OAuth embutidas no OmniRoute estão cadastradas**apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (pl.: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:```
+#### Por que o OAuth do Antigravity / Gemini CLI falha em servidores remotos?
+
+Os provedores **Antigravity** e **Gemini CLI** usam **Google OAuth 2.0** para autenticação. O Google exige que a `redirect_uri` usada no fluxo OAuth seja **exatamente** uma das URIs pré-cadastradas no Google Cloud Console do aplicativo.
+
+As credenciais OAuth embutidas no OmniRoute estão cadastradas **apenas para `localhost`**. Quando você acessa o OmniRoute em um servidor remoto (ex: `https://omniroute.meuservidor.com`), o Google rejeita a autenticação com:
+
+```
Error 400: redirect_uri_mismatch
-````
+```
#### Solução: Configure suas próprias credenciais OAuth
-Você precisa criar um**OAuth 2.0 ügyfél-azonosító**nincs Google Cloud Console com egy URI do seu servidor.#### Passo a passo
+Você precisa criar um **OAuth 2.0 Client ID** no Google Cloud Console com a URI do seu servidor.
+
+#### Passo a passo
**1. Acesse o Google Cloud Console**
Abra: [https://console.cloud.google.com/apis/credentials](https://console.cloud.google.com/apis/credentials)
-**2. Crie um novo OAuth 2.0 ügyfél-azonosító**
+**2. Crie um novo OAuth 2.0 Client ID**
-- Kattintson a gombra**"+ Hitelesítési adatok létrehozása"**→**"OAuth-kliens-azonosító"**
-- Tipo de Aplicativo:**"Web alkalmazás"**
-- Név: escolha qualquer nome (pl.: "OmniRoute Remote")
+- Clique em **"+ Create Credentials"** → **"OAuth client ID"**
+- Tipo de aplicativo: **"Web application"**
+- Nome: escolha qualquer nome (ex: `OmniRoute Remote`)
-**3. Adicione mint engedélyezett átirányítási URI**
+**3. Adicione as Authorized Redirect URIs**
-No campo**"Engedélyezett átirányítási URI-k"**, kiegészítés:```
+No campo **"Authorized redirect URIs"**, adicione:
+
+```
https://seu-servidor.com/callback
+```
-````
+> Substitua `seu-servidor.com` pelo domínio ou IP do seu servidor (inclua a porta se necessário, ex: `http://45.33.32.156:20128/callback`).
-> Helyettesítse a "seu-servidor.com" pelo domínio vagy IP do seu servidor címet (beleértve a porta se necessário-t is, pl.: "http://45.33.32.156:20128/callback").
+**4. Salve e copie as credenciais**
-**4. Másolat mentése hitelesítésként**
+Após criar, o Google mostrará o **Client ID** e o **Client Secret**.
-Após criar, o Google mostrará o**Client ID**e o**Client Secret**.
+**5. Configure as variáveis de ambiente**
-**5. Konfigurálás variáveis de ambienteként**
+No seu `.env` (ou nas variáveis de ambiente do Docker):
-No seu `.env` (ou nas variáveis de ambiente do Docker):```bash
+```bash
# Para Antigravity:
ANTIGRAVITY_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
@@ -1877,37 +2131,39 @@ ANTIGRAVITY_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_OAUTH_CLIENT_ID=seu-client-id.apps.googleusercontent.com
GEMINI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
GEMINI_CLI_OAUTH_CLIENT_SECRET=GOCSPX-seu-secret
-````
+```
-**6. Reinicie o OmniRoute**```bash
+**6. Reinicie o OmniRoute**
+```bash
# Se usando npm:
-
npm run dev
# Se usando Docker:
-
docker restart omniroute
-
-````
+```
**7. Tente conectar novamente**
-Irányítópult → Szolgáltatók → Antigravity (vagy Gemini CLI) → OAuth
+Dashboard → Providers → Antigravity (ou Gemini CLI) → OAuth
-Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` és autenticação funcionará.---
+Agora o Google redirecionará corretamente para `https://seu-servidor.com/callback` e a autenticação funcionará.
+
+---
#### Workaround temporário (sem configurar credenciais próprias)
-Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo**manual de URL**:
+Se não quiser criar credenciais próprias agora, ainda é possível usar o fluxo **manual de URL**:
-1. OmniRoute abrirá a Google autorização URL-jét
-2. Após você autorizar, o Google tentará redirecionar para "localhost" (que falha no servidor remoto)
-3.**Teljes URL másolása**da barra de endereço do seu browser (mesmo que a página não carregue)
+1. O OmniRoute abrirá a URL de autorização do Google
+2. Após você autorizar, o Google tentará redirecionar para `localhost` (que falha no servidor remoto)
+3. **Copie a URL completa** da barra de endereço do seu browser (mesmo que a página não carregue)
4. Cole essa URL no campo que aparece no modal de conexão do OmniRoute
-5. Kattintson a**"Connect"**gombra
+5. Clique em **"Connect"**
-> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+> Este workaround funciona porque o código de autorização na URL é válido independente do redirect ter carregado ou não.
+
+
---
@@ -1915,64 +2171,73 @@ Se não quiser criar credenciais próprias agora, ainda é possível usar o flux
## 🛠️ Tech Stack
-
-Kattintson ide a technológiai verem részleteinek kibontásához
+
+Click to expand tech stack details
--**Futtatási idejű**: Node.js 18–22 LTS (⚠️ A Node.js 24+**nem támogatott**– a `better-sqlite3` natív binárisok nem kompatibilisek)
--**Nyelv**: TypeScript 5.9 –**100% TypeScript**az „src/” és az „open-sse/” protokollokon keresztül (nulla „bármilyen” az alapmodulokban a v2.0 óta)
--**Keretrendszer**: Next.js 16 + React 19 + Tailwind CSS 4
--**Adatbázis**: LowDB (JSON) + SQLite (tartomány állapota + proxynaplók + MCP-audit + útválasztási döntések)
--**Sémák**: Zod (MCP-eszköz I/O-ellenőrzése, API-szerződések)
--**Protokollok**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
--**Streaming**: Szerver által küldött események (SSE)
--**Auth**: OAuth 2.0 (PKCE) + JWT + API-kulcsok + MCP-hatókörű engedélyezés
--**Tesztelés**: Node.js tesztfutó + Vitest (900+ teszt, beleértve az egységet, az integrációt, az E2E-t)
--**CI/CD**: GitHub Actions (automatikus npm közzététel + Docker Hub kiadáskor)
--**Webhely**: [omniroute.online](https://omniroute.online)
--**Csomag**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
--**Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
--**Rugalmasság**: megszakító, exponenciális visszakapcsolás, mennydörgés elleni csorda, TLS-hamisítás, automatikus kombinált öngyógyítás
+- **Runtime**: Node.js 18–22 LTS (⚠️ Node.js 24+ is **not supported** — `better-sqlite3` native binaries are incompatible)
+- **Language**: TypeScript 5.9 — **100% TypeScript** across `src/` and `open-sse/` (zero `any` in core modules since v2.0)
+- **Framework**: Next.js 16 + React 19 + Tailwind CSS 4
+- **Database**: LowDB (JSON) + SQLite (domain state + proxy logs + MCP audit + routing decisions)
+- **Schemas**: Zod (MCP tool I/O validation, API contracts)
+- **Protocols**: MCP (stdio/HTTP) + A2A v0.3 (JSON-RPC 2.0 + SSE)
+- **Streaming**: Server-Sent Events (SSE)
+- **Auth**: OAuth 2.0 (PKCE) + JWT + API Keys + MCP Scoped Authorization
+- **Testing**: Node.js test runner + Vitest (900+ tests including unit, integration, E2E)
+- **CI/CD**: GitHub Actions (auto npm publish + Docker Hub on release)
+- **Website**: [omniroute.online](https://omniroute.online)
+- **Package**: [npmjs.com/package/omniroute](https://www.npmjs.com/package/omniroute)
+- **Docker**: [hub.docker.com/r/diegosouzapw/omniroute](https://hub.docker.com/r/diegosouzapw/omniroute)
+- **Resilience**: Circuit breaker, exponential backoff, anti-thundering herd, TLS spoofing, auto-combo self-healing
+
+
---
## Dokumentáció
-| dokumentum | Leírás |
-| ----------------------------------------------- | ---------------------------------------------------- |
-| [Felhasználói útmutató](docs/USER_GUIDE.md) | Szolgáltatók, kombók, CLI-integráció, telepítés |
-| [API-referencia](docs/API_REFERENCE.md) | Minden végpont példákkal |
-| [MCP-kiszolgáló](open-sse/mcp-server/README.md) | 16 MCP-eszköz, IDE konfigurációk, Python/TS/Go kliensek |
-| [A2A szerver](src/lib/a2a/README.md) | JSON-RPC 2.0 protokoll, készségek, adatfolyam, feladat mgmt |
-| [Auto-Combo Engine](docs/auto-combo.md) | 6 faktoros pontozás, módcsomagok, öngyógyító |
-| [Hibaelhárítás](docs/TROUBLESHOOTING.md) | Gyakori problémák és megoldások |
-| [Architektúra](docs/ARCHITECTURE.md) | Rendszerarchitektúra és belső elemek |
-| [Hozzájárulás](CONTRIBUTING.md) | Fejlesztési beállítások és irányelvek |
-| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specifikáció |
-| [Biztonsági politika](SECURITY.md) | Sebezhetőségi jelentések és biztonsági gyakorlatok |
-| [VM-telepítés](docs/VM_DEPLOYMENT_GUIDE.md) | Teljes útmutató: VM + nginx + Cloudflare beállítás |
-| [Features Gallery](docs/FEATURES.md) | Vizuális irányítópult bemutató képernyőképekkel |
-| [Kiadási ellenőrzőlista](docs/RELEASE_CHECKLIST.md) | Kiadás előtti érvényesítési lépések |---
+| Document | Description |
+| ---------------------------------------------- | --------------------------------------------------- |
+| [User Guide](docs/USER_GUIDE.md) | Providers, combos, CLI integration, deployment |
+| [API Reference](docs/API_REFERENCE.md) | All endpoints with examples |
+| [MCP Server](open-sse/mcp-server/README.md) | 25 MCP tools, IDE configs, Python/TS/Go clients |
+| [A2A Server](src/lib/a2a/README.md) | JSON-RPC 2.0 protocol, skills, streaming, task mgmt |
+| [Auto-Combo Engine](docs/auto-combo.md) | 6-factor scoring, mode packs, self-healing |
+| [Context Relay](docs/features/context-relay.md)| Session handoff strategy for account rotation |
+| [Troubleshooting](docs/TROUBLESHOOTING.md) | Common problems and solutions |
+| [Architecture](docs/ARCHITECTURE.md) | System architecture and internals |
+| [Contributing](CONTRIBUTING.md) | Development setup and guidelines |
+| [OpenAPI Spec](docs/openapi.yaml) | OpenAPI 3.0 specification |
+| [Security Policy](SECURITY.md) | Vulnerability reporting and security practices |
+| [VM Deployment](docs/VM_DEPLOYMENT_GUIDE.md) | Complete guide: VM + nginx + Cloudflare setup |
+| [Features Gallery](docs/FEATURES.md) | Visual dashboard tour with screenshots |
+| [Release Checklist](docs/RELEASE_CHECKLIST.md) | Pre-release validation steps |
+
+---
## 🗺️ Roadmap
-Az OmniRoute**210+ funkciót tervez**több fejlesztési fázisban. Íme a legfontosabb területek:
+OmniRoute has **210+ features planned** across multiple development phases. Here are the key areas:
-| Kategória | Tervezett funkciók | Kiemelések |
-| ------------------------------ | ----------------- | -------------------------------------------------------------------------------------- |
-| 🧠**Útválasztás és intelligencia**| 25+ | Legkisebb késleltetésű útválasztás, címke alapú útválasztás, kvóta elővizsgálat, P2C-fiók kiválasztása |
-| 🔒**Biztonság és megfelelőség**| 20+ | SSRF keményítés, hitelesítő adatok álcázása, végpontonkénti sebességkorlát, felügyeleti kulcs hatóköre |
-| 📊**Megfigyelhetőség**| 15+ | OpenTelemetry integráció, valós idejű kvótafigyelés, modellenkénti költségkövetés |
-| 🔄**Szolgáltatói integrációk**| 20+ | Dinamikus modellnyilvántartás, szolgáltatói leállások, többfiókos Codex, másodpilóta kvótaelemzés |
-| ⚡**Teljesítmény**| 15+ | Kettős gyorsítótárréteg, gyorsítótár, válaszgyorsítótár, folyamatos adatfolyam, kötegelt API |
-| 🌐**Ökoszisztéma**| 10+ | WebSocket API, config hot-reload, elosztott konfigurációs tároló, kereskedelmi mód |### 🔜 Coming Soon
+| Category | Planned Features | Highlights |
+| ----------------------------- | ---------------- | -------------------------------------------------------------------------------------- |
+| 🧠 **Routing & Intelligence** | 25+ | Lowest-latency routing, tag-based routing, quota preflight, P2C account selection |
+| 🔒 **Security & Compliance** | 20+ | SSRF hardening, credential cloaking, rate-limit per endpoint, management key scoping |
+| 📊 **Observability** | 15+ | OpenTelemetry integration, real-time quota monitoring, cost tracking per model |
+| 🔄 **Provider Integrations** | 20+ | Dynamic model registry, provider cooldowns, multi-account Codex, Copilot quota parsing |
+| ⚡ **Performance** | 15+ | Dual cache layer, prompt cache, response cache, streaming keepalive, batch API |
+| 🌐 **Ecosystem** | 10+ | WebSocket API, config hot-reload, distributed config store, commercial mode |
-- 🔗**OpenCode integráció**- Natív szolgáltatói támogatás az OpenCode AI kódoló IDE-hez
-- 🔗**TRAE integráció**— A TRAE AI fejlesztési keret teljes támogatása
-- 📦**Batch API**- Aszinkron kötegelt feldolgozás tömeges kérésekhez
-- 🎯**Címke alapú útválasztás**- Egyéni címkéken és metaadatokon alapuló útvonalkérések
-- 💰**Legalacsonyabb költségű stratégia**- Automatikusan válassza ki a legolcsóbb elérhető szolgáltatót
+### 🔜 Coming Soon
-> 📝 A teljes funkcióspecifikáció elérhető a [`docs/new-features/`](docs/new-features/) oldalon (217 részletes specifikáció)---
+- 🔗 **OpenCode Integration** — Native provider support for the OpenCode AI coding IDE
+- 🔗 **TRAE Integration** — Full support for the TRAE AI development framework
+- 📦 **Batch API** — Asynchronous batch processing for bulk requests
+- 🎯 **Tag-Based Routing** — Route requests based on custom tags and metadata
+- 💰 **Lowest-Cost Strategy** — Automatically select the cheapest available provider
+
+> 📝 Full feature specifications available in [`docs/new-features/`](docs/new-features/) (217 detailed specs)
+
+---
## 👥 Contributors
@@ -1980,18 +2245,20 @@ Az OmniRoute**210+ funkciót tervez**több fejlesztési fázisban. Íme a legfon
### How to Contribute
-1. Fork a tároló
-2. Hozza létre a szolgáltatási ágat (`git checkout -b feature/amazing-feature`)
-3. Végezze el a változtatásokat (`git commit -m 'Elképesztő funkció hozzáadása'`)
-4. Nyomja le az ágra (`git push origin funkció/csodálatos szolgáltatás`)
-5. Nyisson meg egy lehívási kérelmet
+1. Fork the repository
+2. Create your feature branch (`git checkout -b feature/amazing-feature`)
+3. Commit your changes (`git commit -m 'Add amazing feature'`)
+4. Push to the branch (`git push origin feature/amazing-feature`)
+5. Open a Pull Request
-A részletes útmutatásért lásd: [CONTRIBUTING.md](CONTRIBUTING.md).### Releasing a New Version
+See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines.
+
+### Releasing a New Version
```bash
# Create a release — npm publish happens automatically
gh release create v2.0.0 --title "v2.0.0" --generate-notes
-````
+```
---
@@ -2003,13 +2270,17 @@ gh release create v2.0.0 --title "v2.0.0" --generate-notes
## 🙏 Acknowledgments
-Külön köszönet a**[decolua](https://github.com/decolua)\*\***[9router](https://github.com/decolua/9router)\*\*-nak – az eredeti projektnek, amely ezt a villát inspirálta. Az OmniRoute erre a hihetetlen alapra épít további funkciókkal, multimodális API-kkal és teljes TypeScript-újraírással.
+Special thanks to **[9router](https://github.com/decolua/9router)** by **[decolua](https://github.com/decolua)** — the original project that inspired this fork. OmniRoute builds upon that incredible foundation with additional features, multi-modal APIs, and a full TypeScript rewrite.
-Külön köszönet a**[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)**-nak – az eredeti Go implementációnak, amely ihlette ezt a JavaScript-portot.---
+Special thanks to **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — the original Go implementation that inspired this JavaScript port.
+
+---
## Licenc
-MIT-licenc – részletekért lásd: [LICENSE](LICENSE).---
+MIT License - see [LICENSE](LICENSE) for details.
+
+---
Built with ❤️ for developers who code 24/7
diff --git a/docs/i18n/hu/docs/ARCHITECTURE.md b/docs/i18n/hu/docs/ARCHITECTURE.md
index f659c6ecb4..aab37b02ed 100644
--- a/docs/i18n/hu/docs/ARCHITECTURE.md
+++ b/docs/i18n/hu/docs/ARCHITECTURE.md
@@ -4,80 +4,93 @@
---
-_Utolsó frissítés: 2026-03-28_## Executive Summary
-Az OmniRoute egy helyi mesterséges intelligencia-útválasztó átjáró és irányítópult, amely a Next.js-re épül.
-Egyetlen OpenAI-kompatibilis végpontot (`/v1/*`) biztosít, és a forgalmat több upstream szolgáltató között irányítja át fordítással, tartalékkal, tokenfrissítéssel és használati követéssel.
-Alapvető képességek:
+_Last updated: 2026-03-28_
-- OpenAI-kompatibilis API felület a CLI-hez/eszközökhöz (28 szolgáltató)
-- Fordítás kérése/válaszolása a szolgáltatói formátumok között
-- Model kombinált tartalék (több modell sorozat)
-- Fiókszintű tartalék (szolgáltatónként több fiók)
-- OAuth + API-kulcs szolgáltatói kapcsolatkezelés
-- Beágyazás generálása a „/v1/embeddings” fájlon keresztül (6 szolgáltató, 9 modell)
-- Képgenerálás a `/v1/images/generations' fájlon keresztül (4 szolgáltató, 9 modell)
-- Gondoljon a címkeelemzésre (`...`) az érvelési modellekhez
-- Válasz fertőtlenítés a szigorú OpenAI SDK-kompatibilitás érdekében
-- Szerepnormalizálás (fejlesztő→rendszer, rendszer→felhasználó) a szolgáltatók közötti kompatibilitás érdekében
-- Strukturált kimenet átalakítás (json_schema → Gemini responseSchema)
-- Helyi kitartás a szolgáltatók, kulcsok, álnevek, kombinációk, beállítások, árképzés számára
-- Használat/költségkövetés és kérések naplózása
-- Opcionális felhőszinkronizálás több eszköz/állapot szinkronizáláshoz
-- IP engedélyezési/blokkolási lista API hozzáférés-vezérléshez
-- Átgondolt költségvetés-kezelés (áthaladó/automatikus/egyéni/adaptív)
-- Globális rendszer azonnali befecskendezése
-- Munkamenet követés és ujjlenyomat
-- Fiókonként továbbfejlesztett díjkorlátozás szolgáltató-specifikus profilokkal
-- Megszakító minta a szolgáltatói rugalmasság érdekében
-- Mennydörgés elleni állományvédelem mutex zárral
-- Aláírás alapú kérés deduplikációs gyorsítótár
-- Domain réteg: modell elérhetősége, költségszabályok, tartalék házirend, kizárási szabályzat
-- Tartomány állapotának fennmaradása (SQLite átírási gyorsítótár tartalékok, költségvetések, zárolások, megszakítók számára)
-- Házirend motor a kérelmek központosított értékeléséhez (zárás → költségvetés → tartalék)
-- Telemetria kérése p50/p95/p99 késleltetési összesítéssel
-- Korrelációs azonosító (X-Request-Id) a végpontok közötti nyomkövetéshez
-- Megfelelőségi naplózás API-kulcsonkénti leiratkozással
-- Eval keretrendszer az LLM minőségbiztosításhoz
-- Rugalmas UI műszerfal valós idejű megszakító állapottal
-- Moduláris OAuth-szolgáltatók (12 különálló modul az `src/lib/oauth/providers/` alatt)
+## Executive Summary
-Elsődleges futásidejű modell:
+OmniRoute is a local AI routing gateway and dashboard built on Next.js.
+It provides a single OpenAI-compatible endpoint (`/v1/*`) and routes traffic across multiple upstream providers with translation, fallback, token refresh, and usage tracking.
-- A Next.js alkalmazásútvonalai az `src/app/api/*` alatt mind az irányítópult API-kat, mind a kompatibilitási API-kat megvalósítják
-- Egy megosztott SSE/routing mag az `src/sse/*` + `open-sse/*` állományban kezeli a szolgáltató végrehajtását, fordítását, adatfolyamát, tartalékát és használatát## Scope and Boundaries
+Core capabilities:
+
+- OpenAI-compatible API surface for CLI/tools (28 providers)
+- Request/response translation across provider formats
+- Model combo fallback (multi-model sequence)
+- Account-level fallback (multi-account per provider)
+- OAuth + API-key provider connection management
+- Embedding generation via `/v1/embeddings` (6 providers, 9 models)
+- Image generation via `/v1/images/generations` (4 providers, 9 models)
+- Think tag parsing (`...`) for reasoning models
+- Response sanitization for strict OpenAI SDK compatibility
+- Role normalization (developer→system, system→user) for cross-provider compatibility
+- Structured output conversion (json_schema → Gemini responseSchema)
+- Local persistence for providers, keys, aliases, combos, settings, pricing
+- Usage/cost tracking and request logging
+- Optional cloud sync for multi-device/state sync
+- IP allowlist/blocklist for API access control
+- Thinking budget management (passthrough/auto/custom/adaptive)
+- Global system prompt injection
+- Session tracking and fingerprinting
+- Per-account enhanced rate limiting with provider-specific profiles
+- Circuit breaker pattern for provider resilience
+- Anti-thundering herd protection with mutex locking
+- Signature-based request deduplication cache
+- Domain layer: model availability, cost rules, fallback policy, lockout policy
+- Context Relay: session handoff summaries for account rotation continuity
+- Domain state persistence (SQLite write-through cache for fallbacks, budgets, lockouts, circuit breakers)
+- Policy engine for centralized request evaluation (lockout → budget → fallback)
+- Request telemetry with p50/p95/p99 latency aggregation
+- Correlation ID (X-Request-Id) for end-to-end tracing
+- Compliance audit logging with opt-out per API key
+- Eval framework for LLM quality assurance
+- Resilience UI dashboard with real-time circuit breaker status
+- Modular OAuth providers (12 individual modules under `src/lib/oauth/providers/`)
+
+Primary runtime model:
+
+- Next.js app routes under `src/app/api/*` implement both dashboard APIs and compatibility APIs
+- A shared SSE/routing core in `src/sse/*` + `open-sse/*` handles provider execution, translation, streaming, fallback, and usage
+
+## Scope and Boundaries
### In Scope
-- Helyi átjáró futásidejű
-- Irányítópult-kezelő API-k
-- Szolgáltató hitelesítése és token frissítése
-- Fordítás és SSE streaming kérése
-- Helyi állapot + használat tartóssága
-- Opcionális felhőszinkronizálás### Out of Scope
+- Local gateway runtime
+- Dashboard management APIs
+- Provider authentication and token refresh
+- Request translation and SSE streaming
+- Local state + usage persistence
+- Optional cloud sync orchestration
-- Felhőszolgáltatás megvalósítása a `NEXT_PUBLIC_CLOUD_URL' mögött
-- Szolgáltató SLA/vezérlő síkja a helyi folyamaton kívül
-- Maguk a külső CLI binárisok (Claude CLI, Codex CLI stb.)## Dashboard Surface (Current)
+### Out of Scope
-Főoldalak az `src/app/(dashboard)/dashboard/` alatt:
+- Cloud service implementation behind `NEXT_PUBLIC_CLOUD_URL`
+- Provider SLA/control plane outside local process
+- External CLI binaries themselves (Claude CLI, Codex CLI, etc.)
-- `/dashboard` — gyorsindítás + szolgáltató áttekintése
-- "/dashboard/endpoint" - végpont proxy + MCP + A2A + API végpont lapjai
-- "/dashboard/providers" – szolgáltatói kapcsolatok és hitelesítő adatok
-- "/dashboard/combos" - kombinált stratégiák, sablonok, modell-útválasztási szabályok
-- "/dashboard/costs" – a költségek összesítése és az árképzés láthatósága
-- "/dashboard/analytics" — használati elemzések és kiértékelések
-- "/dashboard/limits" – kvóta/kamat szabályozás
-- "/dashboard/cli-tools" - CLI-beépítés, futásidejű észlelés, konfiguráció generálása
-- "/dashboard/agents" — észlelt ACP ügynökök + egyéni ügynök regisztráció
-- `/dashboard/media` — kép/videó/zene játszótér
-- "/dashboard/search-tools" – a keresőszolgáltató tesztelése és előzményei
-- "/dashboard/health" – üzemidő, megszakítók, sebességkorlátok
-- "/dashboard/logs" — kérés/proxy/audit/konzolnaplók
-- "/dashboard/settings" – rendszerbeállítások lapjai (általános, útválasztás, kombinált alapértelmezések stb.)
-- `/dashboard/api-manager` – API kulcs életciklusa és modellengedélyei## High-Level System Context
+## Dashboard Surface (Current)
+
+Main pages under `src/app/(dashboard)/dashboard/`:
+
+- `/dashboard` — quick start + provider overview
+- `/dashboard/endpoint` — endpoint proxy + MCP + A2A + API endpoint tabs
+- `/dashboard/providers` — provider connections and credentials
+- `/dashboard/combos` — combo strategies, templates, model routing rules
+- `/dashboard/costs` — cost aggregation and pricing visibility
+- `/dashboard/analytics` — usage analytics and evaluations
+- `/dashboard/limits` — quota/rate controls
+- `/dashboard/cli-tools` — CLI onboarding, runtime detection, config generation
+- `/dashboard/agents` — detected ACP agents + custom agent registration
+- `/dashboard/media` — image/video/music playground
+- `/dashboard/search-tools` — search provider testing and history
+- `/dashboard/health` — uptime, circuit breakers, rate limits
+- `/dashboard/logs` — request/proxy/audit/console logs
+- `/dashboard/settings` — system settings tabs (general, routing, combo defaults, etc.)
+- `/dashboard/api-manager` — API key lifecycle and model permissions
+
+## High-Level System Context
```mermaid
flowchart LR
@@ -129,139 +142,151 @@ flowchart LR
## 1) API and Routing Layer (Next.js App Routes)
-Fő könyvtárak:
+Main directories:
-- `src/app/api/v1/*` és `src/app/api/v1beta/*` a kompatibilitási API-khoz
-- `src/app/api/*` a felügyeleti/konfigurációs API-khoz
-- Következő átírja a `next.config.mjs` `/v1/*` leképezését `/api/v1/*`-re
+- `src/app/api/v1/*` and `src/app/api/v1beta/*` for compatibility APIs
+- `src/app/api/*` for management/configuration APIs
+- Next rewrites in `next.config.mjs` map `/v1/*` to `/api/v1/*`
-Fontos kompatibilitási útvonalak:
+Important compatibility routes:
-- "src/app/api/v1/chat/completions/route.ts".
+- `src/app/api/v1/chat/completions/route.ts`
- `src/app/api/v1/messages/route.ts`
- `src/app/api/v1/responses/route.ts`
-- "src/app/api/v1/models/route.ts" - egyéni modelleket tartalmaz "custom: true"
-- "src/app/api/v1/embeddings/route.ts" - beágyazás generálása (6 szolgáltató)
-- "src/app/api/v1/images/generations/route.ts" - képgenerálás (4+ szolgáltató, beleértve az Antigravitációt/Nebiust)
+- `src/app/api/v1/models/route.ts` — includes custom models with `custom: true`
+- `src/app/api/v1/embeddings/route.ts` — embedding generation (6 providers)
+- `src/app/api/v1/images/generations/route.ts` — image generation (4+ providers incl. Antigravity/Nebius)
- `src/app/api/v1/messages/count_tokens/route.ts`
-- "src/app/api/v1/providers/[provider]/chat/completions/route.ts" – dedikált szolgáltatónkénti csevegés
-- "src/app/api/v1/providers/[szolgáltató]/embeddings/route.ts" – dedikált szolgáltatónkénti beágyazások
-- "src/app/api/v1/providers/[szolgáltató]/images/generations/route.ts" – szolgáltatónként dedikált képek
+- `src/app/api/v1/providers/[provider]/chat/completions/route.ts` — dedicated per-provider chat
+- `src/app/api/v1/providers/[provider]/embeddings/route.ts` — dedicated per-provider embeddings
+- `src/app/api/v1/providers/[provider]/images/generations/route.ts` — dedicated per-provider images
- `src/app/api/v1beta/models/route.ts`
-- `src/app/api/v1beta/models/[...útvonal]/route.ts`
+- `src/app/api/v1beta/models/[...path]/route.ts`
-Kezelési tartományok:
+Management domains:
-- Hitelesítés/beállítások: `src/app/api/auth/*`, `src/app/api/settings/*`
-- Szolgáltatók/kapcsolatok: `src/app/api/providers*`
-- Szolgáltatói csomópontok: `src/app/api/provider-nodes*`
-- Egyéni modellek: "src/app/api/provider-models" (GET/POST/DELETE)
-- Modellkatalógus: `src/app/api/models/route.ts` (GET)
-- Proxy konfigurációja: "src/app/api/settings/proxy" (GET/PUT/DELETE) + "src/app/api/settings/proxy/test" (POST)
+- Auth/settings: `src/app/api/auth/*`, `src/app/api/settings/*`
+- Providers/connections: `src/app/api/providers*`
+- Provider nodes: `src/app/api/provider-nodes*`
+- Custom models: `src/app/api/provider-models` (GET/POST/DELETE)
+- Model catalog: `src/app/api/models/route.ts` (GET)
+- Proxy config: `src/app/api/settings/proxy` (GET/PUT/DELETE) + `src/app/api/settings/proxy/test` (POST)
- OAuth: `src/app/api/oauth/*`
-- Keys/aliases/combos/pricing: "src/app/api/keys*", "src/app/api/models/alias", "src/app/api/combos*", "src/app/api/pricing"
-- Használat: `src/app/api/usage/*`
-- Szinkronizálás/felhő: `src/app/api/sync/*`, `src/app/api/cloud/*`
-- CLI-eszközök segédei: `src/app/api/cli-tools/*`
-- IP-szűrő: "src/app/api/settings/ip-filter" (GET/PUT)
-- Gondolkodási költségkeret: `src/app/api/settings/thinking-budget' (GET/PUT)
-- Rendszerprompt: "src/app/api/settings/system-prompt" (GET/PUT)
-- Munkamenetek: `src/app/api/sessions' (GET)
-- Díjkorlátok: "src/app/api/rate-limits" (GET)
-- Rugalmasság: "src/app/api/resilience" (GET/PATCH) – szolgáltatói profilok, megszakító, sebességkorlát állapot
-- Rugalmasság visszaállítása: `src/app/api/resilience/reset' (POST) – megszakítók visszaállítása + lehűlés
-- Gyorsítótár statisztikái: "src/app/api/cache/stats" (GET/DELETE)
-- A modell elérhetősége: "src/app/api/models/availability" (GET/POST)
-- Telemetria: "src/app/api/telemetry/summary" (GET)
-- Költségkeret: `src/app/api/usage/budget' (GET/POST)
-- Tartalék láncok: `src/app/api/fallback/chains' (GET/POST/DELETE)
-- Megfelelőségi ellenőrzés: `src/app/api/compliance/audit-log' (GET)
-- Evals: "src/app/api/evals" (GET/POST), "src/app/api/evals/[suiteId]" (GET)
-- Irányelvek: `src/app/api/policies' (GET/POST)## 2) SSE + Translation Core
+- Keys/aliases/combos/pricing: `src/app/api/keys*`, `src/app/api/models/alias`, `src/app/api/combos*`, `src/app/api/pricing`
+- Usage: `src/app/api/usage/*`
+- Sync/cloud: `src/app/api/sync/*`, `src/app/api/cloud/*`
+- CLI tooling helpers: `src/app/api/cli-tools/*`
+- IP filter: `src/app/api/settings/ip-filter` (GET/PUT)
+- Thinking budget: `src/app/api/settings/thinking-budget` (GET/PUT)
+- System prompt: `src/app/api/settings/system-prompt` (GET/PUT)
+- Sessions: `src/app/api/sessions` (GET)
+- Rate limits: `src/app/api/rate-limits` (GET)
+- Resilience: `src/app/api/resilience` (GET/PATCH) — provider profiles, circuit breaker, rate limit state
+- Resilience reset: `src/app/api/resilience/reset` (POST) — reset breakers + cooldowns
+- Cache stats: `src/app/api/cache/stats` (GET/DELETE)
+- Model availability: `src/app/api/models/availability` (GET/POST)
+- Telemetry: `src/app/api/telemetry/summary` (GET)
+- Budget: `src/app/api/usage/budget` (GET/POST)
+- Fallback chains: `src/app/api/fallback/chains` (GET/POST/DELETE)
+- Compliance audit: `src/app/api/compliance/audit-log` (GET)
+- Evals: `src/app/api/evals` (GET/POST), `src/app/api/evals/[suiteId]` (GET)
+- Policies: `src/app/api/policies` (GET/POST)
-Fő áramlási modulok:
+## 2) SSE + Translation Core
-- Bejegyzés: `src/sse/handlers/chat.ts`
-- Alapvető hangszerelés: "open-sse/handlers/chatCore.ts"
-- Szolgáltató végrehajtási adapterei: `open-sse/executors/*`
-- Formátumészlelés/szolgáltató konfigurációja: "open-sse/services/provider.ts"
-- Modell parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts`
-- A fiók tartalék logikája: `open-sse/services/accountFallback.ts`
-- Fordítási nyilvántartás: "open-sse/translator/index.ts".
-- Adatfolyam-átalakítások: "open-sse/utils/stream.ts", "open-sse/utils/streamHandler.ts"
-- Használat kibontása/normalizálása: "open-sse/utils/usageTracking.ts"
+Main flow modules:
+
+- Entry: `src/sse/handlers/chat.ts`
+- Core orchestration: `open-sse/handlers/chatCore.ts`
+- Provider execution adapters: `open-sse/executors/*`
+- Format detection/provider config: `open-sse/services/provider.ts`
+- Model parse/resolve: `src/sse/services/model.ts`, `open-sse/services/model.ts`
+- Account fallback logic: `open-sse/services/accountFallback.ts`
+- Translation registry: `open-sse/translator/index.ts`
+- Stream transformations: `open-sse/utils/stream.ts`, `open-sse/utils/streamHandler.ts`
+- Usage extraction/normalization: `open-sse/utils/usageTracking.ts`
- Think tag parser: `open-sse/utils/thinkTagParser.ts`
-- Beágyazáskezelő: `open-sse/handlers/embeddings.ts`
-- Beágyazási szolgáltató nyilvántartása: "open-sse/config/embeddingRegistry.ts"
-- Képgeneráló kezelő: `open-sse/handlers/imageGeneration.ts`
-- Képszolgáltató regisztrációs adatbázisa: `open-sse/config/imageRegistry.ts`
-- A válaszok fertőtlenítése: "open-sse/handlers/responseSanitizer.ts"
-- Szerepkör normalizálása: `open-sse/services/roleNormalizer.ts`
+- Embedding handler: `open-sse/handlers/embeddings.ts`
+- Embedding provider registry: `open-sse/config/embeddingRegistry.ts`
+- Image generation handler: `open-sse/handlers/imageGeneration.ts`
+- Image provider registry: `open-sse/config/imageRegistry.ts`
+- Response sanitization: `open-sse/handlers/responseSanitizer.ts`
+- Role normalization: `open-sse/services/roleNormalizer.ts`
-Szolgáltatások (üzleti logika):
+Services (business logic):
-- Fiókválasztás/pontozás: "open-sse/services/accountSelector.ts"
-- Kontextus-életciklus-kezelés: `open-sse/services/contextManager.ts`
-- IP-szűrő végrehajtása: `open-sse/services/ipFilter.ts`
-- Munkamenetkövetés: `open-sse/services/sessionManager.ts`
-- Deduplikáció kérése: "open-sse/services/signatureCache.ts"
-- Rendszerprompt injekció: `open-sse/services/systemPrompt.ts`
-- Gondolkodó költségvetés-kezelés: `open-sse/services/thinkingBudget.ts`
-- Helyettesítő karakteres modell-útválasztás: "open-sse/services/wildcardRouter.ts"
-- Díjkorlát kezelése: `open-sse/services/rateLimitManager.ts`
-- Megszakító: "open-sse/services/circuitBreaker.ts"
+- Account selection/scoring: `open-sse/services/accountSelector.ts`
+- Context lifecycle management: `open-sse/services/contextManager.ts`
+- IP filter enforcement: `open-sse/services/ipFilter.ts`
+- Session tracking: `open-sse/services/sessionManager.ts`
+- Request deduplication: `open-sse/services/signatureCache.ts`
+- System prompt injection: `open-sse/services/systemPrompt.ts`
+- Thinking budget management: `open-sse/services/thinkingBudget.ts`
+- Wildcard model routing: `open-sse/services/wildcardRouter.ts`
+- Rate limit management: `open-sse/services/rateLimitManager.ts`
+- Circuit breaker: `open-sse/services/circuitBreaker.ts`
+- Context handoff: `open-sse/services/contextHandoff.ts` — handoff summary generation and injection for context-relay strategy
+- Codex quota fetcher: `open-sse/services/codexQuotaFetcher.ts` — fetches Codex quota for context-relay handoff decisions
-Domain réteg modulok:
+Domain layer modules:
-- A modell elérhetősége: `src/lib/domain/modelAvailability.ts`
-- Költségszabályok/költségkeretek: `src/lib/domain/costRules.ts`
-- Tartalék házirend: `src/lib/domain/fallbackPolicy.ts`
-- Kombinált feloldó: `src/lib/domain/comboResolver.ts`
-- Kizárási szabályzat: `src/lib/domain/lockoutPolicy.ts`
-- Házirend motor: `src/domain/policyEngine.ts` — központosított zárolás → költségvetés → tartalék kiértékelés
-- Hibakód-katalógus: "src/lib/domain/errorCodes.ts".
-- Kérelemazonosító: `src/lib/domain/requestId.ts`
-- Lekérési időtúllépés: `src/lib/domain/fetchTimeout.ts`
-- Telemetria kérése: `src/lib/domain/requestTelemetry.ts`
-- Megfelelőség/ellenőrzés: `src/lib/domain/compliance/index.ts`
+- Model availability: `src/lib/domain/modelAvailability.ts`
+- Cost rules/budgets: `src/lib/domain/costRules.ts`
+- Fallback policy: `src/lib/domain/fallbackPolicy.ts`
+- Combo resolver: `src/lib/domain/comboResolver.ts`
+- Lockout policy: `src/lib/domain/lockoutPolicy.ts`
+- Policy engine: `src/domain/policyEngine.ts` — centralized lockout → budget → fallback evaluation
+- Error codes catalog: `src/lib/domain/errorCodes.ts`
+- Request ID: `src/lib/domain/requestId.ts`
+- Fetch timeout: `src/lib/domain/fetchTimeout.ts`
+- Request telemetry: `src/lib/domain/requestTelemetry.ts`
+- Compliance/audit: `src/lib/domain/compliance/index.ts`
- Eval runner: `src/lib/domain/evalRunner.ts`
-- Tartomány állapotának fennmaradása: `src/lib/db/domainState.ts' — SQLite CRUD tartalék láncokhoz, költségvetésekhez, költségelőzményekhez, zárolási állapothoz, megszakítókhoz
+- Domain state persistence: `src/lib/db/domainState.ts` — SQLite CRUD for fallback chains, budgets, cost history, lockout state, circuit breakers
-OAuth-szolgáltató modulok (12 külön fájl az `src/lib/oauth/providers/` alatt):
+OAuth provider modules (12 individual files under `src/lib/oauth/providers/`):
- Registry index: `src/lib/oauth/providers/index.ts`
-- Egyéni szolgáltatók: "claude.ts", "codex.ts", "gemini.ts", "antigravity.ts", "qoder.ts", "qwen.ts", "kimi-coding.ts", "github.ts", "kiro.ts", "codex". `cline.ts`
-- Vékony burkolóanyag: `src/lib/oauth/providers.ts' – újraexportálás az egyes modulokból## 3) Persistence Layer
+- Individual providers: `claude.ts`, `codex.ts`, `gemini.ts`, `antigravity.ts`, `qoder.ts`, `qwen.ts`, `kimi-coding.ts`, `github.ts`, `kiro.ts`, `cursor.ts`, `kilocode.ts`, `cline.ts`
+- Thin wrapper: `src/lib/oauth/providers.ts` — re-exports from individual modules
-Elsődleges állapotú DB (SQLite):
+## 3) Persistence Layer
-- Alapvető infrastruktúra: `src/lib/db/core.ts` (better-sqlite3, migrációk, WAL)
-- Homlokzat újraexportálása: "src/lib/localDb.ts" (vékony kompatibilitási réteg a hívók számára)
-- fájl: `${DATA_DIR}/storage.sqlite` (vagy `$XDG_CONFIG_HOME/omniroute/storage.sqlite`, ha be van állítva, különben `~/.omniroute/storage.sqlite`)
-- entitások (táblák + KV névterek): szolgáltatói kapcsolatok, szolgáltató csomópontok, modellálnevek, kombók, apiKeys, beállítások, árképzés,**customModels**,**proxyConfig**,**ipFilter**,**thhinkingBudget**,**systemPrompt**
+Primary state DB (SQLite):
-Használat tartóssága:
+- Core infra: `src/lib/db/core.ts` (better-sqlite3, migrations, WAL)
+- Re-export facade: `src/lib/localDb.ts` (thin compatibility layer for callers)
+- file: `${DATA_DIR}/storage.sqlite` (or `$XDG_CONFIG_HOME/omniroute/storage.sqlite` when set, else `~/.omniroute/storage.sqlite`)
+- entities (tables + KV namespaces): providerConnections, providerNodes, modelAliases, combos, apiKeys, settings, pricing, **customModels**, **proxyConfig**, **ipFilter**, **thinkingBudget**, **systemPrompt**
-- homlokzat: `src/lib/usageDb.ts` (bontott modulok az `src/lib/usage/*` fájlban)
-- SQLite táblák a "storage.sqlite" fájlban: "használati_előzmények", "hívásnaplók", "proxy_naplók"
-- az opcionális melléktermékek a kompatibilitáshoz/hibakereséshez maradnak (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`)
-- A régebbi JSON-fájlokat a rendszer indítási migrációval költözteti az SQLite-ba, ha vannak
+Usage persistence:
+
+- facade: `src/lib/usageDb.ts` (decomposed modules in `src/lib/usage/*`)
+- SQLite tables in `storage.sqlite`: `usage_history`, `call_logs`, `proxy_logs`
+- optional file artifacts remain for compatibility/debug (`${DATA_DIR}/log.txt`, `${DATA_DIR}/call_logs/`, `/logs/...`)
+- legacy JSON files are migrated to SQLite by startup migrations when present
Domain State DB (SQLite):
-- "src/lib/db/domainState.ts" - CRUD műveletek a tartomány állapotához
-- Táblázatok (az "src/lib/db/core.ts" fájlban létrehozva): "domain_fallback_chains", "domain_budgets", "domain_cost_history", "domain_lockout_state", "domain_circuit_breakers"
-- Átírási gyorsítótár minta: a memórián belüli térképek mérvadóak futás közben; a mutációk szinkronban íródnak az SQLite-ba; állapot visszaáll a DB-ből hidegindításkor## 4) Auth + Security Surfaces
+- `src/lib/db/domainState.ts` — CRUD operations for domain state
+- Tables (created in `src/lib/db/core.ts`): `domain_fallback_chains`, `domain_budgets`, `domain_cost_history`, `domain_lockout_state`, `domain_circuit_breakers`
+- Write-through cache pattern: in-memory Maps are authoritative at runtime; mutations are written synchronously to SQLite; state is restored from DB on cold start
-- Az irányítópult cookie hitelesítése: "src/proxy.ts", "src/app/api/auth/login/route.ts"
-- API-kulcs létrehozása/ellenőrzése: `src/shared/utils/apiKey.ts`
-- A szolgáltatói titkok megmaradtak a "providerConnections" bejegyzésekben
-- Kimenő proxy támogatása az "open-sse/utils/proxyFetch.ts" (env vars) és az "open-sse/utils/networkProxy.ts" segítségével (szolgáltatónként vagy globálisan konfigurálható)## 5) Cloud Sync
+## 4) Auth + Security Surfaces
-- Ütemező init: "src/lib/initCloudSync.ts", "src/shared/services/initializeCloudSync.ts", "src/shared/services/modelSyncScheduler.ts"
-- Időszakos feladat: `src/shared/services/cloudSyncScheduler.ts`
-- Időszakos feladat: `src/shared/services/modelSyncScheduler.ts`
-- Útvonal vezérlése: "src/app/api/sync/cloud/route.ts"## Request Lifecycle (`/v1/chat/completions`)
+- Dashboard cookie auth: `src/proxy.ts`, `src/app/api/auth/login/route.ts`
+- API key generation/verification: `src/shared/utils/apiKey.ts`
+- Provider secrets persisted in `providerConnections` entries
+- Outbound proxy support via `open-sse/utils/proxyFetch.ts` (env vars) and `open-sse/utils/networkProxy.ts` (configurable per-provider or global)
+
+## 5) Cloud Sync
+
+- Scheduler init: `src/lib/initCloudSync.ts`, `src/shared/services/initializeCloudSync.ts`, `src/shared/services/modelSyncScheduler.ts`
+- Periodic task: `src/shared/services/cloudSyncScheduler.ts`
+- Periodic task: `src/shared/services/modelSyncScheduler.ts`
+- Control route: `src/app/api/sync/cloud/route.ts`
+
+## Request Lifecycle (`/v1/chat/completions`)
```mermaid
sequenceDiagram
@@ -338,7 +363,9 @@ flowchart TD
Q -- No --> R[Return all unavailable]
```
-A tartalék döntéseket az `open-sse/services/accountFallback.ts` vezérli állapotkódok és hibaüzenet-heurisztika használatával. A kombinált útválasztás egy plusz védelmet ad: a szolgáltatói hatókörű 400-asokat, mint például az upstream tartalomblokkolás és a szerepérvényesítési hibák, modellhelyi hibákként kezelik, így a későbbi kombinált célok továbbra is futhatnak.## OAuth Onboarding and Token Refresh Lifecycle
+Fallback decisions are driven by `open-sse/services/accountFallback.ts` using status codes and error-message heuristics. Combo routing adds one extra guard: provider-scoped 400s such as upstream content-block and role-validation failures are treated as model-local failures so later combo targets can still run.
+
+## OAuth Onboarding and Token Refresh Lifecycle
```mermaid
sequenceDiagram
@@ -368,7 +395,9 @@ sequenceDiagram
Test-->>UI: validation result
```
-Az élő forgalom alatti frissítés az `open-sse/handlers/chatCore.ts` fájlban, a `refreshCredentials()` végrehajtón keresztül történik.## Cloud Sync Lifecycle (Enable / Sync / Disable)
+Refresh during live traffic is executed inside `open-sse/handlers/chatCore.ts` via executor `refreshCredentials()`.
+
+## Cloud Sync Lifecycle (Enable / Sync / Disable)
```mermaid
sequenceDiagram
@@ -400,7 +429,9 @@ sequenceDiagram
Sync-->>UI: disabled
```
-Az időszakos szinkronizálást a „CloudSyncScheduler” indítja el, ha a felhő engedélyezve van.## Data Model and Storage Map
+Periodic sync is triggered by `CloudSyncScheduler` when cloud is enabled.
+
+## Data Model and Storage Map
```mermaid
erDiagram
@@ -501,12 +532,14 @@ erDiagram
}
```
-Fizikai tároló fájlok:
+Physical storage files:
-- elsődleges futásidejű DB: `${DATA_DIR}/storage.sqlite`
-- kérésnapló sorai: `${DATA_DIR}/log.txt` (kompat/debug melléktermék)
-- Strukturált hívások hasznosadat-archívuma: `${DATA_DIR}/call_logs/`
-- opcionális fordítói/hibakereső munkamenetek kérése: `/logs/...`## Deployment Topology
+- primary runtime DB: `${DATA_DIR}/storage.sqlite`
+- request log lines: `${DATA_DIR}/log.txt` (compat/debug artifact)
+- structured call payload archives: `${DATA_DIR}/call_logs/`
+- optional translator/request debug sessions: `/logs/...`
+
+## Deployment Topology
```mermaid
flowchart LR
@@ -541,205 +574,249 @@ flowchart LR
### Route and API Modules
-- `src/app/api/v1/*`, `src/app/api/v1beta/*`: kompatibilitási API-k
-- `src/app/api/v1/providers/[szolgáltató]/\*: szolgáltatónként dedikált útvonalak (csevegés, beágyazások, képek)
-- `src/app/api/providers*`: szolgáltató CRUD, érvényesítés, tesztelés
-- `src/app/api/provider-nodes*`: egyéni kompatibilis csomópontkezelés
-- "src/app/api/provider-models": egyéni modellkezelés (CRUD)
-- "src/app/api/models/route.ts": modellkatalógus API (álnevek + egyéni modellek)
-- `src/app/api/oauth/*`: OAuth/eszközkód folyamatok
-- `src/app/api/keys*`: helyi API kulcs életciklusa
-- `src/app/api/models/alias`: alias kezelése
-- `src/app/api/combos*`: tartalék kombinált kezelés
-- "src/app/api/pricing": az árképzés felülbírálása a költségszámításhoz
-- "src/app/api/settings/proxy": proxykonfiguráció (GET/PUT/DELETE)
-- "src/app/api/settings/proxy/test": kimenő proxykapcsolati teszt (POST)
-- `src/app/api/usage/*`: használati és naplózási API-k
-- `src/app/api/sync/*` + `src/app/api/cloud/*`: felhőszinkronizálás és felhő felé néző segítők
-- `src/app/api/cli-tools/*`: helyi CLI konfigurációs írók/ellenőrzők
-- "src/app/api/settings/ip-filter": IP engedélyezési lista/blokkolista (GET/PUT)
-- "src/app/api/settings/thhinking-budget": gondolkodási jogkivonat költségvetési konfigurációja (GET/PUT)
-- "src/app/api/settings/system-prompt": globális rendszerprompt (GET/PUT)
-- "src/app/api/sessions": aktív munkamenet-lista (GET)
-- "src/app/api/rate-limits": fiókonkénti kamatkorlát állapota (GET)### Routing and Execution Core
+- `src/app/api/v1/*`, `src/app/api/v1beta/*`: compatibility APIs
+- `src/app/api/v1/providers/[provider]/*`: dedicated per-provider routes (chat, embeddings, images)
+- `src/app/api/providers*`: provider CRUD, validation, testing
+- `src/app/api/provider-nodes*`: custom compatible node management
+- `src/app/api/provider-models`: custom model management (CRUD)
+- `src/app/api/models/route.ts`: model catalog API (aliases + custom models)
+- `src/app/api/oauth/*`: OAuth/device-code flows
+- `src/app/api/keys*`: local API key lifecycle
+- `src/app/api/models/alias`: alias management
+- `src/app/api/combos*`: fallback combo management
+- `src/app/api/pricing`: pricing overrides for cost calculation
+- `src/app/api/settings/proxy`: proxy configuration (GET/PUT/DELETE)
+- `src/app/api/settings/proxy/test`: outbound proxy connectivity test (POST)
+- `src/app/api/usage/*`: usage and logs APIs
+- `src/app/api/sync/*` + `src/app/api/cloud/*`: cloud sync and cloud-facing helpers
+- `src/app/api/cli-tools/*`: local CLI config writers/checkers
+- `src/app/api/settings/ip-filter`: IP allowlist/blocklist (GET/PUT)
+- `src/app/api/settings/thinking-budget`: thinking token budget config (GET/PUT)
+- `src/app/api/settings/system-prompt`: global system prompt (GET/PUT)
+- `src/app/api/sessions`: active session listing (GET)
+- `src/app/api/rate-limits`: per-account rate limit status (GET)
-- `src/sse/handlers/chat.ts`: kéréselemzés, kombinált kezelés, fiókválasztó hurok
-- "open-sse/handlers/chatCore.ts": fordítás, végrehajtó feladás, újrapróbálkozás/frissítés kezelése, adatfolyam beállítása
-- `open-sse/executors/*`: szolgáltató-specifikus hálózat- és formátumviselkedés### Translation Registry and Format Converters
+### Routing and Execution Core
-- "open-sse/translator/index.ts": fordítói nyilvántartás és hangszerelés
-- Fordítók kérése: `open-sse/translator/request/*`
-- Válasz fordítók: `open-sse/translator/response/*`
-- Formátumkonstansok: `open-sse/translator/formats.ts`### Persistence
+- `src/sse/handlers/chat.ts`: request parse, combo handling, account selection loop
+- `open-sse/handlers/chatCore.ts`: translation, executor dispatch, retry/refresh handling, stream setup
+- `open-sse/executors/*`: provider-specific network and format behavior
-- `src/lib/db/*`: állandó konfiguráció/állapot és tartomány fennmaradása az SQLite-on
-- `src/lib/localDb.ts`: DB modulok kompatibilitási újraexportálása
-- `src/lib/usageDb.ts`: a használati előzmények/hívásnaplók homlokzata az SQLite táblák tetején## Provider Executor Coverage (Strategy Pattern)
+### Translation Registry and Format Converters
-Minden szolgáltató rendelkezik egy speciális végrehajtóval, amely kiterjeszti a „BaseExecutort” (az „open-sse/executors/base.ts” fájlban), amely URL-építést, fejléc-építést, újrapróbálkozást exponenciális visszalépéssel, hitelesítő adatok frissítését és az „execute()” hangszerelési metódust biztosítja.
+- `open-sse/translator/index.ts`: translator registry and orchestration
+- Request translators: `open-sse/translator/request/*`
+- Response translators: `open-sse/translator/response/*`
+- Format constants: `open-sse/translator/formats.ts`
-| Végrehajtó | Szolgáltató(k) | Különleges kezelés |
-| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------------- |
-| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dinamikus URL/fejléc konfiguráció szolgáltatónként |
-| "AntigravityExecutor" | Google Antigravitáció | Egyéni projekt/munkamenet azonosítók, Újrapróbálkozás-elemzés után |
-| "CodexExecutor" | OpenAI Codex | Rendszerutasításokat szúr be, érvelési erőfeszítést kényszerít |
-| "CursorExecutor" | Kurzor IDE | ConnectRPC protokoll, Protobuf kódolás, kérés aláírása ellenőrző összeggel |
-| "GithubExecutor" | GitHub másodpilóta | Másodpilóta token frissítése, VSCode-utánzó fejlécek |
-| "KiroExecutor" | AWS CodeWhisperer/Kiro | AWS EventStream bináris formátum → SSE konverzió |
-| "GeminiCLIExecutor" | Gemini CLI | Google OAuth-token frissítési ciklus |
+### Persistence
-Minden más szolgáltató (beleértve az egyéni kompatibilis csomópontokat is) a "DefaultExecutor"-t használja.## Provider Compatibility Matrix
+- `src/lib/db/*`: persistent config/state and domain persistence on SQLite
+- `src/lib/localDb.ts`: compatibility re-export for DB modules
+- `src/lib/usageDb.ts`: usage history/call logs facade on top of SQLite tables
-| Szolgáltató | Formátum | Auth | Stream | Nem adatfolyam | Token Refresh | Használati API |
-| ------------------ | ---------------- | ------------------------- | ------------------ | -------------- | ------------- | ------------------------- | ------------------------------ |
-| Claude | claude | API kulcs / OAuth | ✅ | ✅ | ✅ | ⚠️ Csak adminisztrátor |
-| Ikrek | ikrek | API kulcs / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
-| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
-| Antigravitáció | antigravitáció | OAuth | ✅ | ✅ | ✅ | ✅ Teljes kvóta API |
-| OpenAI | openai | API kulcs | ✅ | ✅ | ❌ | ❌ |
-| Codex | openai-responses | OAuth | ✅ kényszer | ❌ | ✅ | ✅ Díjkorlátok |
-| GitHub másodpilóta | openai | OAuth + másodpilóta token | ✅ | ✅ | ✅ | ✅ Kvóta pillanatképek |
-| Kurzor | kurzor | Egyéni ellenőrző összeg | ✅ | ✅ | ❌ | ❌ |
-| Kiro | kiro | AWS SSO OIDC | ✅ (Eseményfolyam) | ❌ | ✅ | ✅ Felhasználási korlátok |
-| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Kérésre |
-| Qoder | openai | OAuth (alap) | ✅ | ✅ | ✅ | ⚠️ Kérésre |
-| OpenRouter | openai | API kulcs | ✅ | ✅ | ❌ | ❌ |
-| GLM/Kimi/MiniMax | claude | API kulcs | ✅ | ✅ | ❌ | ❌ |
-| DeepSeek | openai | API kulcs | ✅ | ✅ | ❌ | ❌ |
-| Groq | openai | API kulcs | ✅ | ✅ | ❌ | ❌ |
-| xAI (Grok) | openai | API kulcs | ✅ | ✅ | ❌ | ❌ |
-| Mistral | openai | API kulcs | ✅ | ✅ | ❌ | ❌ |
-| Zavartság | openai | API kulcs | ✅ | ✅ | ❌ | ❌ |
-| Együtt AI | openai | API kulcs | ✅ | ✅ | ❌ | ❌ |
-| Tűzijáték AI | openai | API kulcs | ✅ | ✅ | ❌ | ❌ |
-| Cerebrák | openai | API kulcs | ✅ | ✅ | ❌ | ❌ |
-| Cohere | openai | API kulcs | ✅ | ✅ | ❌ | ❌ |
-| NVIDIA NIM | openai | API kulcs | ✅ | ✅ | ❌ | ❌ | ## Format Translation Coverage |
+## Provider Executor Coverage (Strategy Pattern)
-Az észlelt forrásformátumok a következők:
+Each provider has a specialized executor extending `BaseExecutor` (in `open-sse/executors/base.ts`), which provides URL building, header construction, retry with exponential backoff, credential refresh hooks, and the `execute()` orchestration method.
-- "openai".
-- "Openai-responses".
+| Executor | Provider(s) | Special Handling |
+| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- |
+| `DefaultExecutor` | OpenAI, Claude, Gemini, Qwen, Qoder, OpenRouter, GLM, Kimi, MiniMax, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA | Dynamic URL/header config per provider |
+| `AntigravityExecutor` | Google Antigravity | Custom project/session IDs, Retry-After parsing |
+| `CodexExecutor` | OpenAI Codex | Injects system instructions, forces reasoning effort |
+| `CursorExecutor` | Cursor IDE | ConnectRPC protocol, Protobuf encoding, request signing via checksum |
+| `GithubExecutor` | GitHub Copilot | Copilot token refresh, VSCode-mimicking headers |
+| `KiroExecutor` | AWS CodeWhisperer/Kiro | AWS EventStream binary format → SSE conversion |
+| `GeminiCLIExecutor` | Gemini CLI | Google OAuth token refresh cycle |
+
+All other providers (including custom compatible nodes) use the `DefaultExecutor`.
+
+## Provider Compatibility Matrix
+
+| Provider | Format | Auth | Stream | Non-Stream | Token Refresh | Usage API |
+| ---------------- | ---------------- | --------------------- | ---------------- | ---------- | ------------- | ------------------ |
+| Claude | claude | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Admin only |
+| Gemini | gemini | API Key / OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Gemini CLI | gemini-cli | OAuth | ✅ | ✅ | ✅ | ⚠️ Cloud Console |
+| Antigravity | antigravity | OAuth | ✅ | ✅ | ✅ | ✅ Full quota API |
+| OpenAI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Codex | openai-responses | OAuth | ✅ forced | ❌ | ✅ | ✅ Rate limits |
+| GitHub Copilot | openai | OAuth + Copilot Token | ✅ | ✅ | ✅ | ✅ Quota snapshots |
+| Cursor | cursor | Custom checksum | ✅ | ✅ | ❌ | ❌ |
+| Kiro | kiro | AWS SSO OIDC | ✅ (EventStream) | ❌ | ✅ | ✅ Usage limits |
+| Qwen | openai | OAuth | ✅ | ✅ | ✅ | ⚠️ Per request |
+| Qoder | openai | OAuth (Basic) | ✅ | ✅ | ✅ | ⚠️ Per request |
+| OpenRouter | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| GLM/Kimi/MiniMax | claude | API Key | ✅ | ✅ | ❌ | ❌ |
+| DeepSeek | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Groq | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| xAI (Grok) | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Mistral | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Perplexity | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Together AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Fireworks AI | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cerebras | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| Cohere | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+| NVIDIA NIM | openai | API Key | ✅ | ✅ | ❌ | ❌ |
+
+## Format Translation Coverage
+
+Detected source formats include:
+
+- `openai`
+- `openai-responses`
- `claude`
-- "ikrek".
+- `gemini`
-A célformátumok a következők:
+Target formats include:
-- OpenAI chat/válaszok
+- OpenAI chat/Responses
- Claude
-- Gemini/Gemini-CLI/Antigravitációs boríték
+- Gemini/Gemini-CLI/Antigravity envelope
- Kiro
-- Kurzor
+- Cursor
-A fordítások az**OpenAI-t használják hub-formátumként**– minden konverzió köztesként az OpenAI-n megy keresztül:```
+Translations use **OpenAI as the hub format** — all conversions go through OpenAI as intermediate:
+
+```
Source Format → OpenAI (hub) → Target Format
+```
-````
+Translations are selected dynamically based on source payload shape and provider target format.
-A fordítások kiválasztása dinamikusan történik a forrás hasznos adat alakja és a szolgáltató célformátuma alapján.
+Additional processing layers in the translation pipeline:
-További feldolgozási rétegek a fordítási folyamatban:
+- **Response sanitization** — Strips non-standard fields from OpenAI-format responses (both streaming and non-streaming) to ensure strict SDK compliance
+- **Role normalization** — Converts `developer` → `system` for non-OpenAI targets; merges `system` → `user` for models that reject the system role (GLM, ERNIE)
+- **Think tag extraction** — Parses `...` blocks from content into `reasoning_content` field
+- **Structured output** — Converts OpenAI `response_format.json_schema` to Gemini's `responseMimeType` + `responseSchema`
--**Választisztítás**– Megszünteti a nem szabványos mezőket az OpenAI-formátumú válaszoktól (mind az adatfolyam-, mind a nem adatfolyam-küldéstől) a szigorú SDK-megfelelőség biztosítása érdekében
--**Szerepkör normalizálása**— Átalakítja a "fejlesztő" → "rendszert" nem OpenAI-célokhoz; egyesíti a "rendszer" → "felhasználó" paramétert a rendszerszerepkört elutasító modellekhez (GLM, ERNIE)
--**Think címke kivonatolás**– A tartalomból a `...` blokkokat elemzi a `reasoning_content` mezőbe
--**Strukturált kimenet**- Az OpenAI `response_format.json_schema`-t a Gemini `responseMimeType` + `responseSchema`-jává alakítja## Supported API Endpoints
+## Supported API Endpoints
-| Végpont | Formátum | Kezelő |
-| --------------------------------------------------- | ------------------- | -------------------------------------------------------------------- |
-| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` |
-| `POST /v1/messages` | Claude Üzenetek | Ugyanaz a kezelő (automatikusan észlelve) |
-| `POST /v1/responses` | OpenAI válaszok | `open-sse/handlers/responsesHandler.ts` |
-| `POST /v1/embeddings` | OpenAI beágyazások | `open-sse/handlers/embeddings.ts` |
-| `GET /v1/embeddings` | Modell lista | API útvonal |
-| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` |
-| `GET /v1/images/generations` | Modell lista | API útvonal |
-| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedikált szolgáltatónként modellellenőrzéssel |
-| `POST /v1/providers/{provider}/embeddings` | OpenAI beágyazások | Dedikált szolgáltatónként modellellenőrzéssel |
-| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedikált szolgáltatónként modellellenőrzéssel |
-| `POST /v1/messages/count_tokens` | Claude Token Count | API útvonal |
-| `GET /v1/models` | OpenAI modellek listája | API útvonal (csevegés + beágyazás + kép + egyéni modellek) |
-| `GET /api/models/catalog` | Katalógus | Minden modell szolgáltató + típus szerint csoportosítva |
-| `POST /v1beta/models/*:streamGenerateContent` | Ikrek bennszülött | API route |
-| `GET/PUT/DELETE /api/settings/proxy` | Proxy konfiguráció | Hálózati proxy konfiguráció |
-| `POST /api/settings/proxy/test` | Proxy kapcsolat | Proxy állapot/kapcsolati teszt végpontja |
-| `GET/POST/DELETE /api/provider-models` | Szolgáltatói modellek | A szolgáltatói modell metaadatainak háttere egyéni és felügyelt elérhető modellek |## Bypass Handler
+| Endpoint | Format | Handler |
+| -------------------------------------------------- | ------------------ | ------------------------------------------------------------------- |
+| `POST /v1/chat/completions` | OpenAI Chat | `src/sse/handlers/chat.ts` |
+| `POST /v1/messages` | Claude Messages | Same handler (auto-detected) |
+| `POST /v1/responses` | OpenAI Responses | `open-sse/handlers/responsesHandler.ts` |
+| `POST /v1/embeddings` | OpenAI Embeddings | `open-sse/handlers/embeddings.ts` |
+| `GET /v1/embeddings` | Model listing | API route |
+| `POST /v1/images/generations` | OpenAI Images | `open-sse/handlers/imageGeneration.ts` |
+| `GET /v1/images/generations` | Model listing | API route |
+| `POST /v1/providers/{provider}/chat/completions` | OpenAI Chat | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/embeddings` | OpenAI Embeddings | Dedicated per-provider with model validation |
+| `POST /v1/providers/{provider}/images/generations` | OpenAI Images | Dedicated per-provider with model validation |
+| `POST /v1/messages/count_tokens` | Claude Token Count | API route |
+| `GET /v1/models` | OpenAI Models list | API route (chat + embedding + image + custom models) |
+| `GET /api/models/catalog` | Catalog | All models grouped by provider + type |
+| `POST /v1beta/models/*:streamGenerateContent` | Gemini native | API route |
+| `GET/PUT/DELETE /api/settings/proxy` | Proxy Config | Network proxy configuration |
+| `POST /api/settings/proxy/test` | Proxy Connectivity | Proxy health/connectivity test endpoint |
+| `GET/POST/DELETE /api/provider-models` | Provider Models | Provider model metadata backing custom and managed available models |
-A bypass kezelő (`open-sse/utils/bypassHandler.ts`) elfogja a Claude CLI ismert "kidobási" kéréseit – bemelegítő pingeket, címkivonatokat és tokenszámlálásokat –, és**hamis választ**ad vissza anélkül, hogy felemészti a szolgáltatói tokeneket. Ez csak akkor aktiválódik, ha a „User-Agent” tartalmazza a „claude-cli”-t.## Request Logger Pipeline
+## Bypass Handler
-A kérésnaplózó (`open-sse/utils/requestLogger.ts`) egy 7 szakaszból álló hibakeresési naplózási folyamatot biztosít, amely alapértelmezés szerint le van tiltva, és az `ENABLE_REQUEST_LOGS=true` paraméterrel engedélyezve van:```
+The bypass handler (`open-sse/utils/bypassHandler.ts`) intercepts known "throwaway" requests from Claude CLI — warmup pings, title extractions, and token counts — and returns a **fake response** without consuming upstream provider tokens. This is triggered only when `User-Agent` contains `claude-cli`.
+
+## Request Logger Pipeline
+
+The request logger (`open-sse/utils/requestLogger.ts`) provides a 7-stage debug logging pipeline, disabled by default, enabled via `ENABLE_REQUEST_LOGS=true`:
+
+```
1_req_client.json → 2_req_source.json → 3_req_openai.json → 4_req_target.json
→ 5_res_provider.txt → 6_res_openai.txt → 7_res_client.txt
-````
+```
-A fájlok a `/logs/