From 7c23dab64de929686d2326f6824527ba9c746ca9 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza <8016841+diegosouzapw@users.noreply.github.com> Date: Mon, 29 Jun 2026 08:40:06 -0300 Subject: [PATCH] Release v3.8.40 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit v3.8.40 cycle integration → main. All test gates green (Unit/Integration/Coverage/Node-compat/Quality-Ratchet). The only red check, 'PR Test Policy', is the test-masking heuristic firing on the cumulative ~57-commit release diff (legitimate assert consolidations already reviewed per-PR — Gemini CLI removal #5246, retired GPT models #5280, provider catalog refreshes); overridden with --admin per the documented release-PR convention. CodeQL/SonarQube advisory scans non-blocking; #5278's code already passed CodeQL on main. Homologated on VPS 192.168.0.15 (v3.8.40 healthy). --- .env.example | 19 +- .github/workflows/docker-publish.yml | 12 +- .github/workflows/wiki-sync.yml | 2 +- .mcp.json.example | 17 - @omniroute/opencode-plugin/README.md | 2 +- @omniroute/opencode-plugin/src/index.ts | 11 +- .../opencode-plugin/tests/config-shim.test.ts | 11 +- .../tests/gemini-sanitize.test.ts | 4 +- AGENTS.md | 68 +- CHANGELOG.md | 77 ++ CLAUDE.md | 109 ++- CONTRIBUTING.md | 6 +- Dockerfile | 4 +- README.md | 36 +- bin/_ops-common.sh | 4 +- bin/cli/commands/registry.mjs | 2 - bin/cli/commands/serve.mjs | 10 +- bin/cli/commands/setup-gemini.mjs | 148 ---- bin/cli/commands/setup-qwen.mjs | 19 +- bin/cli/tray/index.mjs | 6 +- bin/cli/tray/traySystray.mjs | 40 +- bin/cold-start-bench.sh | 6 +- bin/nodeRuntimeSupport.mjs | 7 +- bin/restore-data.sh | 4 +- bin/restore-policies.sh | 2 +- bin/rollback.sh | 2 +- bin/snapshot-data.sh | 4 +- config/quality/complexity-baseline.json | 3 +- config/quality/dependency-allowlist.json | 3 +- config/quality/file-size-baseline.json | 26 +- config/quality/quality-baseline.json | 3 +- docs/DOCUMENTATION_OVERHAUL_PLAN.md | 389 ---------- docs/INCIDENT_RESPONSE.md | 174 ----- docs/PERF_BUDGETS.md | 222 ------ docs/README.md | 98 ++- docs/SUBMIT_PR.md | 127 --- docs/THREAT_MODEL.md | 346 --------- docs/architecture/ARCHITECTURE.md | 16 +- docs/architecture/AUTHZ_GUIDE.md | 24 +- docs/architecture/CODEBASE_DOCUMENTATION.md | 48 +- docs/architecture/MONITORING_SECTIONS.md | 62 +- docs/architecture/QUALITY_GATES.md | 56 +- docs/architecture/REPOSITORY_MAP.md | 39 +- docs/architecture/RESILIENCE_GUIDE.md | 40 +- docs/architecture/cluster-decisions.md | 54 +- docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md | 6 +- docs/comparison/meta.json | 5 + ...unified-compression-config-panel-design.md | 240 ------ ...0-unified-compression-config-panel-plan.md | 495 ------------ ...ompression-phase2-named-profiles-design.md | 207 ----- ...-compression-phase2-named-profiles-plan.md | 468 ----------- ...ompression-phase3-request-header-design.md | 229 ------ ...-compression-phase3-request-header-plan.md | 730 ------------------ docs/compression/COMPRESSION_ENGINES.md | 16 +- docs/compression/COMPRESSION_GUIDE.md | 49 +- .../compression/COMPRESSION_LANGUAGE_PACKS.md | 11 +- docs/compression/COMPRESSION_RULES_FORMAT.md | 4 +- docs/compression/CONTEXT_EDITING.md | 25 +- docs/compression/EXTENDING_COMPRESSION.md | 115 +-- docs/compression/RTK_COMPRESSION.md | 99 ++- docs/diagrams/README.md | 26 +- ...bo-9factor.mmd => auto-combo-12factor.mmd} | 25 +- .../diagrams/exported/auto-combo-12factor.svg | 1 + docs/diagrams/exported/auto-combo-9factor.svg | 1 - docs/diagrams/exported/mcp-tools-87.svg | 1 - docs/diagrams/exported/mcp-tools-94.svg | 1 + .../{mcp-tools-87.mmd => mcp-tools-94.mmd} | 9 +- docs/frameworks/A2A-SERVER.md | 20 +- docs/frameworks/ACP.md | 3 +- docs/frameworks/AGENT-SKILLS.md | 140 ++-- docs/frameworks/AGENTBRIDGE.md | 209 ++--- docs/frameworks/AGENT_PROTOCOLS_GUIDE.md | 11 +- docs/frameworks/CLOUD_AGENT.md | 16 +- docs/frameworks/EMBEDDED-SERVICES.md | 2 +- docs/frameworks/EVALS.md | 6 +- docs/frameworks/GAMIFICATION.md | 6 +- docs/frameworks/MCP-SERVER.md | 36 +- docs/frameworks/MEMORY.md | 254 +++--- docs/frameworks/NOTION_CONTEXT.md | 4 +- docs/frameworks/OBSIDIAN_CONTEXT.md | 4 +- docs/frameworks/OPENCODE.md | 4 +- docs/frameworks/OPEN_SSE_ARCHITECTURE.md | 113 +-- docs/frameworks/PLAYGROUND_STUDIO.md | 98 +-- .../{dev/plugins.md => frameworks/PLUGINS.md} | 6 + docs/frameworks/PLUGIN_MARKETPLACE.md | 6 +- docs/{plugins => frameworks}/PLUGIN_SDK.md | 73 +- docs/frameworks/SEARCH_TOOLS_STUDIO.md | 94 +-- docs/frameworks/SKILLS.md | 83 +- docs/frameworks/TRAFFIC_INSPECTOR.md | 227 +++--- docs/frameworks/WEBHOOKS.md | 6 +- docs/frameworks/meta.json | 2 + docs/getting-started/TROUBLESHOOTING.md | 34 +- docs/guides/CLAUDE-CODE-CONFIGURATION.md | 24 +- docs/guides/CLI-INTEGRATIONS.md | 12 +- docs/guides/CODEX-CLI-CONFIGURATION.md | 207 ++--- docs/guides/COST_TRACKING.md | 4 +- docs/guides/DOCKER_GUIDE.md | 4 +- docs/guides/ELECTRON_GUIDE.md | 6 +- docs/guides/FEATURES.md | 12 +- docs/guides/FREE_PROVIDER_RANKINGS.md | 4 +- docs/guides/I18N.md | 4 +- docs/guides/PWA_GUIDE.md | 4 +- docs/guides/REMOTE-MODE.md | 30 +- docs/guides/SETUP_GUIDE.md | 7 +- docs/guides/TERMUX_GUIDE.md | 10 +- docs/{marketing => guides}/TIERS.md | 5 +- docs/guides/TROUBLESHOOTING.md | 34 +- docs/guides/UNINSTALL.md | 4 +- .../{features => guides}/USAGE_QUOTA_GUIDE.md | 153 ++-- docs/guides/USER_GUIDE.md | 80 +- docs/guides/meta.json | 2 + docs/i18n/ar/CHANGELOG.md | 6 + docs/i18n/ar/README.md | 65 +- .../i18n/ar/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/ar/docs/guides/FEATURES.md | 5 +- docs/i18n/ar/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/ar/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/ar/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/ar/llm.txt | 7 +- docs/i18n/az/CHANGELOG.md | 6 + docs/i18n/az/README.md | 65 +- .../i18n/az/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/az/docs/guides/FEATURES.md | 5 +- docs/i18n/az/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/az/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/az/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/az/llm.txt | 7 +- docs/i18n/bg/CHANGELOG.md | 6 + docs/i18n/bg/README.md | 65 +- .../i18n/bg/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/bg/docs/guides/FEATURES.md | 5 +- docs/i18n/bg/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/bg/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/bg/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/bg/llm.txt | 7 +- docs/i18n/bn/CHANGELOG.md | 6 + docs/i18n/bn/README.md | 65 +- .../i18n/bn/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/bn/docs/guides/FEATURES.md | 5 +- docs/i18n/bn/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/bn/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/bn/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/bn/llm.txt | 7 +- docs/i18n/cs/CHANGELOG.md | 6 + docs/i18n/cs/README.md | 65 +- .../i18n/cs/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/cs/docs/guides/FEATURES.md | 5 +- docs/i18n/cs/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/cs/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/cs/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/cs/llm.txt | 7 +- docs/i18n/da/CHANGELOG.md | 6 + docs/i18n/da/README.md | 65 +- .../i18n/da/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/da/docs/guides/FEATURES.md | 5 +- docs/i18n/da/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/da/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/da/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/da/llm.txt | 7 +- docs/i18n/de/CHANGELOG.md | 6 + docs/i18n/de/README.md | 65 +- .../i18n/de/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/de/docs/guides/FEATURES.md | 5 +- docs/i18n/de/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/de/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/de/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/de/llm.txt | 7 +- docs/i18n/es/CHANGELOG.md | 6 + docs/i18n/es/README.md | 65 +- .../i18n/es/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/es/docs/guides/FEATURES.md | 5 +- docs/i18n/es/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/es/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/es/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/es/llm.txt | 7 +- docs/i18n/fa/CHANGELOG.md | 6 + docs/i18n/fa/README.md | 65 +- .../i18n/fa/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/fa/docs/guides/FEATURES.md | 5 +- docs/i18n/fa/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/fa/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/fa/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/fa/llm.txt | 7 +- docs/i18n/fi/CHANGELOG.md | 6 + docs/i18n/fi/README.md | 65 +- .../i18n/fi/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/fi/docs/guides/FEATURES.md | 5 +- docs/i18n/fi/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/fi/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/fi/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/fi/llm.txt | 7 +- docs/i18n/fr/CHANGELOG.md | 6 + docs/i18n/fr/README.md | 65 +- .../i18n/fr/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/fr/docs/guides/FEATURES.md | 5 +- docs/i18n/fr/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/fr/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/fr/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/fr/llm.txt | 7 +- docs/i18n/gu/CHANGELOG.md | 6 + docs/i18n/gu/README.md | 65 +- .../i18n/gu/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/gu/docs/guides/FEATURES.md | 5 +- docs/i18n/gu/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/gu/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/gu/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/gu/llm.txt | 7 +- docs/i18n/he/CHANGELOG.md | 6 + docs/i18n/he/README.md | 65 +- .../i18n/he/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/he/docs/guides/FEATURES.md | 5 +- docs/i18n/he/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/he/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/he/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/he/llm.txt | 7 +- docs/i18n/hi/CHANGELOG.md | 6 + docs/i18n/hi/README.md | 65 +- .../i18n/hi/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/hi/docs/guides/FEATURES.md | 5 +- docs/i18n/hi/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/hi/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/hi/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/hi/llm.txt | 7 +- docs/i18n/hu/CHANGELOG.md | 6 + docs/i18n/hu/README.md | 65 +- .../i18n/hu/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/hu/docs/guides/FEATURES.md | 5 +- docs/i18n/hu/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/hu/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/hu/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/hu/llm.txt | 7 +- docs/i18n/id/CHANGELOG.md | 6 + docs/i18n/id/README.md | 63 +- .../i18n/id/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/id/docs/guides/FEATURES.md | 3 - docs/i18n/id/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/id/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/id/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/id/llm.txt | 7 +- docs/i18n/in/CHANGELOG.md | 6 + docs/i18n/in/README.md | 65 +- .../i18n/in/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/in/docs/guides/FEATURES.md | 5 +- docs/i18n/in/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/in/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/in/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/in/llm.txt | 7 +- docs/i18n/it/CHANGELOG.md | 6 + docs/i18n/it/README.md | 65 +- .../i18n/it/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/it/docs/guides/FEATURES.md | 5 +- docs/i18n/it/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/it/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/it/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/it/llm.txt | 7 +- docs/i18n/ja/CHANGELOG.md | 6 + docs/i18n/ja/README.md | 65 +- .../i18n/ja/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/ja/docs/guides/FEATURES.md | 5 +- docs/i18n/ja/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/ja/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/ja/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/ja/llm.txt | 7 +- docs/i18n/ko/CHANGELOG.md | 6 + docs/i18n/ko/README.md | 65 +- .../i18n/ko/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/ko/docs/guides/FEATURES.md | 5 +- docs/i18n/ko/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/ko/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/ko/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/ko/llm.txt | 7 +- docs/i18n/mr/CHANGELOG.md | 6 + docs/i18n/mr/README.md | 65 +- .../i18n/mr/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/mr/docs/guides/FEATURES.md | 5 +- docs/i18n/mr/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/mr/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/mr/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/mr/llm.txt | 7 +- docs/i18n/ms/CHANGELOG.md | 6 + docs/i18n/ms/README.md | 65 +- .../i18n/ms/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/ms/docs/guides/FEATURES.md | 5 +- docs/i18n/ms/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/ms/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/ms/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/ms/llm.txt | 7 +- docs/i18n/nl/CHANGELOG.md | 6 + docs/i18n/nl/README.md | 65 +- .../i18n/nl/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/nl/docs/guides/FEATURES.md | 5 +- docs/i18n/nl/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/nl/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/nl/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/nl/llm.txt | 7 +- docs/i18n/no/CHANGELOG.md | 6 + docs/i18n/no/README.md | 65 +- .../i18n/no/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/no/docs/guides/FEATURES.md | 5 +- docs/i18n/no/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/no/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/no/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/no/llm.txt | 7 +- docs/i18n/phi/CHANGELOG.md | 6 + docs/i18n/phi/README.md | 65 +- .../phi/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/phi/docs/guides/FEATURES.md | 5 +- docs/i18n/phi/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/phi/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/phi/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/phi/llm.txt | 7 +- docs/i18n/pl/CHANGELOG.md | 6 + docs/i18n/pl/README.md | 65 +- .../i18n/pl/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/pl/docs/guides/FEATURES.md | 5 +- docs/i18n/pl/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/pl/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/pl/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/pl/llm.txt | 7 +- docs/i18n/pt-BR/CHANGELOG.md | 6 + docs/i18n/pt-BR/README.md | 65 +- .../pt-BR/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/pt-BR/docs/guides/FEATURES.md | 5 +- .../i18n/pt-BR/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/pt-BR/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/pt-BR/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/pt-BR/llm.txt | 7 +- docs/i18n/pt/CHANGELOG.md | 6 + docs/i18n/pt/README.md | 65 +- .../i18n/pt/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/pt/docs/guides/FEATURES.md | 5 +- docs/i18n/pt/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/pt/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/pt/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/pt/llm.txt | 7 +- docs/i18n/ro/CHANGELOG.md | 6 + docs/i18n/ro/README.md | 65 +- .../i18n/ro/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/ro/docs/guides/FEATURES.md | 5 +- docs/i18n/ro/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/ro/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/ro/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/ro/llm.txt | 7 +- docs/i18n/ru/CHANGELOG.md | 6 + docs/i18n/ru/README.md | 65 +- .../i18n/ru/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/ru/docs/guides/FEATURES.md | 5 +- docs/i18n/ru/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/ru/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/ru/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/ru/llm.txt | 7 +- docs/i18n/sk/CHANGELOG.md | 6 + docs/i18n/sk/README.md | 65 +- .../i18n/sk/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/sk/docs/guides/FEATURES.md | 5 +- docs/i18n/sk/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/sk/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/sk/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/sk/llm.txt | 7 +- docs/i18n/sv/CHANGELOG.md | 6 + docs/i18n/sv/README.md | 65 +- .../i18n/sv/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/sv/docs/guides/FEATURES.md | 5 +- docs/i18n/sv/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/sv/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/sv/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/sv/llm.txt | 7 +- docs/i18n/sw/CHANGELOG.md | 6 + docs/i18n/sw/README.md | 65 +- .../i18n/sw/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/sw/docs/guides/FEATURES.md | 5 +- docs/i18n/sw/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/sw/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/sw/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/sw/llm.txt | 7 +- docs/i18n/ta/CHANGELOG.md | 6 + docs/i18n/ta/README.md | 65 +- .../i18n/ta/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/ta/docs/guides/FEATURES.md | 5 +- docs/i18n/ta/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/ta/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/ta/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/ta/llm.txt | 7 +- docs/i18n/te/CHANGELOG.md | 6 + docs/i18n/te/README.md | 65 +- .../i18n/te/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/te/docs/guides/FEATURES.md | 5 +- docs/i18n/te/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/te/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/te/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/te/llm.txt | 7 +- docs/i18n/th/CHANGELOG.md | 6 + docs/i18n/th/README.md | 65 +- .../i18n/th/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/th/docs/guides/FEATURES.md | 5 +- docs/i18n/th/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/th/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/th/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/th/llm.txt | 7 +- docs/i18n/tr/CHANGELOG.md | 6 + docs/i18n/tr/README.md | 65 +- .../i18n/tr/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/tr/docs/guides/FEATURES.md | 5 +- docs/i18n/tr/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/tr/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/tr/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/tr/llm.txt | 7 +- docs/i18n/uk-UA/CHANGELOG.md | 6 + docs/i18n/uk-UA/README.md | 65 +- .../uk-UA/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/uk-UA/docs/guides/FEATURES.md | 5 +- .../i18n/uk-UA/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/uk-UA/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/uk-UA/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/uk-UA/llm.txt | 7 +- docs/i18n/ur/CHANGELOG.md | 6 + docs/i18n/ur/README.md | 65 +- .../i18n/ur/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/ur/docs/guides/FEATURES.md | 5 +- docs/i18n/ur/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/ur/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/ur/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/ur/llm.txt | 7 +- docs/i18n/vi/CHANGELOG.md | 6 + docs/i18n/vi/README.md | 65 +- .../i18n/vi/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/vi/docs/guides/FEATURES.md | 5 +- docs/i18n/vi/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/vi/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/vi/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/vi/llm.txt | 7 +- docs/i18n/zh-CN/CHANGELOG.md | 6 + docs/i18n/zh-CN/README.md | 3 - .../zh-CN/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/zh-CN/docs/guides/FEATURES.md | 5 +- .../i18n/zh-CN/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/zh-CN/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/zh-CN/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/zh-CN/llm.txt | 7 +- docs/i18n/zh-TW/CHANGELOG.md | 6 + docs/i18n/zh-TW/README.md | 3 - .../zh-TW/docs/architecture/ARCHITECTURE.md | 4 +- .../architecture/CODEBASE_DOCUMENTATION.md | 4 - docs/i18n/zh-TW/docs/guides/FEATURES.md | 5 +- .../i18n/zh-TW/docs/guides/TROUBLESHOOTING.md | 3 +- docs/i18n/zh-TW/docs/guides/USER_GUIDE.md | 25 +- docs/i18n/zh-TW/docs/reference/ENVIRONMENT.md | 4 - docs/i18n/zh-TW/llm.txt | 7 +- docs/meta.json | 2 + docs/openapi.yaml | 2 +- docs/ops/COVERAGE_PLAN.md | 70 +- docs/ops/DATABASE_GUIDE.md | 147 ++-- docs/ops/DOCUMENTATION_AUDIT_REPORT.md | 241 ------ docs/ops/E2E_DASHBOARD_SHAKEDOWN_v3.8.0.md | 350 --------- docs/ops/FLY_IO_DEPLOYMENT_GUIDE.md | 4 +- docs/ops/MONITORING_GUIDE.md | 171 ++-- docs/ops/MUTATION_GATE_SPIKE_VERDICT.md | 51 -- docs/ops/PROXY_GUIDE.md | 63 +- docs/ops/RELEASE_CHECKLIST.md | 26 +- docs/ops/TUNNELS_GUIDE.md | 6 +- docs/ops/VM_DEPLOYMENT_GUIDE.md | 4 +- docs/ops/meta.json | 4 +- docs/providers/AGENTROUTER.md | 10 +- .../{PROVIDERS.md => providers/CLAUDE_WEB.md} | 22 +- docs/providers/ZED-DOCKER.md | 6 + docs/providers/meta.json | 5 + docs/reference/API_REFERENCE.md | 288 ++++--- docs/reference/CLI-TOOLS.md | 141 ++-- docs/reference/ENVIRONMENT.md | 442 ++++++----- docs/reference/FEATURE_FLAGS.md | 24 +- docs/reference/FREE_TIERS.md | 64 +- docs/reference/PROVIDER_REFERENCE.md | 523 ++++++------- docs/releases/v3.8.0.md | 140 ---- docs/research/DISCOVERY_TOOL_DESIGN.md | 142 ---- docs/research/UNLIMITED_LLM_ACCESS.md | 322 -------- docs/routing/AUTO-COMBO.md | 182 ++--- docs/routing/QUOTA_SHARE.md | 78 +- docs/routing/REASONING_REPLAY.md | 8 +- docs/security/CLI_TOKEN.md | 18 + docs/security/CLI_TOKEN_AUTH.md | 47 -- docs/security/COMPLIANCE.md | 6 +- docs/security/EGRESS_POLICY.md | 4 +- docs/security/ERROR_SANITIZATION.md | 8 +- docs/security/GUARDRAILS.md | 28 +- docs/security/MITM-TPROXY-DECRYPT.md | 116 +-- docs/security/PUBLIC_CREDS.md | 7 +- docs/security/STEALTH_GUIDE.md | 29 +- docs/security/meta.json | 3 +- electron/package-lock.json | 4 +- electron/package.json | 2 +- images/tier-flow-dark.svg | 2 +- images/tier-flow-light.svg | 2 +- llm.txt | 7 +- open-sse/config/anthropicHeaders.ts | 12 +- open-sse/config/cliFingerprints.ts | 13 - open-sse/config/freeModelCatalog.data.ts | 9 - open-sse/config/freeTierCatalog.ts | 5 +- open-sse/config/glmProvider.ts | 2 +- open-sse/config/providerHeaderProfiles.ts | 12 +- open-sse/config/providers/index.ts | 4 +- .../providers/registry/chatgpt-web/index.ts | 4 - .../config/providers/registry/codex/index.ts | 1 - .../providers/registry/command-code/index.ts | 20 +- .../registry/copilot-m365-web/index.ts | 12 + .../providers/registry/gemini/cli/index.ts | 34 - .../providers/registry/grok-cli/index.ts | 14 +- .../providers/registry/kilocode/index.ts | 9 + open-sse/config/providers/shared.ts | 9 + open-sse/executors/adapta-web.ts | 36 +- open-sse/executors/antigravity.ts | 2 +- open-sse/executors/blackbox-web.ts | 36 +- open-sse/executors/chatgpt-web.ts | 19 +- open-sse/executors/chipotle.ts | 43 +- open-sse/executors/claude-web.ts | 4 +- open-sse/executors/claudeIdentity.ts | 2 +- open-sse/executors/codex.ts | 4 +- open-sse/executors/commandCode.ts | 51 +- open-sse/executors/copilot-m365-connection.ts | 55 +- open-sse/executors/copilot-m365-frames.ts | 31 +- open-sse/executors/copilot-m365-web.ts | 308 ++++++++ open-sse/executors/copilot-web.ts | 2 +- open-sse/executors/deepseek-web.ts | 2 +- open-sse/executors/default.ts | 2 +- open-sse/executors/doubao-web.ts | 18 +- open-sse/executors/duckduckgo-web.ts | 4 +- open-sse/executors/gemini-business.ts | 27 +- open-sse/executors/gemini-cli.ts | 587 -------------- open-sse/executors/gemini-web.ts | 2 +- open-sse/executors/grok-cli.ts | 13 +- open-sse/executors/grok-web.ts | 12 +- open-sse/executors/huggingchat.ts | 63 +- open-sse/executors/index.ts | 7 +- open-sse/executors/inner-ai.ts | 31 +- open-sse/executors/kimi-web.ts | 18 +- open-sse/executors/lmarena.ts | 21 +- open-sse/executors/mimocode.ts | 11 +- open-sse/executors/muse-spark-web.ts | 30 +- open-sse/executors/perplexity-web.ts | 39 +- open-sse/executors/phind.ts | 5 +- open-sse/executors/poe-web.ts | 2 +- open-sse/executors/qwen-web.ts | 2 +- open-sse/executors/t3-chat-web.ts | 182 +++-- open-sse/executors/theoldllm.ts | 58 +- open-sse/executors/trae.ts | 2 +- open-sse/executors/v0-vercel-web.ts | 2 +- open-sse/executors/venice-web.ts | 2 +- open-sse/executors/veoaifree-web.ts | 9 +- open-sse/handlers/chatCore.ts | 88 ++- open-sse/handlers/chatCore/attemptLogging.ts | 3 + .../chatCore/compressionAnalyticsWrite.ts | 36 + open-sse/handlers/responseSanitizer.ts | 112 ++- open-sse/handlers/responseTranslator.ts | 6 +- open-sse/mcp-server/README.md | 2 +- .../__tests__/essentialTools.test.ts | 4 +- .../__tests__/httpAuthContext.test.ts | 143 ++++ .../__tests__/toolSearch.catalog.test.ts | 27 + .../__tests__/toolSearch.search.test.ts | 26 + .../__tests__/toolSearch.signature.test.ts | 26 + .../__tests__/toolSearch.tool.test.ts | 39 + open-sse/mcp-server/httpAuthContext.ts | 42 + open-sse/mcp-server/httpTransport.ts | 11 +- open-sse/mcp-server/schemas/toolDefinition.ts | 31 + open-sse/mcp-server/schemas/toolSearch.ts | 37 + open-sse/mcp-server/schemas/tools.ts | 36 +- open-sse/mcp-server/server.ts | 73 +- open-sse/mcp-server/toolSearch/catalog.ts | 98 +++ open-sse/mcp-server/toolSearch/handler.ts | 18 + open-sse/mcp-server/toolSearch/index.ts | 5 + open-sse/mcp-server/toolSearch/register.ts | 36 + open-sse/mcp-server/toolSearch/search.ts | 96 +++ open-sse/mcp-server/toolSearch/signature.ts | 92 +++ open-sse/mcp-server/tools/advancedTools.ts | 4 +- open-sse/package.json | 2 +- open-sse/services/AGENTS.md | 2 +- .../services/__tests__/tierResolver.test.ts | 1 - open-sse/services/antigravityHeaderScrub.ts | 2 +- open-sse/services/antigravityHeaders.ts | 4 +- open-sse/services/autoRefreshDaemon.ts | 2 +- open-sse/services/browserBackedChat.ts | 4 +- open-sse/services/browserPool.ts | 4 +- open-sse/services/ccBridgeTransforms.ts | 4 +- open-sse/services/claudeCodeCompatible.ts | 4 +- open-sse/services/claudeTlsClient.ts | 2 +- open-sse/services/claudeTurnstileSolver.ts | 2 +- open-sse/services/cloudCodeThinking.ts | 2 +- open-sse/services/combo.ts | 29 +- .../combo/__tests__/targetExhaustion.test.ts | 311 ++++++++ .../services/compression/cacheAwareConfig.ts | 23 + open-sse/services/compression/diffHelper.ts | 101 ++- .../services/compression/engineCatalog.ts | 7 + .../compression/engines/headroom/index.ts | 6 +- .../services/compression/engines/index.ts | 2 + .../engines/relevance/configSchema.ts | 63 ++ .../compression/engines/relevance/index.ts | 185 +++++ .../compression/engines/relevance/scorer.ts | 85 ++ .../engines/rtk/commandDetector.ts | 7 +- .../compression/engines/rtk/configSchema.ts | 117 +++ .../services/compression/engines/rtk/index.ts | 127 +-- .../engines/rtk/renderers/gitDiff.ts | 39 + .../engines/rtk/renderers/index.ts | 50 ++ .../engines/rtk/renderers/structuredTable.ts | 96 +++ .../engines/rtk/renderers/terraformPlan.ts | 40 + .../engines/rtk/renderers/testGreen.ts | 70 ++ .../engines/rtk/renderers/types.ts | 13 + .../engines/rtk/splitCompositeCommand.ts | 111 +++ .../services/compression/entrypointWrap.ts | 43 ++ open-sse/services/compression/hardBudget.ts | 196 +++++ .../services/compression/planResolution.ts | 24 +- open-sse/services/compression/preservation.ts | Bin 6610 -> 7665 bytes .../services/compression/quantumLock/index.ts | 17 + .../compression/quantumLock/quantumLock.ts | 58 ++ .../quantumLock/quantumLockStep.ts | 72 ++ .../quantumLock/quantumPatterns.ts | 80 ++ .../compression/quantumLock/strategyWrap.ts | 72 ++ open-sse/services/compression/resultMemo.ts | 83 ++ .../services/compression/riskGate/index.ts | 3 + .../services/compression/riskGate/riskGate.ts | 136 ++++ .../compression/riskGate/riskGateStep.ts | 115 +++ .../compression/riskGate/riskPatterns.ts | 61 ++ .../compression/riskGate/strategyWrap.ts | 44 ++ .../services/compression/strategySelector.ts | 201 ++++- open-sse/services/compression/types.ts | 37 + open-sse/services/contextManager.ts | 30 +- open-sse/services/freeWebSearch.ts | 2 +- open-sse/services/geminiCliHeaders.ts | 43 -- open-sse/services/grokTlsClient.ts | 2 +- open-sse/services/model.ts | 4 - open-sse/services/opencodeOllamaUsage.ts | 4 +- open-sse/services/payloadRules.ts | 2 +- open-sse/services/provider.ts | 2 +- open-sse/services/qoderCli.ts | 14 +- .../sessionPool/fingerprintRotator.ts | 30 +- open-sse/services/tierConfig.ts | 1 - open-sse/services/tierDefaults.json | 1 - open-sse/services/tokenRefresh.ts | 96 ++- open-sse/services/usage.ts | 194 +---- open-sse/services/usage/antigravity.ts | 2 +- open-sse/transformer/responsesTransformer.ts | 57 +- open-sse/translator/formats.ts | 1 - open-sse/translator/index.ts | 4 +- .../translator/request/claude-to-gemini.ts | 2 +- .../translator/request/gemini-to-openai.ts | 1 - .../translator/request/openai-responses.ts | 9 +- .../translator/request/openai-to-gemini.ts | 180 ++--- open-sse/translator/request/openai-to-kiro.ts | 8 +- .../translator/response/gemini-to-claude.ts | 1 - .../translator/response/gemini-to-openai.ts | 34 +- .../translator/response/kiro-to-openai.ts | 3 +- .../translator/response/openai-responses.ts | 58 +- .../response/openai-to-gemini-sse.ts | 15 +- open-sse/types.d.ts | 2 +- open-sse/utils/aiSdkCompat.ts | 27 + open-sse/utils/cursorVersionDetector.ts | 2 +- open-sse/utils/proxyFallback.ts | 51 +- open-sse/utils/publicCreds.ts | 10 +- open-sse/utils/stream.ts | 4 +- open-sse/utils/streamPayloadCollector.ts | 1 - open-sse/utils/usageTracking.ts | 2 +- package-lock.json | 186 ++--- package.json | 10 +- public/providers/gemini-cli.svg | 1 - scripts/ad-hoc/diag-trae-auth.mjs | 2 +- scripts/ad-hoc/resolve_all_conflicts.js | 24 - scripts/build/bootstrap-env.mjs | 4 - scripts/build/pack-artifact-policy.ts | 11 +- scripts/check/check-docs-symbols.mjs | 14 +- scripts/check/check-fabricated-docs.mjs | 5 - scripts/check/check-known-symbols.ts | 41 +- scripts/check/check-test-discovery.mjs | 1 + scripts/ci/should-promote-latest.sh | 44 ++ scripts/dev/responses-ws-proxy.mjs | 2 +- scripts/dev/system-info.mjs | 1 - scripts/docs/sync-wiki.mjs | 58 +- scripts/i18n/i18n_autotranslate.py | 2 +- scripts/quality/mutation-radiography.mjs | 7 +- skills/cli-serve/SKILL.md | 3 +- .../(dashboard)/dashboard/acp-agents/page.tsx | 1 - .../analytics/CompressionAnalyticsTab.tsx | 15 +- .../cli-code/components/CodexToolCard.tsx | 2 - src/app/(dashboard)/dashboard/combos/page.tsx | 4 +- .../studio/CompressionAnnotation.tsx | 36 + .../compression/studio/CompressionCockpit.tsx | 12 + .../dashboard/compression/studio/PlayView.tsx | 33 +- .../compression/studio/PlaygroundInput.tsx | 16 +- .../compression/studio/QuantumLockBadge.tsx | 19 + .../compression/studio/RiskGateBadge.tsx | 15 + .../compression/studio/SaliencyHeatmap.tsx | 45 ++ .../studio/compressionFlowModel.ts | 27 + .../(dashboard)/dashboard/context/page.tsx | 36 + .../dashboard/onboarding/steps/TierTour.tsx | 2 +- .../[id]/ProviderDetailPageClient.tsx | 16 - .../[id]/components/ConnectionRow.tsx | 53 +- .../components/ConnectionsHeaderToolbar.tsx | 12 - .../[id]/components/ConnectionsListPanel.tsx | 90 +-- .../[id]/components/CustomModelsSection.tsx | 2 - .../EmptyConnectionsPlaceholder.tsx | 11 - .../[id]/components/ProviderModalsPanel.tsx | 33 - .../components/modals/EditConnectionModal.tsx | 13 +- .../modals/ImportGeminiAuthModal.tsx | 706 ----------------- .../__tests__/authImportModals.test.tsx | 8 +- .../[id]/hooks/useAuthFileHandlers.ts | 93 +-- .../providers/[id]/providerPageHelpers.ts | 1 - .../components/NoAuthProvidersSection.tsx | 159 ++++ .../onboarding/providerOnboardingCatalog.ts | 1 - .../(dashboard)/dashboard/providers/page.tsx | 112 +-- .../usage/components/ProviderLimits/index.tsx | 2 - .../ProviderLimits/providerColumns.ts | 2 +- .../components/ProviderLimits/quotaParsing.ts | 7 - .../guide-settings/[toolId]/route.ts | 2 +- src/app/api/compression/preview/route.ts | 69 +- .../api/internal/codex-responses-ws/route.ts | 2 +- .../api/oauth/[provider]/[action]/route.ts | 2 +- .../[id]/gemini-cli-auth/apply-local/route.ts | 76 -- .../[id]/gemini-cli-auth/export/route.ts | 42 - src/app/api/providers/[id]/models/route.ts | 91 +-- src/app/api/providers/[id]/test/route.ts | 7 - .../providers/agy-auth/zip-extract/route.ts | 6 +- .../gemini-cli-auth/import-bulk/route.ts | 114 --- .../providers/gemini-cli-auth/import/route.ts | 99 --- .../gemini-cli-auth/zip-extract/route.ts | 52 -- src/app/api/settings/authz-inventory/route.ts | 7 +- src/app/api/v1/messages/count_tokens/route.ts | 2 +- src/app/api/v1/models/catalog.ts | 5 +- .../v1/vscode/[token]/modelPresentation.ts | 3 +- .../vscode/raw/[token]/modelPresentation.ts | 3 +- src/app/docs/[...slug]/page.tsx | 13 +- src/domain/providerExpiration.ts | 2 +- src/hooks/usePreviewCompression.ts | 15 +- src/i18n/messages/ar.json | 53 +- src/i18n/messages/az.json | 51 +- src/i18n/messages/bg.json | 51 +- src/i18n/messages/bn.json | 51 +- src/i18n/messages/cs.json | 51 +- src/i18n/messages/da.json | 51 +- src/i18n/messages/de.json | 51 +- src/i18n/messages/en.json | 52 +- src/i18n/messages/es.json | 51 +- src/i18n/messages/fa.json | 47 +- src/i18n/messages/fi.json | 51 +- src/i18n/messages/fr.json | 51 +- src/i18n/messages/gu.json | 51 +- src/i18n/messages/he.json | 51 +- src/i18n/messages/hi.json | 51 +- src/i18n/messages/hu.json | 51 +- src/i18n/messages/id.json | 51 +- src/i18n/messages/in.json | 51 +- src/i18n/messages/it.json | 43 +- src/i18n/messages/ja.json | 49 +- src/i18n/messages/ko.json | 49 +- src/i18n/messages/mr.json | 49 +- src/i18n/messages/ms.json | 49 +- src/i18n/messages/nl.json | 49 +- src/i18n/messages/no.json | 49 +- src/i18n/messages/phi.json | 49 +- src/i18n/messages/pl.json | 49 +- src/i18n/messages/pt-BR.json | 47 +- src/i18n/messages/pt.json | 49 +- src/i18n/messages/ro.json | 49 +- src/i18n/messages/ru.json | 47 +- src/i18n/messages/sk.json | 49 +- src/i18n/messages/sv.json | 49 +- src/i18n/messages/sw.json | 51 +- src/i18n/messages/ta.json | 51 +- src/i18n/messages/te.json | 51 +- src/i18n/messages/th.json | 51 +- src/i18n/messages/tr.json | 55 +- src/i18n/messages/uk-UA.json | 40 +- src/i18n/messages/ur.json | 51 +- src/i18n/messages/vi.json | 51 +- src/i18n/messages/zh-CN.json | 46 +- src/lib/acp/registry.ts | 9 - src/lib/cloudflaredTunnel.ts | 6 +- src/lib/copilot/engine.ts | 2 +- src/lib/copilot/systemPrompt.ts | 2 +- src/lib/db/compressionAnalytics.ts | 94 ++- src/lib/db/core.ts | 3 +- .../109_call_logs_correlation_id.sql | 2 + src/lib/db/schemaColumns.ts | 5 + src/lib/docsI18nPath.ts | 33 + src/lib/docsSanitizer.ts | 42 + src/lib/modelAliasSeed.ts | 17 +- src/lib/modelsDevSync.ts | 2 +- src/lib/oauth/constants/oauth.ts | 31 +- src/lib/oauth/credentialBlob.ts | 2 +- src/lib/oauth/pasteCredentials.ts | 2 +- src/lib/oauth/providers.ts | 19 +- src/lib/oauth/providers/gemini.ts | 99 --- src/lib/oauth/providers/grok-cli.ts | 59 +- src/lib/oauth/providers/index.ts | 2 - src/lib/oauth/providers/windsurf.ts | 2 +- src/lib/oauth/services/gemini.ts | 237 ------ src/lib/oauth/services/index.ts | 1 - src/lib/oauth/utils/agyAuthImport.ts | 8 +- src/lib/oauth/utils/cliProxyAuthImport.ts | 7 +- src/lib/oauth/utils/geminiAuthFile.ts | 412 ---------- src/lib/oauth/utils/geminiAuthImport.ts | 234 ------ ...iniAuthZipExtract.ts => jsonZipExtract.ts} | 2 +- src/lib/pricingSync.ts | 8 +- src/lib/providers/validation/headers.ts | 8 +- src/lib/providers/validation/metaAi.ts | 5 +- src/lib/providers/validation/openaiFormat.ts | 31 +- src/lib/providers/validation/urlHelpers.ts | 2 +- src/lib/providers/validation/webProvidersA.ts | 17 +- src/lib/providers/validation/webProvidersB.ts | 4 +- src/lib/proxyHealth.ts | 10 +- src/lib/services/installers/utils.ts | 2 +- src/lib/system/versionCheck.ts | 61 +- src/lib/tokenHealthCheck.ts | 4 +- src/lib/usage/callLogs.ts | 14 +- src/lib/usage/fetcher.ts | 33 - src/lib/usage/providerLimits.ts | 2 +- src/lib/usage/usageHistory.ts | 5 + src/mitm/handlers/antigravity.ts | 40 +- src/proxy.ts | 2 + src/server/authz/classify.ts | 13 + src/shared/components/OAuthModal.tsx | 12 +- src/shared/components/ProviderIcon.tsx | 1 - src/shared/components/lobeProviderIcons.ts | 36 +- src/shared/constants/cliCompatProviders.ts | 6 - src/shared/constants/cliTools.ts | 53 +- src/shared/constants/colors.ts | 1 - src/shared/constants/pricing.ts | 3 +- .../constants/pricing/oauth-subscriptions.ts | 60 +- src/shared/constants/providers.ts | 1 - src/shared/constants/providers/oauth.ts | 16 +- src/shared/constants/providers/web-cookie.ts | 13 + src/shared/providers/webSessionCredentials.ts | 7 + src/shared/schemas/cliCatalog.ts | 2 +- src/shared/services/cliRuntime.ts | 11 - src/shared/utils/noAuthProviders.ts | 24 + src/shared/utils/nodeRuntimeSupport.ts | 7 +- src/shared/validation/schemas/auth.ts | 34 +- src/shared/validation/schemas/provider.ts | 2 +- src/sse/handlers/chat.ts | 92 ++- src/sse/handlers/chatHelpers.ts | 19 + src/sse/services/auth.ts | 98 ++- src/sse/services/model.ts | 5 +- tests/integration/chat-pipeline.test.ts | 79 +- tests/snapshots/provider/translate-path.json | 121 +-- tests/unit/antigravity-client-profile.test.ts | 22 +- tests/unit/api/authz-inventory.test.ts | 7 + .../auth-antigravity-account-retry-v2.test.ts | 164 ++++ tests/unit/authz/classify.test.ts | 19 + tests/unit/authz/pipeline.test.ts | 32 + tests/unit/authz/proxy-contract.test.ts | 1 + .../build/should-promote-latest-5301.test.ts | 69 ++ tests/unit/chat-route-coverage.test.ts | 25 +- .../chatcore-memory-skills-injection.test.ts | 7 +- tests/unit/chatcore-skills-format.test.ts | 2 +- tests/unit/chatgpt-web.test.ts | 23 +- tests/unit/check-docs-symbols.test.ts | 13 +- tests/unit/check-known-symbols.test.ts | 15 +- ...claude-codex-identity-version-sync.test.ts | 4 +- tests/unit/cli-catalog-acpspawnable.test.ts | 1 - tests/unit/cli-catalog-counts.test.ts | 49 +- tests/unit/cli-catalog-newentries.test.ts | 14 +- tests/unit/cli-plugin-system.test.ts | 6 +- .../cli-runtime-imports-packaged-5227.test.ts | 86 +++ tests/unit/cli-tools-schema.test.ts | 38 +- tests/unit/cli-tools.test.ts | 38 +- tests/unit/cli/setup-gemini.test.ts | 24 - tests/unit/cliproxy-auth-import-1934.test.ts | 19 +- .../combo-body-specific-400-stop-4279.test.ts | 32 +- .../combo-rr-session-stickiness-3825.test.ts | 116 +++ tests/unit/combo-strategies.test.ts | 36 + tests/unit/command-code-executor.test.ts | 36 +- ...mmand-code-maxtokens-negative-5166.test.ts | 76 ++ .../compression/compressionAnalytics.test.ts | 63 ++ .../compression/compressionAnnotation.test.ts | 90 +++ tests/unit/compression/hard-budget.test.ts | 359 +++++++++ tests/unit/compression/heatmap-build.test.ts | 99 +++ .../compression/quantumLockDetect.test.ts | 97 +++ .../quantumLockIntegration.test.ts | 50 ++ .../unit/compression/quantumLockStep.test.ts | 121 +++ .../unit/compression/relevance-engine.test.ts | 361 +++++++++ .../unit/compression/relevance-scorer.test.ts | 70 ++ tests/unit/compression/result-memo.test.ts | 343 ++++++++ tests/unit/compression/riskGateDetect.test.ts | 105 +++ .../compression/riskGateIntegration.test.ts | 78 ++ tests/unit/compression/riskGateStep.test.ts | 61 ++ .../compression/rtk-command-detector.test.ts | 11 + .../compression/rtk-render-gitdiff.test.ts | 36 + .../unit/compression/rtk-render-table.test.ts | 32 + .../compression/rtk-render-terraform.test.ts | 37 + .../compression/rtk-render-testgreen.test.ts | 40 + .../rtk-renderers-integration.test.ts | 43 ++ .../compression/splitCompositeCommand.test.ts | 64 ++ tests/unit/context-manager.test.ts | 32 + tests/unit/copilot-m365-web-executor.test.ts | 158 ++++ .../context-parent-redirect-5298.test.ts | 23 + tests/unit/dockerfile-build-heap-4076.test.ts | 20 +- tests/unit/dockerignore-docs-coverage.test.ts | 2 +- tests/unit/error-classification.test.ts | 1 - tests/unit/executor-antigravity.test.ts | 2 +- tests/unit/executor-codex.test.ts | 20 +- tests/unit/executor-gemini-cli.test.ts | 527 ------------- tests/unit/executor-github.test.ts | 31 +- tests/unit/free-tier-catalog.test.ts | 2 +- tests/unit/gemini-import-route.test.ts | 356 --------- .../unit/gemini-midstream-error-4177.test.ts | 2 +- tests/unit/gemini-usage-projectid.test.ts | 145 ---- tests/unit/geminiAuthFile.test.ts | 479 ------------ tests/unit/geminiAuthImport.test.ts | 455 ----------- tests/unit/glm-executor.test.ts | 38 +- tests/unit/grok-cli-oauth.test.ts | 52 ++ tests/unit/grok-cli-strip-params.test.ts | 53 ++ tests/unit/guide-settings-route.test.ts | 4 +- .../kilocode-anonymous-fallback-4019.test.ts | 70 ++ tests/unit/kiro-continue-filler-5231.test.ts | 64 ++ tests/unit/m365-bizchat-frames-4042.test.ts | 4 +- tests/unit/m365-connection-4042.test.ts | 1 + .../unit/mcp/tool-definition-reexport.test.ts | 23 + tests/unit/mitm-handler-antigravity.test.ts | 62 ++ tests/unit/model-alias-seed.test.ts | 14 +- tests/unit/modelsDevSync.test.ts | 4 +- .../noauth-blocked-partition-5183.test.ts | 69 ++ tests/unit/node-runtime-support.test.ts | 14 +- tests/unit/oauth-providers-config.test.ts | 81 +- .../unit/oauth-redirect-uri-mismatch.test.ts | 185 +---- tests/unit/ops-scripts.test.ts | 4 +- tests/unit/pack-artifact-policy.test.ts | 1 + tests/unit/pricing-sync.test.ts | 4 +- tests/unit/provider-health-autopilot.test.ts | 20 +- tests/unit/provider-models-route.test.ts | 97 --- .../provider-request-failure-pipeline.test.ts | 177 ++++- .../provider-target-format-badge-4475.test.ts | 3 +- .../unit/provider-validation-branches.test.ts | 46 -- tests/unit/providers-page-utils.test.ts | 4 +- tests/unit/proxy-connection-test.test.ts | 7 +- tests/unit/proxy-fallback-cache-key.test.ts | 42 + .../unit/qoder-jobtoken-exchange-4683.test.ts | 67 ++ tests/unit/refresh-serializer.test.ts | 4 +- tests/unit/response-sanitizer.test.ts | 108 ++- tests/unit/responses-transformer.test.ts | 31 +- .../unit/responses-translation-fixes.test.ts | 36 +- tests/unit/save-call-log-persistence.test.ts | 96 +++ .../unit/security/docs-path-traversal.test.ts | 52 ++ tests/unit/security/docs-sanitization.test.ts | 27 + tests/unit/service-token-refresh.test.ts | 7 +- tests/unit/sse-nonstream-accept-5305.test.ts | 69 ++ tests/unit/system-transforms.test.ts | 2 +- tests/unit/system-version-check-4100.test.ts | 47 +- tests/unit/t14-proxy-fast-fail.test.ts | 25 + .../t19-codex-responses-empty-content.test.ts | 4 +- tests/unit/t20-t22-provider-headers.test.ts | 72 -- tests/unit/t28-model-catalog-updates.test.ts | 10 +- tests/unit/token-health-check.test.ts | 10 +- .../unit/token-refresh-cas-guard-4038.test.ts | 114 +++ .../translator-gemini-audio-input.test.ts | 2 +- .../translator-openai-responses-req.test.ts | 34 +- .../translator-openai-to-gemini-sse.test.ts | 2 +- .../unit/translator-openai-to-gemini.test.ts | 70 +- tests/unit/translator-openai-to-kiro.test.ts | 4 +- ...resp-antigravity-thinking-boundary.test.ts | 75 ++ .../translator-resp-kiro-to-openai.test.ts | 5 +- .../translator-resp-openai-responses.test.ts | 30 +- tests/unit/tray-systray-loader-4605.test.ts | 80 ++ tests/unit/ui/compressionAnnotation.test.tsx | 67 ++ tests/unit/ui/quantumLockBadge.test.tsx | 37 + tests/unit/ui/riskGateBadge.test.tsx | 42 + tests/unit/ui/saliencyHeatmap.test.tsx | 101 +++ tests/unit/usage-providers.test.ts | 19 +- tests/unit/usage-service-hardening.test.ts | 175 +---- tests/unit/usage-utils.test.ts | 11 - tests/unit/vscode-token-routes.test.ts | 27 +- vitest.mcp.config.ts | 1 + 1007 files changed, 16451 insertions(+), 21343 deletions(-) delete mode 100644 .mcp.json.example delete mode 100644 bin/cli/commands/setup-gemini.mjs delete mode 100644 docs/DOCUMENTATION_OVERHAUL_PLAN.md delete mode 100644 docs/INCIDENT_RESPONSE.md delete mode 100644 docs/PERF_BUDGETS.md delete mode 100644 docs/SUBMIT_PR.md delete mode 100644 docs/THREAT_MODEL.md create mode 100644 docs/comparison/meta.json delete mode 100644 docs/compression/2026-06-20-unified-compression-config-panel-design.md delete mode 100644 docs/compression/2026-06-20-unified-compression-config-panel-plan.md delete mode 100644 docs/compression/2026-06-21-compression-phase2-named-profiles-design.md delete mode 100644 docs/compression/2026-06-21-compression-phase2-named-profiles-plan.md delete mode 100644 docs/compression/2026-06-21-compression-phase3-request-header-design.md delete mode 100644 docs/compression/2026-06-21-compression-phase3-request-header-plan.md rename docs/diagrams/{auto-combo-9factor.mmd => auto-combo-12factor.mmd} (52%) create mode 100644 docs/diagrams/exported/auto-combo-12factor.svg delete mode 100644 docs/diagrams/exported/auto-combo-9factor.svg delete mode 100644 docs/diagrams/exported/mcp-tools-87.svg create mode 100644 docs/diagrams/exported/mcp-tools-94.svg rename docs/diagrams/{mcp-tools-87.mmd => mcp-tools-94.mmd} (63%) rename docs/{dev/plugins.md => frameworks/PLUGINS.md} (97%) rename docs/{plugins => frameworks}/PLUGIN_SDK.md (62%) rename docs/{marketing => guides}/TIERS.md (97%) rename docs/{features => guides}/USAGE_QUOTA_GUIDE.md (65%) delete mode 100644 docs/ops/DOCUMENTATION_AUDIT_REPORT.md delete mode 100644 docs/ops/E2E_DASHBOARD_SHAKEDOWN_v3.8.0.md delete mode 100644 docs/ops/MUTATION_GATE_SPIKE_VERDICT.md rename docs/{PROVIDERS.md => providers/CLAUDE_WEB.md} (88%) create mode 100644 docs/providers/meta.json delete mode 100644 docs/releases/v3.8.0.md delete mode 100644 docs/research/DISCOVERY_TOOL_DESIGN.md delete mode 100644 docs/research/UNLIMITED_LLM_ACCESS.md delete mode 100644 docs/security/CLI_TOKEN_AUTH.md create mode 100644 open-sse/config/providers/registry/copilot-m365-web/index.ts delete mode 100644 open-sse/config/providers/registry/gemini/cli/index.ts create mode 100644 open-sse/executors/copilot-m365-web.ts delete mode 100644 open-sse/executors/gemini-cli.ts create mode 100644 open-sse/mcp-server/__tests__/httpAuthContext.test.ts create mode 100644 open-sse/mcp-server/__tests__/toolSearch.catalog.test.ts create mode 100644 open-sse/mcp-server/__tests__/toolSearch.search.test.ts create mode 100644 open-sse/mcp-server/__tests__/toolSearch.signature.test.ts create mode 100644 open-sse/mcp-server/__tests__/toolSearch.tool.test.ts create mode 100644 open-sse/mcp-server/httpAuthContext.ts create mode 100644 open-sse/mcp-server/schemas/toolDefinition.ts create mode 100644 open-sse/mcp-server/schemas/toolSearch.ts create mode 100644 open-sse/mcp-server/toolSearch/catalog.ts create mode 100644 open-sse/mcp-server/toolSearch/handler.ts create mode 100644 open-sse/mcp-server/toolSearch/index.ts create mode 100644 open-sse/mcp-server/toolSearch/register.ts create mode 100644 open-sse/mcp-server/toolSearch/search.ts create mode 100644 open-sse/mcp-server/toolSearch/signature.ts create mode 100644 open-sse/services/combo/__tests__/targetExhaustion.test.ts create mode 100644 open-sse/services/compression/cacheAwareConfig.ts create mode 100644 open-sse/services/compression/engines/relevance/configSchema.ts create mode 100644 open-sse/services/compression/engines/relevance/index.ts create mode 100644 open-sse/services/compression/engines/relevance/scorer.ts create mode 100644 open-sse/services/compression/engines/rtk/configSchema.ts create mode 100644 open-sse/services/compression/engines/rtk/renderers/gitDiff.ts create mode 100644 open-sse/services/compression/engines/rtk/renderers/index.ts create mode 100644 open-sse/services/compression/engines/rtk/renderers/structuredTable.ts create mode 100644 open-sse/services/compression/engines/rtk/renderers/terraformPlan.ts create mode 100644 open-sse/services/compression/engines/rtk/renderers/testGreen.ts create mode 100644 open-sse/services/compression/engines/rtk/renderers/types.ts create mode 100644 open-sse/services/compression/engines/rtk/splitCompositeCommand.ts create mode 100644 open-sse/services/compression/entrypointWrap.ts create mode 100644 open-sse/services/compression/hardBudget.ts create mode 100644 open-sse/services/compression/quantumLock/index.ts create mode 100644 open-sse/services/compression/quantumLock/quantumLock.ts create mode 100644 open-sse/services/compression/quantumLock/quantumLockStep.ts create mode 100644 open-sse/services/compression/quantumLock/quantumPatterns.ts create mode 100644 open-sse/services/compression/quantumLock/strategyWrap.ts create mode 100644 open-sse/services/compression/resultMemo.ts create mode 100644 open-sse/services/compression/riskGate/index.ts create mode 100644 open-sse/services/compression/riskGate/riskGate.ts create mode 100644 open-sse/services/compression/riskGate/riskGateStep.ts create mode 100644 open-sse/services/compression/riskGate/riskPatterns.ts create mode 100644 open-sse/services/compression/riskGate/strategyWrap.ts delete mode 100644 open-sse/services/geminiCliHeaders.ts delete mode 100644 public/providers/gemini-cli.svg create mode 100755 scripts/ci/should-promote-latest.sh create mode 100644 src/app/(dashboard)/dashboard/compression/studio/CompressionAnnotation.tsx create mode 100644 src/app/(dashboard)/dashboard/compression/studio/QuantumLockBadge.tsx create mode 100644 src/app/(dashboard)/dashboard/compression/studio/RiskGateBadge.tsx create mode 100644 src/app/(dashboard)/dashboard/compression/studio/SaliencyHeatmap.tsx create mode 100644 src/app/(dashboard)/dashboard/context/page.tsx delete mode 100644 src/app/(dashboard)/dashboard/providers/[id]/components/modals/ImportGeminiAuthModal.tsx create mode 100644 src/app/(dashboard)/dashboard/providers/components/NoAuthProvidersSection.tsx delete mode 100644 src/app/api/providers/[id]/gemini-cli-auth/apply-local/route.ts delete mode 100644 src/app/api/providers/[id]/gemini-cli-auth/export/route.ts delete mode 100644 src/app/api/providers/gemini-cli-auth/import-bulk/route.ts delete mode 100644 src/app/api/providers/gemini-cli-auth/import/route.ts delete mode 100644 src/app/api/providers/gemini-cli-auth/zip-extract/route.ts create mode 100644 src/lib/db/migrations/109_call_logs_correlation_id.sql create mode 100644 src/lib/docsI18nPath.ts create mode 100644 src/lib/docsSanitizer.ts delete mode 100644 src/lib/oauth/providers/gemini.ts delete mode 100644 src/lib/oauth/services/gemini.ts delete mode 100644 src/lib/oauth/utils/geminiAuthFile.ts delete mode 100644 src/lib/oauth/utils/geminiAuthImport.ts rename src/lib/oauth/utils/{geminiAuthZipExtract.ts => jsonZipExtract.ts} (98%) create mode 100644 tests/unit/auth-antigravity-account-retry-v2.test.ts create mode 100644 tests/unit/build/should-promote-latest-5301.test.ts create mode 100644 tests/unit/cli-runtime-imports-packaged-5227.test.ts delete mode 100644 tests/unit/cli/setup-gemini.test.ts create mode 100644 tests/unit/combo-rr-session-stickiness-3825.test.ts create mode 100644 tests/unit/command-code-maxtokens-negative-5166.test.ts create mode 100644 tests/unit/compression/compressionAnnotation.test.ts create mode 100644 tests/unit/compression/hard-budget.test.ts create mode 100644 tests/unit/compression/heatmap-build.test.ts create mode 100644 tests/unit/compression/quantumLockDetect.test.ts create mode 100644 tests/unit/compression/quantumLockIntegration.test.ts create mode 100644 tests/unit/compression/quantumLockStep.test.ts create mode 100644 tests/unit/compression/relevance-engine.test.ts create mode 100644 tests/unit/compression/relevance-scorer.test.ts create mode 100644 tests/unit/compression/result-memo.test.ts create mode 100644 tests/unit/compression/riskGateDetect.test.ts create mode 100644 tests/unit/compression/riskGateIntegration.test.ts create mode 100644 tests/unit/compression/riskGateStep.test.ts create mode 100644 tests/unit/compression/rtk-render-gitdiff.test.ts create mode 100644 tests/unit/compression/rtk-render-table.test.ts create mode 100644 tests/unit/compression/rtk-render-terraform.test.ts create mode 100644 tests/unit/compression/rtk-render-testgreen.test.ts create mode 100644 tests/unit/compression/rtk-renderers-integration.test.ts create mode 100644 tests/unit/compression/splitCompositeCommand.test.ts create mode 100644 tests/unit/copilot-m365-web-executor.test.ts create mode 100644 tests/unit/dashboard/context-parent-redirect-5298.test.ts delete mode 100644 tests/unit/executor-gemini-cli.test.ts delete mode 100644 tests/unit/gemini-import-route.test.ts delete mode 100644 tests/unit/gemini-usage-projectid.test.ts delete mode 100644 tests/unit/geminiAuthFile.test.ts delete mode 100644 tests/unit/geminiAuthImport.test.ts create mode 100644 tests/unit/grok-cli-strip-params.test.ts create mode 100644 tests/unit/kilocode-anonymous-fallback-4019.test.ts create mode 100644 tests/unit/kiro-continue-filler-5231.test.ts create mode 100644 tests/unit/mcp/tool-definition-reexport.test.ts create mode 100644 tests/unit/noauth-blocked-partition-5183.test.ts create mode 100644 tests/unit/proxy-fallback-cache-key.test.ts create mode 100644 tests/unit/save-call-log-persistence.test.ts create mode 100644 tests/unit/security/docs-path-traversal.test.ts create mode 100644 tests/unit/security/docs-sanitization.test.ts create mode 100644 tests/unit/sse-nonstream-accept-5305.test.ts delete mode 100644 tests/unit/t20-t22-provider-headers.test.ts create mode 100644 tests/unit/token-refresh-cas-guard-4038.test.ts create mode 100644 tests/unit/translator-resp-antigravity-thinking-boundary.test.ts create mode 100644 tests/unit/tray-systray-loader-4605.test.ts create mode 100644 tests/unit/ui/compressionAnnotation.test.tsx create mode 100644 tests/unit/ui/quantumLockBadge.test.tsx create mode 100644 tests/unit/ui/riskGateBadge.test.tsx create mode 100644 tests/unit/ui/saliencyHeatmap.test.tsx diff --git a/.env.example b/.env.example index d38fa84c92..77c4a23dea 100644 --- a/.env.example +++ b/.env.example @@ -413,7 +413,6 @@ NEXT_PUBLIC_CLOUD_URL= # debugging or when routing through a corporate mirror. Used by: # open-sse/services/usage.ts. #OMNIROUTE_CROF_USAGE_URL=https://crof.ai/usage_api/ -#OMNIROUTE_GEMINI_CLI_USAGE_URL=https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist #OMNIROUTE_CODEWHISPERER_BASE_URL=https://codewhisperer.us-east-1.amazonaws.com #OMNIROUTE_OPENCODE_QUOTA_URL=https://opencode.ai/zen/go/v1/quota #OMNIROUTE_OPENCODE_GO_QUOTA_URL=https://api.z.ai/api/monitor/usage/quota/limit @@ -721,7 +720,7 @@ CODEX_OAUTH_CLIENT_ID=app_EMoamEEZ73f0CkXaXp7hrann # Used by: open-sse/executors/theoldllm.ts. Default: 30000 (30s). # THEOLDLLM_NAV_TIMEOUT_MS=30000 -# ── Gemini / Gemini CLI / Antigravity / Windsurf (all Google-based) ── +# ── Gemini / Antigravity / Windsurf (all Google-based) ── # These providers ship public OAuth client_id/secret values (or Firebase Web # keys) embedded in their public CLIs/binaries. Defaults are baked into the # code via open-sse/utils/publicCreds.ts — leave the env vars unset to use @@ -730,8 +729,6 @@ CODEX_OAUTH_CLIENT_ID=app_EMoamEEZ73f0CkXaXp7hrann # # GEMINI_OAUTH_CLIENT_ID= # GEMINI_OAUTH_CLIENT_SECRET= -# GEMINI_CLI_OAUTH_CLIENT_ID= -# GEMINI_CLI_OAUTH_CLIENT_SECRET= # ANTIGRAVITY_OAUTH_CLIENT_ID= # ANTIGRAVITY_OAUTH_CLIENT_SECRET= # WINDSURF_FIREBASE_API_KEY= @@ -819,7 +816,7 @@ GITHUB_OAUTH_CLIENT_ID=Iv1.b507a08c87ecfe98 # VISION_BRIDGE_API_KEY= # ───────────────────────────────────────────────────────────────────────────── -# ⚠️ GOOGLE OAUTH (Antigravity, Gemini CLI) & OTHER PROVIDERS — REMOTE SERVERS +# ⚠️ GOOGLE OAUTH (Antigravity) & OTHER PROVIDERS — REMOTE SERVERS # ───────────────────────────────────────────────────────────────────────────── # The default Client IDs above ONLY work when OmniRoute runs on localhost. # For remote/VPS hosting (including Docker containers on remote servers): @@ -847,7 +844,7 @@ GITHUB_OAUTH_CLIENT_ID=Iv1.b507a08c87ecfe98 # Used by: open-sse/executors/base.ts — buildHeaders() dynamic lookup. # Update these when providers release new CLI versions to avoid blocks. -CLAUDE_USER_AGENT="claude-cli/2.1.187 (external, cli)" +CLAUDE_USER_AGENT="claude-cli/2.1.195 (external, cli)" # Disable the deterministic tool-name cloak applied on both Anthropic-bound paths # (executors/base.ts native OAuth + executors/cliproxyapi.ts CLIProxyAPI) — @@ -857,7 +854,7 @@ CLAUDE_USER_AGENT="claude-cli/2.1.187 (external, cli)" # forward the original names verbatim (debugging only). # CLAUDE_DISABLE_TOOL_NAME_CLOAK=false CODEX_USER_AGENT="codex-cli/0.142.0 (Windows 10.0.26200; x64)" -GITHUB_USER_AGENT="GitHubCopilotChat/0.45.1" +GITHUB_USER_AGENT="GitHubCopilotChat/0.54.0" ANTIGRAVITY_USER_AGENT="antigravity/2.0.1 linux/arm64 google-api-nodejs-client/10.3.0" KIRO_USER_AGENT="AWS-SDK-JS/3.0.0 kiro-ide/1.0.0" # KIRO_VERIFY_FULL_CRC=false # opt-in: full per-frame message CRC validation on the Kiro event stream (debug corrupted streams; prelude CRC + TLS already protect framing) @@ -872,9 +869,8 @@ KIRO_USER_AGENT="AWS-SDK-JS/3.0.0 kiro-ide/1.0.0" # Used by: open-sse/executors/kiro.ts # KIRO_VERIFY_FULL_CRC=false QODER_USER_AGENT="Qoder-Cli" -QWEN_USER_AGENT="QwenCode/0.15.11 (linux; x64)" +QWEN_USER_AGENT="QwenCode/0.19.3 (linux; x64)" CURSOR_USER_AGENT="Cursor/3.4" -GEMINI_CLI_USER_AGENT="google-api-nodejs-client/10.3.0" # Override Codex client version sent in headers independently of the # CODEX_USER_AGENT string. Used by: open-sse/config/codexClient.ts. @@ -1362,6 +1358,11 @@ APP_LOG_TO_FILE=true # Health check result cache TTL (ms). Default: 30000 (30s) # PROXY_HEALTH_CACHE_TTL_MS=30000 +# Unhealthy health check result cache TTL (ms). Default: 2000 (2s) +# Keeps transient fast-fail timeouts from poisoning a proxy for the full +# healthy-result cache window under high concurrency. +# PROXY_HEALTH_UNHEALTHY_CACHE_TTL_MS=2000 + # Allow OAuth and provider validation flows to bypass a pinned proxy and connect # directly when proxy reachability pre-checks fail. Default: false. # Also configurable from Dashboard > Settings > Feature Flags. diff --git a/.github/workflows/docker-publish.yml b/.github/workflows/docker-publish.yml index c21f233051..022a9f270f 100644 --- a/.github/workflows/docker-publish.yml +++ b/.github/workflows/docker-publish.yml @@ -98,11 +98,13 @@ jobs: PROMOTE="${PROMOTE_INPUT:-false}" else git fetch --tags --quiet || true - HIGHEST=$(git tag -l 'v[0-9]*' | sed 's/^v//' | grep -vE -- '-(rc|alpha|beta|pre|next)' | sort -V | tail -1 || echo "") - if [ -n "$HIGHEST" ] && [ "$VERSION" = "$HIGHEST" ]; then - PROMOTE="true" - else - echo "Version $VERSION is not the highest semver tag (highest=${HIGHEST:-}). Not promoting :latest." + # Decide via the extracted helper, which folds VERSION into the + # candidate set so the result is independent of git-tag sync timing + # on `release` events (#5301). Without that, the freshly-created tag + # is often not yet visible here and :latest stays a release behind. + PROMOTE=$(git tag -l 'v[0-9]*' | bash scripts/ci/should-promote-latest.sh "$VERSION") + if [ "$PROMOTE" != "true" ]; then + echo "Version $VERSION is not the highest stable semver. Not promoting :latest." fi fi echo "promote_latest=$PROMOTE" >> "$GITHUB_OUTPUT" diff --git a/.github/workflows/wiki-sync.yml b/.github/workflows/wiki-sync.yml index c3b095f751..da22cccd8d 100644 --- a/.github/workflows/wiki-sync.yml +++ b/.github/workflows/wiki-sync.yml @@ -9,7 +9,7 @@ name: Wiki Sync # It does NOT overwrite existing wiki pages by default: several docs sources still carry # stale counts (e.g. ARCHITECTURE.md says "177 providers" while the wiki cover is 226), # so blind overwrite would regress the wiki. Full content parity (--update-existing) is -# gated on regenerating those sources — see docs/ops/DOCUMENTATION_AUDIT_REPORT.md. +# gated on regenerating those sources first. on: push: diff --git a/.mcp.json.example b/.mcp.json.example deleted file mode 100644 index 0a11e61a62..0000000000 --- a/.mcp.json.example +++ /dev/null @@ -1,17 +0,0 @@ -{ - "$comment_purpose": "OPT-IN agent-lsp / LSP-in-the-loop (Quality Gates Fase 7 Task 15). Copy this file to `.mcp.json` to enable. It exposes a TypeScript language server to coding agents (Claude Code, etc.) so they get diagnostics / hover / go-to-definition / blast-radius BEFORE writing code — turning 'invented symbol' review-catches into impossible-at-edit-time. Pairs with `npm run typecheck:core` as a compile-before-claim check.", - "$comment_safety": "Shipped as `.example` (NOT `.mcp.json`) on purpose so it never auto-loads an unvetted server into everyone's session. Pick an MCP<->LSP bridge you trust and have verified locally, then drop in its package + args below. A broken MCP entry only logs a connection error; it does not break agent sessions. The underlying language server is `typescript-language-server` (npm, mature) — install via `npm i -g typescript-language-server typescript` or rely on npx.", - "mcpServers": { - "typescript-lsp": { - "command": "npx", - "args": [ - "-y", - "", - "--lsp", - "typescript-language-server", - "--stdio" - ], - "$note": "Replace with the concrete MCP<->LSP adapter you chose. It must speak MCP on stdio and proxy to `typescript-language-server --stdio`. Scope it to this repo's tsconfig (open-sse/tsconfig.json / tsconfig.json) for accurate diagnostics." - } - } -} diff --git a/@omniroute/opencode-plugin/README.md b/@omniroute/opencode-plugin/README.md index 55329dec8a..35fc53a242 100644 --- a/@omniroute/opencode-plugin/README.md +++ b/@omniroute/opencode-plugin/README.md @@ -196,7 +196,7 @@ Every field is optional. Defaults mirror v0.1.0 behaviour so existing `opencode. | `combos` | `boolean` | `true` | Discover `/api/combos` and surface them as pseudo-models with LCD capabilities. Combos are keyed under the `combo/` namespace and labelled `Combo: ` in the model picker so they're distinguishable from raw provider/model pairs. | | `enrichment` | `boolean` | `true` | Pull display names from `/api/pricing/models` AND per-million-token pricing (`input`, `output`, `cached` → `cacheRead`, `cache_creation` → `cacheWrite`) from `/api/pricing`, then overlay both onto the live catalog (so the UI shows `Claude 4.7 Opus` with `cost.input: 5`, `cost.output: 25` instead of raw IDs and zeroed cost). | | `compressionMetadata` | `boolean` | `false` | Pull `/api/context/combos` so combo names get tagged with their compression pipeline, e.g. `Combo: claude-primary [rtk🟡 → caveman🟠]`. Intensity tokens render as traffic-light emoji (🟢 lite/minimal · 🟡 standard · 🟠 aggressive/full · 🔴 ultra) so the picker advertises "how compressed" each combo is at a glance. | -| `providerTag` | `boolean` | `true` | Prepend a short upstream-provider label to the enriched display name with `" - "` separator, so `cc/claude-opus-4-7 → Claude - Claude Opus 4.7` differs visibly from `kr/claude-opus-4-7 → Kiro - Claude Opus 4.7` in the OC TUI model picker. Label resolution: use `/api/pricing/models[].name` verbatim when ≤8 chars (e.g. `Claude`, `Kiro`, `Codex`, `Qwen`), otherwise fall back to `UPPER(alias)` (e.g. `GitHub Models` → `GHM`, `Gemini-cli` → `GEMINI-CLI`). Idempotent. Combos intentionally skipped (the `Combo: ` prefix already conveys multi-upstream). | +| `providerTag` | `boolean` | `true` | Prepend a short upstream-provider label to the enriched display name with `" - "` separator, so `cc/claude-opus-4-7 → Claude - Claude Opus 4.7` differs visibly from `kr/claude-opus-4-7 → Kiro - Claude Opus 4.7` in the OC TUI model picker. Label resolution: use `/api/pricing/models[].name` verbatim when ≤8 chars (e.g. `Claude`, `Kiro`, `Codex`, `Qwen`), otherwise fall back to `UPPER(alias)` (e.g. `GitHub Models` → `GHM`, `Gemini` → `GEMINI`). Idempotent. Combos intentionally skipped (the `Combo: ` prefix already conveys multi-upstream). | | `usableOnly` | `boolean` | `false` | Read `/api/providers` and filter the catalog to providers that have at least one connection with `isActive: true` AND `testStatus: 'active'`. Subtract-filter semantics: providers unknown to BOTH the pricing-models catalog AND the connection table pass through (so synthetic prefixes like `agentrouter/*` survive). On fetch failure the filter is disabled for the refresh — never hides the whole catalog. | | `diskCache` | `boolean` | `true` | Persist the last successful `/v1/models` + `/api/combos` + enrichment + connections + compression snapshot to `${OPENCODE_DATA_DIR ?? ~/.local/share/opencode}/plugins/omniroute-.json`. On a subsequent cold start where `/v1/models` throws (network down / IP whitelist drop / 5xx) the static block hydrates from the snapshot so OC's model picker survives offline. Soft-fail on read/write — never blocks publishing. | | `geminiSanitization` | `boolean` | `true` | Strip `$schema`/`$ref`/`additionalProperties` from tool params when the model id matches `gemini` | diff --git a/@omniroute/opencode-plugin/src/index.ts b/@omniroute/opencode-plugin/src/index.ts index 058b1cb8c7..d69383b087 100644 --- a/@omniroute/opencode-plugin/src/index.ts +++ b/@omniroute/opencode-plugin/src/index.ts @@ -1225,7 +1225,7 @@ export interface OmniRouteEnrichmentEntry { cacheWrite?: number; }; /** - * Provider alias prefix seen in `/v1/models` ids (e.g. `cc`, `gemini-cli`). + * Provider alias prefix seen in `/v1/models` ids (e.g. `cc`, `gemini`). * Populated by `defaultOmniRouteEnrichmentFetcher` from * `/api/pricing/models` keys. Drives the `usableOnly` alias↔canonical * resolution. @@ -1233,7 +1233,7 @@ export interface OmniRouteEnrichmentEntry { providerAlias?: string; /** * Canonical provider id used by `/api/providers` connections (e.g. - * `claude`, `gemini-cli`, `kiro`). Populated from the per-provider + * `claude`, `gemini`, `kiro`). Populated from the per-provider * `entry.id` field inside `/api/pricing/models`. */ providerCanonical?: string; @@ -2046,7 +2046,7 @@ export function formatCompressionPipeline(pipeline: OmniRouteCompressionStep[]): export interface OmniRouteProviderConnection { /** Connection UUID. */ id: string; - /** Canonical provider id, e.g. `claude`, `gemini-cli`, `kiro`. Matches `entry.id` in `/api/pricing/models`. */ + /** Canonical provider id, e.g. `claude`, `gemini`, `kiro`. Matches `entry.id` in `/api/pricing/models`. */ provider: string; /** Connection auth flavor, e.g. `apikey`, `oauth`, `cookie`. */ authType?: string; @@ -2125,7 +2125,7 @@ export const defaultOmniRouteProvidersFetcher: OmniRouteProvidersFetcher = async * walk only the namespaced keys to derive the alias↔canonical mapping). * * Returns: - * - `aliases`: set of alias prefixes safe to keep (e.g. `cc`, `gemini-cli`). + * - `aliases`: set of alias prefixes safe to keep (e.g. `cc`, `gemini`). * - `canonicals`: set of canonical provider ids (e.g. `claude`, `kiro`). * * Callers should treat membership in EITHER set as "usable" — raw model @@ -2174,7 +2174,7 @@ export function usableProviderAliasSet( } // Always include every usable canonical as an alias too — handles the // common case where `/v1/models` ids use the canonical id directly - // (e.g. `gemini-cli/gemini-1.5-pro`). + // (e.g. `gemini/gemini-1.5-pro`). for (const canonical of usableCanonicals) aliases.add(canonical); return { aliases, canonicals: usableCanonicals, knownAliases }; } @@ -3174,7 +3174,6 @@ export function sanitizeGeminiToolSchemas(payload: unknown): unknown { * `gemini-2.5-flash`, etc.) * - `models/gemini-…` (Google Generative AI canonical id form) * - `google-vertex/gemini-…` (OpenCode + AI-SDK Vertex routing prefix) - * - `gemini-cli/…` (real OmniRoute alias surfaced on b35 prod `/v1/models`) * * Liberal by design: a false positive (cleaning a payload that didn't * need cleaning) costs only a structuredClone + one walk; a false negative diff --git a/@omniroute/opencode-plugin/tests/config-shim.test.ts b/@omniroute/opencode-plugin/tests/config-shim.test.ts index 63d7c277ac..8bfd5fe91b 100644 --- a/@omniroute/opencode-plugin/tests/config-shim.test.ts +++ b/@omniroute/opencode-plugin/tests/config-shim.test.ts @@ -1311,9 +1311,9 @@ test("config: providerTag (default-on) prepends ' - ' to enriched raw- "gemini-3-flash", { name: "Gemini 3 Flash", - providerAlias: "gemini-cli", - providerCanonical: "gemini-cli", - providerDisplayName: "Gemini-cli", + providerAlias: "gemini", + providerCanonical: "gemini", + providerDisplayName: "Gemini", }, ], ]) @@ -1335,10 +1335,7 @@ test("config: providerTag (default-on) prepends ' - ' to enriched raw- entry.models["opencode-omniroute/claude-sonnet-4-6"].name, "Claude - Claude Sonnet 4.6" ); - assert.equal( - entry.models["opencode-omniroute/gemini-3-flash"].name, - "Gemini-cli - Gemini 3 Flash" - ); + assert.equal(entry.models["opencode-omniroute/gemini-3-flash"].name, "Gemini - Gemini 3 Flash"); // Combos stay untouched — `Combo: ` prefix already conveys multi-upstream. assert.equal(entry.models["opencode-omniroute/claude-tier"].name, "Claude Tier"); }); diff --git a/@omniroute/opencode-plugin/tests/gemini-sanitize.test.ts b/@omniroute/opencode-plugin/tests/gemini-sanitize.test.ts index effd66eae9..3cebfc52a5 100644 --- a/@omniroute/opencode-plugin/tests/gemini-sanitize.test.ts +++ b/@omniroute/opencode-plugin/tests/gemini-sanitize.test.ts @@ -212,8 +212,8 @@ test("shouldSanitizeForGemini: google-vertex/gemini-1.5-flash → true", () => { assert.equal(shouldSanitizeForGemini({ model: "google-vertex/gemini-1.5-flash" }), true); }); -test("shouldSanitizeForGemini: gemini-cli/gemini-2.5-pro → true (real OmniRoute alias)", () => { - assert.equal(shouldSanitizeForGemini({ model: "gemini-cli/gemini-2.5-pro" }), true); +test("shouldSanitizeForGemini: gemini/gemini-2.5-pro → true", () => { + assert.equal(shouldSanitizeForGemini({ model: "gemini/gemini-2.5-pro" }), true); }); test("shouldSanitizeForGemini: claude-sonnet-4 → false", () => { diff --git a/AGENTS.md b/AGENTS.md index 11e52a6a8b..ce665709a4 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -3,14 +3,14 @@ ## Project Unified AI proxy/router — route any LLM through one endpoint. Multi-provider support -with **231 provider entries** (OpenAI, Anthropic, Gemini, DeepSeek, Groq, xAI, Mistral, Fireworks, +with **237 provider entries** (OpenAI, Anthropic, Gemini, DeepSeek, Groq, xAI, Mistral, Fireworks, Cohere, NVIDIA, Cerebras, Pollinations, Puter, Cloudflare AI, HuggingFace, DeepInfra, SambaNova, Meta Llama API, Moonshot AI, AI21 Labs, Databricks, Snowflake, and many more) -with **MCP Server** (87 tools), **A2A v0.3 Protocol**, and **Electron desktop app**. +with **MCP Server** (94 tools), **A2A v0.3 Protocol**, and **Electron desktop app**. -> **Live counts (v3.8.31)**: providers 231 · MCP tools 87 · MCP scopes 30 · A2A skills 6 · -> open-sse services 115 · routing strategies 15 · auto-combo scoring factors 9 · -> DB modules 83 · DB migrations 97 · base tables 17 · search providers 11 · +> **Live counts (v3.8.40)**: providers 237 · MCP tools 94 · MCP scopes 30 · A2A skills 6 · +> open-sse services 298 · routing strategies 17 · auto-combo scoring factors 12 · +> DB modules 94 · DB migrations 106 · base tables 17 · search providers 11 · > i18n locales 42. **Refresh with `npm run check:docs-all`.** ## Doc Accuracy Discipline (read before writing any doc) @@ -267,7 +267,7 @@ Zod schemas, and unit tests aligned when editing. ### Provider Categories -- **Free** (4): Qoder AI, Qwen Code, Gemini CLI (deprecated), Kiro AI +- **Free** (3): Qoder AI, Qwen Code, Kiro AI - **OAuth** (14): Claude Code, Antigravity, Codex, GitHub Copilot, Cursor, Kimi Coding, Kilo Code, Cline, Qwen (⚠️ free tier discontinued 2026-04-15), Kiro, Qoder, Gemini, Windsurf (v3.8), GitLab Duo (v3.8) - **API Key** (120+): OpenAI, Anthropic, Gemini, DeepSeek, Groq, xAI, Mistral, Perplexity, Together, Fireworks, Cerebras, Cohere, NVIDIA, Nebius, SiliconFlow, Hyperbolic, @@ -291,7 +291,7 @@ Providers are registered in `src/shared/constants/providers.ts` with Zod validat ### Executors (`open-sse/executors/`) Provider-specific request executors: `base.ts`, `default.ts`, `cursor.ts`, `codex.ts`, -`antigravity.ts`, `github.ts`, `gemini-cli.ts`, `kiro.ts`, `qoder.ts`, `vertex.ts`, +`antigravity.ts`, `github.ts`, `kiro.ts`, `qoder.ts`, `vertex.ts`, `cloudflare-ai.ts`, `opencode.ts`, `pollinations.ts`, `puter.ts`. #### Executor Internals @@ -391,7 +391,7 @@ Policy engine modules: `policyEngine.ts`, `comboResolver.ts`, `costRules.ts`, ### MCP Server (`open-sse/mcp-server/`) -**87 tools** total (`TOTAL_MCP_TOOL_COUNT`, `open-sse/mcp-server/server.ts`): a 33-entry base registry (`MCP_TOOLS` in `schemas/tools.ts`, bundling the core / cache / compression / 1proxy / advanced tools) **plus** standalone module sets — memory (3), skill (4), agentSkill (3), gamification (8), plugin (8), notion (6), obsidian (22). 3 transports (stdio / SSE / Streamable HTTP). Scoped auth (30 scopes — see `OMNIROUTE_MCP_SCOPES`), Zod schemas. See [`docs/frameworks/MCP-SERVER.md`](docs/frameworks/MCP-SERVER.md). +**94 tools** total (`TOTAL_MCP_TOOL_COUNT`, `open-sse/mcp-server/server.ts`): a 34-entry base registry (`MCP_TOOLS` in `schemas/tools.ts`, bundling the core / cache / compression / 1proxy / advanced tools) **plus** standalone module sets — memory (3), skill (4), agentSkill (3), pool (6), gamification (8), plugin (8), notion (6), obsidian (22). 3 transports (stdio / SSE / Streamable HTTP). Scoped auth (30 scopes — see `OMNIROUTE_MCP_SCOPES`), Zod schemas. See [`docs/frameworks/MCP-SERVER.md`](docs/frameworks/MCP-SERVER.md). **Core tools** (20): get_health, list_combos, get_combo_metrics, switch_combo, check_quota, route_request, cost_report, list_models_catalog, web_search, simulate_route, set_budget_guard, @@ -534,33 +534,33 @@ Cloudflare Quick/Named, ngrok, Tailscale Funnel. See [`docs/ops/TUNNELS_GUIDE.md For any non-trivial change, read the matching deep-dive first: -| Area | Doc | -| ------------------------------------------ | ----------------------------------------------------------------------------------------------------------------------------------- | -| Repo navigation | [`docs/architecture/REPOSITORY_MAP.md`](docs/architecture/REPOSITORY_MAP.md) | -| Architecture | [`docs/architecture/ARCHITECTURE.md`](docs/architecture/ARCHITECTURE.md) | -| Engineering reference | [`docs/architecture/CODEBASE_DOCUMENTATION.md`](docs/architecture/CODEBASE_DOCUMENTATION.md) | -| Auto-Combo (12-factor, 15 strategies) | [`docs/routing/AUTO-COMBO.md`](docs/routing/AUTO-COMBO.md) | -| Resilience (3 layers) | [`docs/architecture/RESILIENCE_GUIDE.md`](docs/architecture/RESILIENCE_GUIDE.md) | -| Skills | [`docs/frameworks/SKILLS.md`](docs/frameworks/SKILLS.md) | -| Memory | [`docs/frameworks/MEMORY.md`](docs/frameworks/MEMORY.md) | -| Cloud agents | [`docs/frameworks/CLOUD_AGENT.md`](docs/frameworks/CLOUD_AGENT.md) | -| Guardrails | [`docs/security/GUARDRAILS.md`](docs/security/GUARDRAILS.md) | -| Evals | [`docs/frameworks/EVALS.md`](docs/frameworks/EVALS.md) | -| Compliance | [`docs/security/COMPLIANCE.md`](docs/security/COMPLIANCE.md) | -| Webhooks | [`docs/frameworks/WEBHOOKS.md`](docs/frameworks/WEBHOOKS.md) | -| Authz | [`docs/architecture/AUTHZ_GUIDE.md`](docs/architecture/AUTHZ_GUIDE.md) | -| Stealth | [`docs/security/STEALTH_GUIDE.md`](docs/security/STEALTH_GUIDE.md) | -| Reasoning replay | [`docs/routing/REASONING_REPLAY.md`](docs/routing/REASONING_REPLAY.md) | -| Agent protocols (A2A / ACP / Cloud) | [`docs/frameworks/AGENT_PROTOCOLS_GUIDE.md`](docs/frameworks/AGENT_PROTOCOLS_GUIDE.md) | -| MCP server | [`docs/frameworks/MCP-SERVER.md`](docs/frameworks/MCP-SERVER.md) | -| A2A server | [`docs/frameworks/A2A-SERVER.md`](docs/frameworks/A2A-SERVER.md) | +| Area | Doc | +| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------- | +| Repo navigation | [`docs/architecture/REPOSITORY_MAP.md`](docs/architecture/REPOSITORY_MAP.md) | +| Architecture | [`docs/architecture/ARCHITECTURE.md`](docs/architecture/ARCHITECTURE.md) | +| Engineering reference | [`docs/architecture/CODEBASE_DOCUMENTATION.md`](docs/architecture/CODEBASE_DOCUMENTATION.md) | +| Auto-Combo (12-factor, 17 strategies) | [`docs/routing/AUTO-COMBO.md`](docs/routing/AUTO-COMBO.md) | +| Resilience (3 layers) | [`docs/architecture/RESILIENCE_GUIDE.md`](docs/architecture/RESILIENCE_GUIDE.md) | +| Skills | [`docs/frameworks/SKILLS.md`](docs/frameworks/SKILLS.md) | +| Memory | [`docs/frameworks/MEMORY.md`](docs/frameworks/MEMORY.md) | +| Cloud agents | [`docs/frameworks/CLOUD_AGENT.md`](docs/frameworks/CLOUD_AGENT.md) | +| Guardrails | [`docs/security/GUARDRAILS.md`](docs/security/GUARDRAILS.md) | +| Evals | [`docs/frameworks/EVALS.md`](docs/frameworks/EVALS.md) | +| Compliance | [`docs/security/COMPLIANCE.md`](docs/security/COMPLIANCE.md) | +| Webhooks | [`docs/frameworks/WEBHOOKS.md`](docs/frameworks/WEBHOOKS.md) | +| Authz | [`docs/architecture/AUTHZ_GUIDE.md`](docs/architecture/AUTHZ_GUIDE.md) | +| Stealth | [`docs/security/STEALTH_GUIDE.md`](docs/security/STEALTH_GUIDE.md) | +| Reasoning replay | [`docs/routing/REASONING_REPLAY.md`](docs/routing/REASONING_REPLAY.md) | +| Agent protocols (A2A / ACP / Cloud) | [`docs/frameworks/AGENT_PROTOCOLS_GUIDE.md`](docs/frameworks/AGENT_PROTOCOLS_GUIDE.md) | +| MCP server | [`docs/frameworks/MCP-SERVER.md`](docs/frameworks/MCP-SERVER.md) | +| A2A server | [`docs/frameworks/A2A-SERVER.md`](docs/frameworks/A2A-SERVER.md) | | API reference | [`docs/reference/API_REFERENCE.md`](docs/reference/API_REFERENCE.md) + [`docs/openapi.yaml`](docs/openapi.yaml) | -| Provider catalog (auto-generated) | [`docs/reference/PROVIDER_REFERENCE.md`](docs/reference/PROVIDER_REFERENCE.md) | -| Tunnels | [`docs/ops/TUNNELS_GUIDE.md`](docs/ops/TUNNELS_GUIDE.md) | -| Electron desktop | [`docs/guides/ELECTRON_GUIDE.md`](docs/guides/ELECTRON_GUIDE.md) | -| Release flow | [`docs/ops/RELEASE_CHECKLIST.md`](docs/ops/RELEASE_CHECKLIST.md) | -| Quality gates (35 gates, allowlist policy) | [`docs/architecture/QUALITY_GATES.md`](docs/architecture/QUALITY_GATES.md) | -| Cluster opt-in profiles (memory, bifrost) | [`docs/architecture/cluster-decisions.md`](docs/architecture/cluster-decisions.md) | +| Provider catalog (auto-generated) | [`docs/reference/PROVIDER_REFERENCE.md`](docs/reference/PROVIDER_REFERENCE.md) | +| Tunnels | [`docs/ops/TUNNELS_GUIDE.md`](docs/ops/TUNNELS_GUIDE.md) | +| Electron desktop | [`docs/guides/ELECTRON_GUIDE.md`](docs/guides/ELECTRON_GUIDE.md) | +| Release flow | [`docs/ops/RELEASE_CHECKLIST.md`](docs/ops/RELEASE_CHECKLIST.md) | +| Quality gates (35 gates, allowlist policy) | [`docs/architecture/QUALITY_GATES.md`](docs/architecture/QUALITY_GATES.md) | +| Cluster opt-in profiles (memory, bifrost) | [`docs/architecture/cluster-decisions.md`](docs/architecture/cluster-decisions.md) | --- diff --git a/CHANGELOG.md b/CHANGELOG.md index eea092fe8a..f6a9ea7442 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,81 @@ --- +## [3.8.40] — TBD + +_In development — bullets added per PR; finalized at release._ + +### ✨ New Features + +- **feat(compression): relevance extractive engine** — a new opt-in compression engine that scores each sentence by term-overlap (Jaccard) with the user's last query minus a length/boilerplate penalty, greedily keeps the most relevant within a budget, and reconstructs the original order. Pure-string, deterministic, ReDoS-safe (char-code tokenization, no `RegExp` over user input), fail-open, default off. Ideal for trimming long pasted RAG context / tool output to what's relevant. Sentences carrying real signal (digits/URLs/errors/code/paths) are never dropped; `overlapThreshold`/`budgetPercent`/`boilerplateWeight` are configurable. Tier-2 item of the compression feature-extraction roadmap (#7). ([#5289](https://github.com/diegosouzapw/OmniRoute/pull/5289)) +- **feat(compression): hard-budget mode — compress to ≤ N tokens** — a deterministic post-pass (`targetTokens` / `targetRatio`, default unset → no-op) that trims a body to a token budget. It ranks sentences/lines by average `scoreToken` ascending and drops the lowest-saliency ones until the body fits (measured by the exact cl100k `countTextTokens`), preserving original order. Lines carrying real signal (digits, URLs, `Error:`-family, code fences, stack `at`-frames, multi-segment paths, `key=value`) are never dropped; the budget is distributed proportionally across messages so the total stays ≤ target; an unreachable target (all-preserved) surfaces a `validationWarnings` note instead of failing silently. Does NOT touch the `estimateCompressionTokens` budget-gate estimator. Tier-3 item of the compression feature-extraction roadmap (#17). ([#5288](https://github.com/diegosouzapw/OmniRoute/pull/5288), follow-up [#5291](https://github.com/diegosouzapw/OmniRoute/pull/5291)) +- **feat(compression): result memoization for deterministic engines (opt-in)** — caches `(input, config) → result` for provably pure, stateless modes (`lite`/`standard`/`rtk` and stacked pipelines of `{lite,caveman,rtk}`) to skip recompute on the hot path. Opt-in via `memoizeCompressionResults` (default off → zero behavior change). Conservative opt-in whitelist (stateful `ccr`/`session-dedup` — which write the cross-request CCR store — and model-backed `ultra`/`aggressive`/`llmlingua` are never cached), principal-scoped (skipped without a principal, so no cross-principal body leak), and clone-on-store + clone-on-read. Tier-3 item of the compression feature-extraction roadmap (#21). ([#5286](https://github.com/diegosouzapw/OmniRoute/pull/5286)) +- **feat(compression): inline transparency annotation** — surfaces `tokens=847→312; rules: filler×8, dedup×2` derived from existing compression stats. The `X-OmniRoute-Compression` response header is extended **append-only** (the `mode; source=X` prefix stays byte-identical, so existing header parsers don't break) and the compression studio cockpit shows a matching badge. Zero new computation — it aggregates the `rulesApplied`/`techniquesUsed` already on the stats. Tier-3 item of the compression feature-extraction roadmap (#18). ([#5284](https://github.com/diegosouzapw/OmniRoute/pull/5284)) +- **feat(compression): saliency heatmap in the compression studio** — the preview studio can now color each token by saliency: `ultra` per-token `scoreToken` (0–1, green→red gradient) or universal kept/removed from the existing diff. A dry-run visualization behind a toggle (no cost on a normal preview; backward-compatible when off). Completes the visualization half of roadmap item #13 (the A/B comparison shipped in [#5080](https://github.com/diegosouzapw/OmniRoute/pull/5080)). ([#5285](https://github.com/diegosouzapw/OmniRoute/pull/5285)) +- **feat(compression): composite-command splitter for RTK detection** — `cd /x && git status` now detects as `git-status` (previously the whole string was treated as one command and matched no filter). A quote-aware top-level tokenizer splits on `&&`/`||`/`;` (never inside quotes or `$(…)`/backtick subshells) and feeds the **last** segment to RTK command detection, so every RTK filter/renderer fires on commands wrapped in `cd … &&`/`||`/`;` chains. O(n), no RegExp over the command (ReDoS-safe). Tier-3 item of the compression feature-extraction roadmap (#16). ([#5283](https://github.com/diegosouzapw/OmniRoute/pull/5283)) +- **feat(mcp): `omniroute_tool_search` tool + one-line TS signatures** — new MCP tool that does lexical keyword search over every MCP tool's name/description and returns the top matches as compact one-line TypeScript signatures (~half the JSON-schema token cost), so agents discover tools on demand instead of carrying all ~88 schemas every turn. Search is ReDoS-safe (substring scoring, never `new RegExp` on the query) and deterministic; `tools/list` stays complete (no hidden tools). Adds the `read:tools` scope. Tier-1 item of the compression feature-extraction roadmap. ([#5269](https://github.com/diegosouzapw/OmniRoute/pull/5269)) +- **feat(compression): RTK semantic command-output renderers (opt-in)** — adds a second, opt-in compaction layer to the RTK engine that rewrites structured command output into a far more compact semantic form: `git diff` → file headers + `@@` hunks + changed lines only; an all-green `pytest`/`jest`/`vitest`/`eslint` run → its one-line summary; `terraform`/`tofu plan` → `Plan: +N ~M -K` plus the resource list; `kubectl`/`aws` JSON arrays → a minimal table. Each renderer is conservative (no-op when the shape doesn't match) and the integration is fail-open; the test-green renderer never collapses output that carries any failure signal. Gated by `RtkConfig.enableRenderers` (default off → zero behavioral change). Eighth item of the compression feature-extraction roadmap. ([#5268](https://github.com/diegosouzapw/OmniRoute/pull/5268)) +- **feat(compression): QuantumLock cache-prefix stabilization (opt-in, default off)** — recovers upstream prompt-cache hits that a volatile fragment in the system prompt would otherwise bust. When a caller injects a session UUID, unix timestamp, request-id, JWT, API-key shape, or long hex digest into the `role:system` message every turn, the longest common prefix across turns ends at that changing byte → the whole system prompt after it is re-billed and re-processed each turn. QuantumLock replaces each non-semantic volatile fragment with a **positional, value-independent** placeholder `⟦Q{i}⟧` and appends the real values in a delimited `⟦QUANTUMLOCK⟧` tail. The rewrite is **sent to the model** (lossless — not restored), so the system-prompt body becomes **byte-identical across turns** and the provider caches the long stable prefix while only the small tail differs. Opt-in, default off, applied only for caching providers (`isCachingProvider && config.quantumLock.enabled`); bounded ReDoS-safe patterns; idempotent; **no date/time patterns** (semantically meaningful — explicit non-goal). Studio gets a toggle + a "🔒 N volatile fragment(s) stabilized" dry-run badge. Seventh item of the compression feature-extraction roadmap (bench: [#5080](https://github.com/diegosouzapw/OmniRoute/pull/5080), gate: [#5127](https://github.com/diegosouzapw/OmniRoute/pull/5127), fuzzy: [#5143](https://github.com/diegosouzapw/OmniRoute/pull/5143), ionizer: [#5148](https://github.com/diegosouzapw/OmniRoute/pull/5148), TOON: [#5163](https://github.com/diegosouzapw/OmniRoute/pull/5163), CCR ranged: [#5187](https://github.com/diegosouzapw/OmniRoute/pull/5187), risk-gate: [#5243](https://github.com/diegosouzapw/OmniRoute/pull/5243)). ([#5260](https://github.com/diegosouzapw/OmniRoute/pull/5260)) +- **kilocode:** anonymous (no-auth) access to Kilo Code's free models, mirroring the `opencode`/`mimocode` pattern. With no Kilo account connected, requests now fall back to the gateway's anonymous tier (`Authorization: Bearer anonymous` on `api.kilo.ai/api/openrouter`) so the free models work without signup; a connected OAuth account is still used unchanged for the paid tier ([#5259](https://github.com/diegosouzapw/OmniRoute/pull/5259), #4019 — thanks @Theadd for the reference implementation) +- **feat(logging): call-log correlation ID (end-to-end)** — every request now gets a unique correlation id, returned in the `X-Correlation-Id` response header, persisted in `call_logs` (migration **109**), filterable via `/api/usage/call-logs`, and surfaced in the dashboard request logger (per-chunk stream timestamps + active-requests-first sort). This is the safe, cohesive core subset of the larger #5275 — landed on its own so the low-risk value isn't blocked by the parts of that PR still under review. ([#5279](https://github.com/diegosouzapw/OmniRoute/pull/5279) — thanks @hartmark) +- **feat(providers): Microsoft 365 Copilot individual provider** — adds the `copilot-m365-web` provider (the 237th), wiring the M365 BizChat framing/connection helpers into a selectable web-session provider backed by `m365.cloud.microsoft/chat` for individual Microsoft 365 plans. Builds on the M365 pure-framing groundwork from #4696. Regression guard: `tests/unit/copilot-m365-web-executor.test.ts`. ([#5302](https://github.com/diegosouzapw/OmniRoute/pull/5302) — thanks @skyzea1) + +### 🔧 Bug Fixes + +- **ci(docker):** re-point the Docker Hub / GHCR `:latest` (and `:latest-web`) tags to the just-published release. On a `release: released` event the freshly-created git tag is often not yet visible to `git fetch --tags` when `docker-publish` runs, so the `:latest`-promotion gate built its candidate set purely from `git tag -l` and resolved the highest semver to the **previous** version — leaving `latest` one release behind (3.8.39 published, `latest` still 3.8.38). The decision now lives in `scripts/ci/should-promote-latest.sh`, which folds the current `VERSION` into the candidate set before picking the highest stable semver, making promotion independent of tag-sync timing (a patch published after a higher minor still won't grab `latest`). Regression guard: `tests/unit/build/should-promote-latest-5301.test.ts` ([#5301](https://github.com/diegosouzapw/OmniRoute/issues/5301)) +- **command-code:** treat a non-positive `max_tokens`/`max_completion_tokens` (e.g. Zoo Code's `-1` "let the server choose") as "no limit" — omit the field instead of forcing it to `1`. `clampMaxTokens` previously did `Math.max(1, …)`, so a client `-1` was sent upstream as `max_tokens: 1`, truncating the response to a single token (the observed `completion_tokens: 1`, `content: null`, `reasoning_content: "The"` with `finish_reason: stop`). Now any value `≤ 0` is dropped so Command Code applies the model's native default; positive values are still floored and clamped to the 200k ceiling. Regression guard: `tests/unit/command-code-maxtokens-negative-5166.test.ts` ([#5166](https://github.com/diegosouzapw/OmniRoute/issues/5166) — thanks @Stazyu) +- **fix(auth): compare-and-swap guard on the OAuth refresh persist** — under multi-agent load, the per-connection refresh mutex makes `[network refresh + DB write]` atomic for **one** connection, but it does not protect against a **third** writer (a sibling request, a concurrent HealthCheck, or a replica) landing a fresher `refresh_token` rotation on the same `connection_id` between the staleness read and the persist. Overwriting that fresher row reverts the sibling's rotation; the next caller then loads the now-consumed token, Auth0/Anthropic flag it as `refresh_token_reused`, and the whole token family gets revoked (the 1352× claude/`aa5dd5cf` invalidation storm). `getAccessToken` now re-reads the row's current `refresh_token` immediately before persisting (inside the mutex) and **skips the write** when it has rotated past the token the caller presented — the caller still receives the freshly-issued access token, only the DB overwrite is skipped. Opt-in via `runWithCasGuard` (no active guard ⇒ byte-identical behavior); skip/persist counters exposed via `getCasGuardStats()`. Regression guard: `tests/unit/token-refresh-cas-guard-4038.test.ts`. ([#4038](https://github.com/diegosouzapw/OmniRoute/issues/4038) — thanks @KooshaPari for the root-cause diagnosis) +- **mcp:** break the `schemas/tools.ts ↔ schemas/toolSearch.ts` import cycle introduced when the `tool_search` defs (#5269) were extracted into their own module — `toolSearch.ts` imported `McpToolDefinition` from `tools.ts` while `tools.ts` imported `toolSearchTool` from `toolSearch.ts`, failing `check:cycles` on `release/v3.8.40`. The shared `AuditLevel` + `McpToolDefinition` types now live in a leaf `schemas/toolDefinition.ts` that both import; `tools.ts` re-exports them for backward compatibility. +- **compression (analytics):** record attempted-but-no-op compression runs so Stacked is no longer invisible when it saves nothing. Previously a `compression_analytics` row was written only on a net-positive saving, so a Stacked (RTK→Caveman) pipeline that ran on already-compact context produced no row — indistinguishable from "never dispatched" (`byMode.stacked.count` stayed flat while Ultra climbed). Such runs are now recorded with `skip_reason` and surfaced as a per-mode `skipped` count plus `totalSkipped`/`bySkipReason` in the analytics summary and the Mode Breakdown; the existing net-saving totals/averages are unchanged (skip rows are excluded from them) (#4268 — thanks @abdulkadirozyurt, @androw) +- **cli (tray):** fix `omniroute server --tray` showing no tray on macOS/Linux with no error printed. The wired Unix tray path loaded `systray2` through an inline loader that called `require("module")` inside an ESM `.mjs` file (`"type":"module"`) → `ReferenceError: require is not defined`, silently swallowed (regressed in v3.8.34); even if it had loaded, `systray2` isn't in `node_modules` (it's lazily installed into `~/.omniroute/runtime`). The loader now delegates to the runtime loader, the icon path (`icon.png`) is corrected, `isTemplateIcon` is `false` (the full-color icon rendered as a white square under macOS template mode), and tray start failures are surfaced to stderr instead of being swallowed (#4605 — thanks @ProgMEM-CC) +- **agent-bridge (antigravity):** unwrap the cloudcode-pa `.request` envelope when converting Antigravity IDE requests. The real IDE sends `cloudcode-pa.googleapis.com/v1internal:generateContent` with the Gemini request nested under `.request` (`{ project, model, request: { contents, systemInstruction, generationConfig } }`), but the bridge read those fields at the top level — yielding an empty conversation, so prompts hung mid-execution. The legacy `/v1beta/models/:generateContent` top-level shape still works (#4294 — thanks @shabeer) +- **dashboard:** add a GitHub releases fallback to the "Update Available" lookup. After the v3.8.28 fix added an npm-registry HTTP fallback, the banner could still stay hidden on networks that reach GitHub (where the news feed already loads) but not `registry.npmjs.org`. `resolveLatestVersion()` now tries npm CLI → npm registry → GitHub releases (`/repos/diegosouzapw/OmniRoute/releases/latest`) before giving up, and logs a warning only when all three fail (#4100) +- **command-code:** omit `max_tokens` when the client omits it so the upstream applies the model's native default, fixing `400 "expected <=200000"` on `/alpha/generate` for high-cap models; an explicit oversized client value is clamped to the 200k endpoint ceiling (#5221 — thanks @adivekar-utexas) +- **combo:** wire session stickiness into the round-robin dispatch path. Multi-turn conversations from clients that send no session id (Codex CLI, Claude Code, most OpenAI-compatible tools) were rotated to a different connection on every turn by round-robin combos, busting the upstream prompt-cache → cold high-reasoning starts, intermittent `504`s and throughput collapse under concurrency. The weighted/priority paths already honored per-conversation stickiness; the round-robin handler returned before reaching it. Round-robin now starts the rotation at the conversation's sticky connection (failover to the other targets is preserved), and different conversations still spread across connections — only intra-conversation rotation is removed ([#5248](https://github.com/diegosouzapw/OmniRoute/pull/5248), #3825 — thanks @bypanghu, @jpsn123, @xz-dev) +- **kiro:** replace the synthesized trailing `"Continue"` turn with a neutral filler (`"..."`) — when an OpenAI→Kiro request ends on an assistant/tool turn, the translator synthesizes the protocol-required trailing user turn, and the literal word `"Continue"` could be read by Kiro/CodeWhisperer as a real user instruction and trigger unintended agent action. A trailing tool-result turn is still promoted as-is (it already collapses to a real user turn); only the assistant-text-ending case is affected. Regression guards: `tests/unit/kiro-continue-filler-5231.test.ts`. ([#5231](https://github.com/diegosouzapw/OmniRoute/issues/5231)) +- **combo:** advance to the next combo target on a `400 "requested model is not supported"` instead of hard-failing. The 400 guard in the priority strategy treated `MODEL_CAPACITY` as a block-fallback reason, so a combo that hit a provider lacking a specific model returned a hard `400` even when other targets (different providers) supported it. Such 400s now fall through to the next target. ([#5249](https://github.com/diegosouzapw/OmniRoute/pull/5249) — thanks @Chewji9875) +- **dashboard:** disabled no-auth providers no longer vanish from the All Providers page. Disabling a no-auth provider (the "No authentication required" toggle, which adds it to `blockedProviders`) silently removed its card because the page _dropped_ blocked no-auth entries from its render list — the only way back was buried under Settings → Security → Blocked Providers. The page now **partitions** no-auth entries: visible providers render as before, blocked ones appear in a "Disabled" sub-group with an **Enable** button that un-blocks them in place. Aggregates, counts and `/v1/models` still consume the visible-only list (blocked providers stay out of routing). Regression guard: `tests/unit/noauth-blocked-partition-5183.test.ts`. ([#5183](https://github.com/diegosouzapw/OmniRoute/issues/5183), follow-up from [#5166](https://github.com/diegosouzapw/OmniRoute/issues/5166) — thanks @WslzGmzs) +- **dashboard:** add a parent `/dashboard/context` page so RSC prefetches of the compression-context hub no longer 404. The route only had sub-routes (`settings`, `combos`, `ultra`, …) and no parent page, so the App Router returned 404 for the bare segment. The parent now redirects to its canonical sub-route (`/dashboard/context/settings`), honoring a legacy `?tab=` query for deep links. Regression guard: `tests/unit/dashboard/context-parent-redirect-5298.test.ts` ([#5298](https://github.com/diegosouzapw/OmniRoute/issues/5298) — thanks @KooshaPari) +- **i18n:** add the missing `sidebar.gamificationGroup` message across all 42 locales — the Gamification sidebar group referenced a `titleKey` that existed in no locale, logging `MISSING_MESSAGE: sidebar.gamificationGroup (en)` at runtime (the group still rendered via its `titleFallback`). The key is now present everywhere so the warning is gone and locale coverage is unaffected ([#5298](https://github.com/diegosouzapw/OmniRoute/issues/5298) — thanks @KooshaPari) +- **api(stream):** `/v1/chat/completions` no longer returns SSE for a non-stream OpenAI-compatible request when `stream` is omitted and the client sends `Accept: application/json, text/event-stream` — the Vercel AI SDK / OpenAI SDK non-stream signature (`doGenerate()`/`generateText()`), which then failed with `Invalid JSON response` (`Unexpected token 'd', "data: {"id"...`). The route-level Accept override (#302) and `resolveStreamFlag` now treat an Accept header that explicitly lists `application/json` as a JSON opt-in even when it also lists `text/event-stream`; only a pure `Accept: text/event-stream` (no `application/json`) still opts an omitted-`stream` request into SSE, and an explicit body `stream` value always wins. The shared decision now lives in `acceptHeaderForcesStream`. Regression guard: `tests/unit/sse-nonstream-accept-5305.test.ts`. ([#5305](https://github.com/diegosouzapw/OmniRoute/issues/5305) — thanks @md-riaz) +- **providers:** drop the retired GPT‑5.2 / GPT‑4.5 models from the direct **ChatGPT‑web** and **Codex** surfaces (OpenAI removed them there), so OmniRoute stops advertising/routing models that no longer exist. Scoped on purpose to those two providers — third‑party proxies that still expose the ids are untouched. ([#5280](https://github.com/diegosouzapw/OmniRoute/pull/5280) — thanks @backryun) +- **codex:** drop the deprecated `local_shell` hosted tool type before forwarding to OpenAI's Responses API, resolving the omni-combo `400 "The local_shell tool is no longer supported."` spike. Inbound Responses `local_shell` is still accepted and mapped to a caller-side Chat `shell` function for compatibility. ([#5250](https://github.com/diegosouzapw/OmniRoute/pull/5250), [#5256](https://github.com/diegosouzapw/OmniRoute/pull/5256) — thanks @KooshaPari) +- **antigravity:** retry excluded accounts via the fallback LRU. The combo same-model retry loop accumulates excluded Antigravity connection ids after account-level failures, but auth selection only treated a single `excludeConnectionId` as a fallback scenario — once exclusions accumulated through `excludedConnectionIds`, selection could fall back to normal sticky/priority behavior instead of LRU-selecting the next eligible account for the same model/family. Any non-empty accumulated exclude set is now treated as fallback mode. Builds on the family-scoped lockout work in #5180 (v3.8.39). ([#5222](https://github.com/diegosouzapw/OmniRoute/pull/5222) — thanks @Ardem2025) +- **grok-cli:** strip unsupported sampling params (`presencePenalty`, `frequencyPenalty`, `logprobs`, `topLogprobs`) before sending to the Grok Build API, fixing `400 'Model does not support parameter presencePenalty'` when clients (MiMoCode, Cursor, etc.) send OpenAI-style params. ([#5273](https://github.com/diegosouzapw/OmniRoute/pull/5273) — thanks @fulorgnas) +- **grok-cli:** accept the full `~/.grok/auth.json` object in the dashboard import-token endpoint. The `oauthImportTokenSchema` only accepted a bare string token while the UI sends the whole auth.json object → `400 Bad Request`; the schema now accepts the object and stores the original under `providerSpecificData.rawAuthJson` for diagnostics and token refresh. ([#5258](https://github.com/diegosouzapw/OmniRoute/pull/5258) — thanks @fulorgnas) +- **qoder:** coalesce concurrent PAT→job-token exchanges per PAT so high-concurrency / multi-agent bursts no longer stampede `openapi.qoder.sh/api/v1/jobToken/exchange` before the first exchange populates the completed-token cache; the shared exchange is also decoupled from any single caller's `AbortSignal` so one aborted waiter can't cancel it for the others. ([#5254](https://github.com/diegosouzapw/OmniRoute/pull/5254), [#5265](https://github.com/diegosouzapw/OmniRoute/pull/5265) — thanks @KooshaPari) +- **proxy:** scope the fallback reachability cache by normalized target **URL** instead of by hostname, so a failed probe for one endpoint on a shared API host no longer suppresses a later probe for a different endpoint on that same host for the full TTL (host fallback is preserved for malformed URLs). ([#5261](https://github.com/diegosouzapw/OmniRoute/pull/5261) — thanks @KooshaPari) +- **proxy:** cache failed fast-fail health probes with a short negative TTL instead of the full positive health TTL, so a single transient timeout/load blip no longer marks a working residential SOCKS5 proxy unreachable for the whole window (#5109 regression coverage added). ([#5255](https://github.com/diegosouzapw/OmniRoute/pull/5255) — thanks @KooshaPari) +- **mcp:** forward HTTP auth to internal tool fetches so MCP tools that call back into the local API surface carry the caller's authorization. ([#5218](https://github.com/diegosouzapw/OmniRoute/pull/5218) — thanks @KooshaPari) +- **logging:** preserve the **outbound provider request** headers in the detailed call-log Provider Request payload (previously the upstream **response** headers were shown there). chatCore now keeps executor-returned request headers when wrapping streaming and non-streaming responses; response headers stay scoped to the `Response`. ([#5257](https://github.com/diegosouzapw/OmniRoute/pull/5257) — thanks @rdself) +- **sse:** scope textual ``/`` tag extraction so generic OpenAI-compatible paths don't rewrite prompt-format content into `reasoning_content`; an explicit opt-in keeps tag-native families (DeepSeek-R1, QwQ) working while Antigravity/Agy stay excluded by provider/model prefix. ([#5224](https://github.com/diegosouzapw/OmniRoute/pull/5224) — thanks @rdself) +- **mcp:** break the `schemas/tools.ts ↔ schemas/toolSearch.ts` import cycle introduced when the `tool_search` defs (#5269) were extracted into their own module — `toolSearch.ts` imported `McpToolDefinition` from `tools.ts` while `tools.ts` imported `toolSearchTool` from `toolSearch.ts`, failing `check:cycles` on `release/v3.8.40`. The shared `AuditLevel` + `McpToolDefinition` types now live in a leaf `schemas/toolDefinition.ts` that both import; `tools.ts` re-exports them for backward compatibility. ([#5282](https://github.com/diegosouzapw/OmniRoute/pull/5282)) +- **compression (analytics):** record attempted-but-no-op compression runs so Stacked is no longer invisible when it saves nothing. A `compression_analytics` row was previously written only on a net-positive saving, so a Stacked pipeline that ran on already-compact context produced no row — indistinguishable from "never dispatched". Such runs are now recorded with `skip_reason` and surfaced as a per-mode `skipped` count plus `totalSkipped`/`bySkipReason`; net-saving totals/averages are unchanged (skip rows excluded). ([#5277](https://github.com/diegosouzapw/OmniRoute/pull/5277), #4268 — thanks @abdulkadirozyurt, @androw) +- **cli (tray):** fix `omniroute server --tray` showing no tray on macOS/Linux with no error printed. The Unix tray path loaded `systray2` through an inline loader that called `require("module")` inside an ESM `.mjs` file → `ReferenceError: require is not defined`, silently swallowed (regressed in v3.8.34); even if loaded, `systray2` is lazily installed into `~/.omniroute/runtime`, not `node_modules`. The loader now delegates to the runtime loader, the icon path is corrected, `isTemplateIcon` is `false` (the full-color icon rendered as a white square under macOS template mode), and tray start failures surface to stderr. ([#5276](https://github.com/diegosouzapw/OmniRoute/pull/5276), #4605 — thanks @ProgMEM-CC) +- **agent-bridge (antigravity):** unwrap the cloudcode-pa `.request` envelope when converting Antigravity IDE requests. The real IDE sends `cloudcode-pa.googleapis.com/v1internal:generateContent` with the Gemini request nested under `.request`, but the bridge read those fields at the top level — yielding an empty conversation, so prompts hung mid-execution. The legacy `/v1beta/models/:generateContent` top-level shape still works. ([#5267](https://github.com/diegosouzapw/OmniRoute/pull/5267), #4294 — thanks @shabeer) +- **dashboard:** add a GitHub releases fallback to the "Update Available" lookup. After the v3.8.28 npm-registry fallback, the banner could still stay hidden on networks that reach GitHub but not `registry.npmjs.org`. `resolveLatestVersion()` now tries npm CLI → npm registry → GitHub releases before giving up. ([#5266](https://github.com/diegosouzapw/OmniRoute/pull/5266), #4100) +- **command-code:** omit `max_tokens` when the client omits it so the upstream applies the model's native default, fixing `400 "expected <=200000"` on `/alpha/generate` for high-cap models; an explicit oversized client value is clamped to the 200k endpoint ceiling. ([#5221](https://github.com/diegosouzapw/OmniRoute/pull/5221) — thanks @adivekar-utexas) + +### 🔒 Security + +- **authz:** require auth for the `/v1beta/*` Gemini-compatible client API. `next.config.mjs` rewrote `/v1beta/:path*` → `/api/v1beta/:path*`, but `src/proxy.ts` didn't match `/v1beta` before the rewrite and `classifyRoute()` didn't classify `/api/v1beta/*` as client API — so unauthenticated `/v1beta/models/...:generateContent` traffic could reach the model-serving route without the central client-API auth policy. Both alias and rewritten forms are now classified `CLIENT_API`, enforcing Bearer auth when `REQUIRE_API_KEY` is enabled. ([#5274](https://github.com/diegosouzapw/OmniRoute/pull/5274) — thanks @rdself) +- **sentinel:** security hardening pass across request handling. ([#5241](https://github.com/diegosouzapw/OmniRoute/pull/5241) — thanks @iamedwardngo) +- **providers:** refresh impersonation User-Agents + TLS fingerprint profiles to current real-client versions; several had drifted or were inconsistent across files, a bot-detection/blocking risk. ([#5237](https://github.com/diegosouzapw/OmniRoute/pull/5237) — thanks @backryun) +- **authz (public origin):** centralize browser-mutation origin validation into `src/server/origin/publicOrigin.ts` and wire it through the authz pipeline, replacing the per-route same-origin-only check that `403`'d dashboard mutations when served behind a reverse proxy on a different public origin. The module resolves the allowed public origin from configured base-URL env vars or trusted forwarded headers (only when `OMNIROUTE_TRUST_PROXY` is set **and** the peer is loopback/LAN via peer-stamp), validates `Sec-Fetch-Site` metadata, and sanitizes `Host`/`Forwarded` inputs (rejects control chars, userinfo, path/query in Host). Regression guards: `tests/unit/authz/public-origin.test.ts` + `tests/unit/authz/pipeline.test.ts`. ([#5278](https://github.com/diegosouzapw/OmniRoute/pull/5278) — thanks @Thinkscape / @abodera) + +### 📝 Maintenance + +- **docs:** reorganize `docs/`, run an accuracy audit, and drop Node 20 from the supported matrix (rebased onto the current release tip). ([#5262](https://github.com/diegosouzapw/OmniRoute/pull/5262)) +- **providers:** remove the discontinued Gemini CLI channel — Google shut it down on 2026-06-18; the supported migration path is Antigravity. ([#5246](https://github.com/diegosouzapw/OmniRoute/pull/5246) — thanks @rdself) +- **dashboard:** add more provider icons from Lobehub. ([#5220](https://github.com/diegosouzapw/OmniRoute/pull/5220) — thanks @backryun) +- **deps:** resolve `npm install` warnings, fix a runtime `Cannot find module` crash on `omniroute serve` (the runtime-env module was missing from the npm `files` allow-list), and clear the 4 moderate `npm audit` advisories. ([#5252](https://github.com/diegosouzapw/OmniRoute/pull/5252) — thanks @yunaamelia), ([#5230](https://github.com/diegosouzapw/OmniRoute/pull/5230), #5227) +- **docker:** harden the base image against container-scan CVEs and skip comment lines in the #4076 builder heap-ordering check. ([#5229](https://github.com/diegosouzapw/OmniRoute/pull/5229), [#5233](https://github.com/diegosouzapw/OmniRoute/pull/5233)) +- **ci:** Trivy advisory scan now ignores unfixed CVEs to cut Security-tab noise. ([#5235](https://github.com/diegosouzapw/OmniRoute/pull/5235)) +- **test(combo):** reconcile the #4279 stop-guard test with the #5249 advance policy. ([#5300](https://github.com/diegosouzapw/OmniRoute/pull/5300)) +- **test(targets):** add 14 unit tests for the shared `combo/targetExhaustion.ts` handler (provider-exhausted / connection-error / transient rate-limited classification), registered in the vitest discovery list. ([#5296](https://github.com/diegosouzapw/OmniRoute/pull/5296) — thanks @KooshaPari) + +--- + ## [3.8.39] — 2026-06-28 ### ✨ New Features @@ -15,6 +90,7 @@ ### 🔧 Bug Fixes +- **fix(cli): global npm install no longer fails with "Cannot find module scripts/build/runtime-env.mjs"** — the v3.8.39 heap auto-calibration fix made `bin/cli/commands/serve.mjs` import `scripts/build/runtime-env.mjs`, but that file was never added to the `files` whitelist in `package.json`, so the published npm tarball shipped the importing CLI without the imported module — breaking **every** `npm install -g omniroute` at startup (regression, all platforms). The file is now whitelisted (`npm pack` confirms it ships) and added to the pack-artifact policy allow/required lists, and a new regression test scans all `bin/**` entrypoints for runtime imports resolving under `scripts/` and asserts each is covered by the package `files` whitelist, so any future unpackaged CLI import fails CI instead of users' installs. ([#5227](https://github.com/diegosouzapw/OmniRoute/issues/5227) — thanks @PriyomSaha, @m-Yaghoubi, @jonlwheat2-gif) - **fix(oauth): Antigravity refresh no longer nulls the stored refresh_token on an empty upstream response** — Google's OAuth token endpoint uses non-rotating refresh tokens: a refresh response normally OMITS `refresh_token` and occasionally returns it as an empty string. The Antigravity executor's `refreshCredentials` used `typeof tokens.refresh_token === "string" ? tokens.refresh_token : credentials.refreshToken`, and because `typeof "" === "string"` is true, an empty-string response overwrote the good token with `""` — nulling it on first refresh. The check now treats a non-string **or empty** value as absent and preserves the stored token, matching the canonical `refreshGoogleToken` (`tokens.refresh_token || refreshToken`) semantics. ([#3850](https://github.com/diegosouzapw/OmniRoute/issues/3850) — thanks @3xa228148) - **fix(api): LAN/Tailscale dashboard access — `ws:` CSP scheme, GET-exempt version route, surface combo field errors** — three failures when opening the dashboard from a non-loopback host: (1) CSP `connect-src` allowed the `ws:` scheme only for loopback origins, blocking the dashboard's `ws://:*` Live WebSocket from LAN/Tailscale clients; the bare `ws:` scheme is now permitted (symmetric with the bare `wss:` already allowed), kept declarative in `next.config.mjs` with no global middleware (the project has none by design); (2) `GET /api/system/version` was blocked by `LOCAL_ONLY_API_PREFIXES` for all methods despite only `POST` spawning child processes (git/npm/pm2) — a new `LOCAL_ONLY_API_GET_EXEMPTIONS` set exempts safe read methods for this path while keeping `POST`/`PUT`/`PATCH`/`DELETE` strictly loopback-only; (3) `COMBO_002` validation errors only surfaced the generic message — `firstField`/`firstMessage` are now extracted from the first Zod issue and included in the response body. ([#5083](https://github.com/diegosouzapw/OmniRoute/issues/5083) — thanks @KooshaPari for the diagnosis and original PR #5084) - **fix(sse): defer `` close so it never leaks before `tool_calls` in Claude→OpenAI streaming** — when a Claude thinking block was followed by a tool_use block, the translator unconditionally emitted a `content: ""` chunk at `content_block_stop`, injecting a spurious assistant text chunk immediately before the `tool_calls` delta and corrupting OpenAI-compatible clients (e.g. Kimi Coding). The close marker is now deferred: it is flushed at the first `text_delta` that follows the thinking block (preserving the #4633 / decolua/9router#454 behavior for Claude Code / Cursor) or at stream finish when no tool_calls were collected. Tool-use streams never get a `text_delta` after the thinking block, so `` is never emitted into content before `tool_calls`. ([#5123](https://github.com/diegosouzapw/OmniRoute/issues/5123)) @@ -45,6 +121,7 @@ ### 📝 Maintenance +- **test(docker): de-brittle the #4076 builder-stage heap-ordering test (false-failing on a comment)** — `dockerfile-build-heap-4076.test.ts` located the `npm run build` step via a raw `findIndex(/npm run build\b/)`, which matched a **comment line** in the builder stage (`# … npm run build fails with "Module not…"`, added with the v3.8.40 workspace-deps Docker fix) sitting before the `NODE_OPTIONS` heap line — making the test report the heap ceiling as set _after_ the build and fail, even though the real `RUN … npm run build` correctly follows the `ENV NODE_OPTIONS` line. Instruction matching now skips Dockerfile comment lines (`# …`), so the ordering guard checks real `ENV`/`RUN` instructions only. The Dockerfile itself was already correct; this was a base-red blocking every PR into `release/v3.8.40`. ([#4076](https://github.com/diegosouzapw/OmniRoute/issues/4076)) - **test(combo): deterministic context-relay universal-handoff coverage** — covers the universal (provider-agnostic) session-handoff path in `context-relay` (`combo.ts:2099–2139`), which previously had only a definition-order assertion and a `TODO(phase-2)`. The test drives the real pipeline via session seams (`x-session-id` → `relayOptions.sessionId` → `maybeGenerateUniversalHandoff`) without live infrastructure. ([#5168](https://github.com/diegosouzapw/OmniRoute/pull/5168)) - **test(combo): end-to-end quota-share DRR routing-decision coverage (matrix parity)** — adds the missing E2E test for the `quota-share` strategy, driving the real `handleChat` → chatCore → `selectQuotaShareTarget` → executor pipeline via in-process seams and asserting which connection is dispatched. The DRR selector already had 29 unit tests; this closes the E2E gap and brings quota-share to parity with the 17-strategy public matrix. ([#5179](https://github.com/diegosouzapw/OmniRoute/pull/5179)) - **test(combo): deterministic context-relay codex quota-handoff coverage (closes last gap)** — covers the codex-specific handoff block of `context-relay` (`combo.ts:2143–2183`), which #5168 left documented-but-untested because it requires a `codex` connection. All seams (`fetchCodexQuota`, handoff generation, session relay) are mocked deterministically without live infra. ([#5195](https://github.com/diegosouzapw/OmniRoute/pull/5195)) diff --git a/CLAUDE.md b/CLAUDE.md index 0ec5ba84ad..d21aa5a08b 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -35,22 +35,22 @@ For full test matrix, see `CONTRIBUTING.md` → "Running Tests". For deep archit ## Project at a Glance -**OmniRoute** — unified AI proxy/router. One endpoint, 231 LLM providers, auto-fallback. +**OmniRoute** — unified AI proxy/router. One endpoint, 237 LLM providers, auto-fallback. -| Layer | Location | Purpose | -| ------------- | ----------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | -| API Routes | `src/app/api/v1/` | Next.js App Router — entry points | -| Handlers | `open-sse/handlers/` | Request processing (chat, embeddings, etc) | -| Executors | `open-sse/executors/` | Provider-specific HTTP dispatch | -| Translators | `open-sse/translator/` | Format conversion (OpenAI↔Claude↔Gemini) | -| Transformer | `open-sse/transformer/` | Responses API ↔ Chat Completions | -| Services | `open-sse/services/` | Combo routing, rate limits, caching, etc | -| Database | `src/lib/db/` | SQLite domain modules (83 files, 97 migrations) | -| Domain/Policy | `src/domain/` | Policy engine, cost rules, fallback logic | -| MCP Server | `open-sse/mcp-server/` | 87 tools (33 base + memory/skill/notion/obsidian/gamification/plugin modules), 3 transports (stdio / SSE / Streamable HTTP), 30 scopes | -| A2A Server | `src/lib/a2a/` | JSON-RPC 2.0 agent protocol | -| Skills | `src/lib/skills/` | Extensible skill framework | -| Memory | `src/lib/memory/` | Persistent conversational memory | +| Layer | Location | Purpose | +| ------------- | ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------ | +| API Routes | `src/app/api/v1/` | Next.js App Router — entry points | +| Handlers | `open-sse/handlers/` | Request processing (chat, embeddings, etc) | +| Executors | `open-sse/executors/` | Provider-specific HTTP dispatch | +| Translators | `open-sse/translator/` | Format conversion (OpenAI↔Claude↔Gemini) | +| Transformer | `open-sse/transformer/` | Responses API ↔ Chat Completions | +| Services | `open-sse/services/` | Combo routing, rate limits, caching, etc | +| Database | `src/lib/db/` | SQLite domain modules (94 files, 106 migrations) | +| Domain/Policy | `src/domain/` | Policy engine, cost rules, fallback logic | +| MCP Server | `open-sse/mcp-server/` | 94 tools (34 base + memory/skill/agentSkill/pool/notion/obsidian/gamification/plugin modules), 3 transports (stdio / SSE / Streamable HTTP), 30 scopes | +| A2A Server | `src/lib/a2a/` | JSON-RPC 2.0 agent protocol | +| Skills | `src/lib/skills/` | Extensible skill framework | +| Memory | `src/lib/memory/` | Persistent conversational memory | Monorepo: `src/` (Next.js 16 app), `open-sse/` (streaming engine workspace), `electron/` (desktop app), `tests/`, `bin/` (CLI entry point). @@ -72,7 +72,7 @@ Client → /v1/chat/completions (Next.js route) API routes follow a consistent pattern: `Route → CORS preflight → Zod body validation → Optional auth (extractApiKey/isValidApiKey) → API key policy enforcement → Handler delegation (open-sse)`. No global Next.js middleware — interception is route-specific. -**Combo routing** (`open-sse/services/combo.ts`): 17 strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion). Each target calls `handleSingleModel()` which wraps `handleChatCore()` with per-target error handling and circuit breaker checks. The `fusion` strategy is the exception: it fans out to a panel of models in parallel, then a judge model synthesizes one final answer (`open-sse/services/fusion.ts`). See `docs/routing/AUTO-COMBO.md` for the 9-factor Auto-Combo scoring + the full strategy table and `docs/architecture/RESILIENCE_GUIDE.md` for the 3 resilience layers. +**Combo routing** (`open-sse/services/combo.ts`): 17 strategies (priority, weighted, fill-first, round-robin, P2C, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, context-relay, fusion). Each target calls `handleSingleModel()` which wraps `handleChatCore()` with per-target error handling and circuit breaker checks. The `fusion` strategy is the exception: it fans out to a panel of models in parallel, then a judge model synthesizes one final answer (`open-sse/services/fusion.ts`). See `docs/routing/AUTO-COMBO.md` for the 12-factor Auto-Combo scoring + the full strategy table and `docs/architecture/RESILIENCE_GUIDE.md` for the 3 resilience layers. --- @@ -327,33 +327,33 @@ connection continue serving other models. For any non-trivial change, read the matching deep-dive first: -| Area | Doc | -| --------------------------------------------- | ----------------------------------------------------------------- | -| Repo navigation | `docs/architecture/REPOSITORY_MAP.md` | -| Architecture | `docs/architecture/ARCHITECTURE.md` | -| Engineering reference | `docs/architecture/CODEBASE_DOCUMENTATION.md` | -| Auto-Combo (9-factor scoring, 17 strategies) | `docs/routing/AUTO-COMBO.md` | -| Resilience (3 mechanisms) | `docs/architecture/RESILIENCE_GUIDE.md` | -| Reasoning replay | `docs/routing/REASONING_REPLAY.md` | -| Skills framework | `docs/frameworks/SKILLS.md` | -| Memory system (FTS5 + Qdrant) | `docs/frameworks/MEMORY.md` | -| Cloud agents | `docs/frameworks/CLOUD_AGENT.md` | -| Guardrails (PII / injection / vision) | `docs/security/GUARDRAILS.md` | -| Public upstream credentials (Gemini/etc.) | `docs/security/PUBLIC_CREDS.md` | -| Error message sanitization | `docs/security/ERROR_SANITIZATION.md` | -| Evals | `docs/frameworks/EVALS.md` | -| Compliance / audit | `docs/security/COMPLIANCE.md` | -| Webhooks | `docs/frameworks/WEBHOOKS.md` | -| Authorization pipeline | `docs/architecture/AUTHZ_GUIDE.md` | -| Stealth (TLS / fingerprint) | `docs/security/STEALTH_GUIDE.md` | -| Agent protocols (A2A / ACP / Cloud) | `docs/frameworks/AGENT_PROTOCOLS_GUIDE.md` | -| MCP server | `docs/frameworks/MCP-SERVER.md` | -| A2A server | `docs/frameworks/A2A-SERVER.md` | +| Area | Doc | +| --------------------------------------------- | ------------------------------------------------------- | +| Repo navigation | `docs/architecture/REPOSITORY_MAP.md` | +| Architecture | `docs/architecture/ARCHITECTURE.md` | +| Engineering reference | `docs/architecture/CODEBASE_DOCUMENTATION.md` | +| Auto-Combo (12-factor scoring, 17 strategies) | `docs/routing/AUTO-COMBO.md` | +| Resilience (3 mechanisms) | `docs/architecture/RESILIENCE_GUIDE.md` | +| Reasoning replay | `docs/routing/REASONING_REPLAY.md` | +| Skills framework | `docs/frameworks/SKILLS.md` | +| Memory system (FTS5 + Qdrant) | `docs/frameworks/MEMORY.md` | +| Cloud agents | `docs/frameworks/CLOUD_AGENT.md` | +| Guardrails (PII / injection / vision) | `docs/security/GUARDRAILS.md` | +| Public upstream credentials (Gemini/etc.) | `docs/security/PUBLIC_CREDS.md` | +| Error message sanitization | `docs/security/ERROR_SANITIZATION.md` | +| Evals | `docs/frameworks/EVALS.md` | +| Compliance / audit | `docs/security/COMPLIANCE.md` | +| Webhooks | `docs/frameworks/WEBHOOKS.md` | +| Authorization pipeline | `docs/architecture/AUTHZ_GUIDE.md` | +| Stealth (TLS / fingerprint) | `docs/security/STEALTH_GUIDE.md` | +| Agent protocols (A2A / ACP / Cloud) | `docs/frameworks/AGENT_PROTOCOLS_GUIDE.md` | +| MCP server | `docs/frameworks/MCP-SERVER.md` | +| A2A server | `docs/frameworks/A2A-SERVER.md` | | API reference + OpenAPI | `docs/reference/API_REFERENCE.md` + `docs/openapi.yaml` | -| Provider catalog (auto-generated) | `docs/reference/PROVIDER_REFERENCE.md` | -| Release flow | `docs/ops/RELEASE_CHECKLIST.md` | -| Embedded services | `docs/frameworks/EMBEDDED-SERVICES.md` | -| Quality gates (~48 scripts, allowlist policy) | `docs/architecture/QUALITY_GATES.md` | +| Provider catalog (auto-generated) | `docs/reference/PROVIDER_REFERENCE.md` | +| Release flow | `docs/ops/RELEASE_CHECKLIST.md` | +| Embedded services | `docs/frameworks/EMBEDDED-SERVICES.md` | +| Quality gates (~48 scripts, allowlist policy) | `docs/architecture/QUALITY_GATES.md` | --- @@ -388,6 +388,31 @@ Why this matters: fixing bug A while opening bug B is worse than not fixing at a --- +## Planning & Research Artifacts (superpowers, deep-research) + +`_tasks/` is a **separate, isolated git repository** that is gitignored by the main +repo (`.gitignore` → `_tasks/`). It is the canonical home for working artifacts — +plans, specs/designs, research, hand-offs — so they stay **versioned in their own +repo** instead of polluting the main OmniRoute tree. + +**Hard rule — never write superpowers / planning / research output under `docs/` or +the repo root.** The superpowers skills ship with defaults that point at `docs/…` +(`writing-plans` → `docs/superpowers/plans/`, `brainstorming` → `docs/superpowers/specs/`). +Those defaults are **overridden here**. Whenever you invoke superpowers (or any +plan/spec/research generator) in this project, save to `_tasks/` instead, using the +same filename convention: + +| Artifact (skill) | Default (do NOT use) | Save here instead | +| ---------------------------------- | ------------------------- | ------------------------------------------------------------- | +| Plans (`writing-plans`) | `docs/superpowers/plans/` | `_tasks/superpowers/plans/YYYY-MM-DD-.md` | +| Specs / design (`brainstorming`) | `docs/superpowers/specs/` | `_tasks/superpowers/specs/YYYY-MM-DD--design.md` | +| Research (`deep-research`, ad-hoc) | `docs/research/` | `_tasks/research/…` | +| Hand-offs (`/handoff`) | — | `_tasks/hands-off/__v_sess-/` | + +When a superpowers skill announces a path like "saved to `docs/superpowers/plans/…`", +rewrite it to the `_tasks/…` equivalent before writing. Commit those artifacts inside +the `_tasks/` repo (`git -C _tasks …`), never in the main repo. + ## Git Workflow ```bash diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index b91144999d..6c3b4acb02 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -160,7 +160,7 @@ npm run test:protocols:e2e # Ecosystem compatibility tests npm run test:ecosystem -# Coverage gate: 75% statements/lines/functions, 70% branches +# Coverage gate: 60% statements/lines/functions/branches npm run test:coverage npm run coverage:report @@ -184,7 +184,7 @@ npm run test:combo:live:vps:failover # adds a real cross-provider failover s Coverage notes: - `npm run test:coverage` measures source coverage for the main unit test suite, excludes `tests/**`, and includes `open-sse/**` -- Pull requests must keep the coverage gate at **75%+** statements/lines/functions and **70%+** branches +- Pull requests must keep the coverage gate at **60%+** statements/lines/functions/branches - If a PR changes production code in `src/`, `open-sse/`, `electron/`, or `bin/`, it must add or update automated tests in the same PR - `npm run coverage:report` prints the detailed file-by-file report from the latest coverage run - `npm run test:coverage:legacy` preserves the older metric for historical comparison @@ -196,7 +196,7 @@ Before opening or merging a PR: - Run `npm run test:unit` - Run `npm run test:coverage` -- Ensure the coverage gate stays at **75%+** statements/lines/functions, **70%+** branches +- Ensure the coverage gate stays at **60%+** statements/lines/functions/branches - Include the changed or added test files in the PR description when production code changed - Check the SonarQube result on the PR when the project secrets are configured in CI diff --git a/Dockerfile b/Dockerfile index 89d0c2fad5..28568268db 100644 --- a/Dockerfile +++ b/Dockerfile @@ -71,8 +71,8 @@ ENV OMNIROUTE_USE_TURBOPACK=0 # Raise the V8 heap ceiling for the build. The webpack production optimization # pass (forced above since Turbopack panics) needs more than V8's default ceiling # (~2 GB) for a codebase this size; a memory-constrained Docker build otherwise -# dies with "FATAL ERROR: ... JavaScript heap out of memory" at `[builder] npm run -# build` (#4076). NODE_OPTIONS propagates to the spawned `next build` child +# dies with "FATAL ERROR: ... JavaScript heap out of memory" during the builder +# stage (#4076). NODE_OPTIONS propagates to the spawned `next build` child # (build-next-isolated.mjs → resolveNextBuildEnv spreads process.env). Build-only; # the runtime heap is set separately on the runner stage (OMNIROUTE_MEMORY_MB). # Override for hosts with more/less RAM: `--build-arg OMNIROUTE_BUILD_MEMORY_MB=6144`. diff --git a/README.md b/README.md index 6f6a90106b..ae9c08b8aa 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,7 @@ # 🚀 OmniRoute — The Free AI Gateway -### Never stop coding. Connect every AI tool to **231 providers** — **50+ free** — through one endpoint. +### Never stop coding. Connect every AI tool to **237 providers** — **50+ free** — through one endpoint. **Plug Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini. Auto-fallback.**
@@ -138,11 +138,11 @@ -> One endpoint. **231 providers.** Never stop building — and let OmniRoute pick the cheapest one that works. +> One endpoint. **237 providers.** Never stop building — and let OmniRoute pick the cheapest one that works. - + @@ -293,7 +293,7 @@ Result: 4 layers of fallback = zero downtime > Recent highlights from **v3.8.20 → v3.8.39**. Full history in [`CHANGELOG.md`](CHANGELOG.md). - **⚖️ Quota-Share routing** — a dedicated combo strategy that spreads load across accounts by _available quota_: Deficit-Round-Robin scheduling, per-connection `max_concurrent` with cooldown-wait queueing, multi-window usage buckets (5h / 7d / per-model), per-(key,model) caps, session stickiness for prompt-cache integrity, and proactive saturation from upstream token-usage headers. → [Resilience Guide](docs/architecture/RESILIENCE_GUIDE.md) -- **🤖 One-command CLI/agent setup** — a dedicated `setup-*` command configures each coding tool to route through OmniRoute (Claude Code, Codex, Cline, Continue, Cursor, Roo Code, Kilo Code, Crush, Goose, Qwen Code, Aider, OpenCode, Gemini CLI); `omniroute launch` / `omniroute launch-codex` are zero-config launchers. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md) +- **🤖 One-command CLI/agent setup** — a dedicated `setup-*` command configures each coding tool to route through OmniRoute (Claude Code, Codex, Cline, Continue, Cursor, Roo Code, Kilo Code, Crush, Goose, Qwen Code, Aider, OpenCode); `omniroute launch` / `omniroute launch-codex` are zero-config launchers. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md) - **🛰️ Remote mode** — drive a remote OmniRoute from any machine with scoped access tokens (`omniroute connect` / `omniroute contexts` / `omniroute tokens`), plus an `omniroute login antigravity` helper that runs Google "native/desktop" OAuth on your own machine and pastes a credential blob into a remote/VPS install (where the loopback redirect is unreachable). → [Remote Mode](docs/guides/REMOTE-MODE.md) - **🧭 Smarter auto-routing** — OpenRouter-style `auto/:` combos (e.g. `auto/coding:fast`, `auto/reasoning:pro`), a **Fusion** strategy (fan out to a panel of models in parallel, then synthesize via a judge), **task-aware routing** (best-fit connection per task type), per-request `X-Route-Model` override, live Arena-ELO + models.dev model intelligence, per-step account allowlists, provider-wildcard combo steps, nested combo-ref execution, sticky weighted selection, and `web_search`-aware routing. → [Auto-Combo](docs/routing/AUTO-COMBO.md) - **🗜️ Pluggable compression** — an async pipeline of **9 composable engines** with Compression Studios, an LLMLingua-2 ONNX engine and a heuristic/SLM two-tier **Ultra**, RTK, delegated Anthropic Context Editing, **Output Styles** (output-axis steering: terse-prose / less-code / terse-CJK), an **adaptive context-budget dial** (escalate only as far as needed to fit the context window), per-request `x-omniroute-compression` control, an opt-in offline eval harness, one-click **Headroom** proxy lifecycle management from the dashboard (Docker sidecar supported), a synthetic **compression playground** (Play lanes + A/B Compare with USD-capped fidelity verdicts), an opt-in **per-step fidelity gate** that rejects a lossy engine before it degrades the prompt, a **best-of-N candidate encoder** (GCF vs TOON — keep whichever is shorter, with an A/B bytes/token table in the studio), **CCR ranged/grep/stats retrieval** (pull an exact byte/line slice or summary of a stored block instead of re-expanding it), and a unified panel with named profiles + an active-profile selector. → [Compression](docs/compression/COMPRESSION_ENGINES.md) @@ -301,7 +301,7 @@ Result: 4 layers of fallback = zero downtime - **💸 Cost telemetry everywhere** — `X-OmniRoute-*` cost/usage headers on every endpoint (including media), a non-token cost engine, a cache-HIT `X-OmniRoute-Cost-Saved` header, and per-key USD spend quotas. → [API Reference](docs/reference/API_REFERENCE.md) - **🧠 Memory you control** — opt-in int8 vector quantization (Qdrant + sqlite-vec), memory off by default, and a per-request `x-omniroute-no-memory` header. → [Memory](docs/frameworks/MEMORY.md) - **🛡️ Security** — a prompt-injection guard across every LLM route (backed by a red-team suite), plus a free DuckDuckGo last-resort web search. → [Guardrails](docs/security/GUARDRAILS.md) -- **🤝 More providers & agents** — Cursor Cloud Agent (a 4th cloud agent), CodeBuddy CN (`copilot.tencent.com`), a Google Flow video-generation provider, new gateways **DGrid** and **Pioneer AI** (Fastino Labs), inbound **xAI Grok** translators plus **Grok Build (xAI)** with an OAuth import-token flow, GPT-4 / GPT-4o-mini on the GitHub Copilot provider, multi-model **Factory Droid**, **ZenMux Free** (session-cookie free tier), **Alibaba DashScope** text-to-video (`wan2.7-t2v`), a refreshed 231-provider catalog (OrcaRouter, Wafer AI, OpenAdapter, dit.ai, TokenRouter, …), Vertex AI media generation (speech / transcription / music / video), and one-click account import from CLIProxyAPI (`~/.cli-proxy-api/`). → [Providers](docs/reference/PROVIDER_REFERENCE.md) +- **🤝 More providers & agents** — Cursor Cloud Agent (a 4th cloud agent), CodeBuddy CN (`copilot.tencent.com`), a Google Flow video-generation provider, new gateways **DGrid** and **Pioneer AI** (Fastino Labs), inbound **xAI Grok** translators plus **Grok Build (xAI)** with an OAuth import-token flow, GPT-4 / GPT-4o-mini on the GitHub Copilot provider, multi-model **Factory Droid**, **ZenMux Free** (session-cookie free tier), **Alibaba DashScope** text-to-video (`wan2.7-t2v`), a refreshed 237-provider catalog (OrcaRouter, Wafer AI, OpenAdapter, dit.ai, TokenRouter, …), Vertex AI media generation (speech / transcription / music / video), and one-click account import from CLIProxyAPI (`~/.cli-proxy-api/`). → [Providers](docs/reference/PROVIDER_REFERENCE.md) - **⚡ Local performance & infra** — a one-click local Redis launcher (`omniroute redis up`, plus a dashboard Redis panel), one-click **Cloudflare Workers** and **Deno Deploy** relay deployers wired into the proxy pool, and an optional Bifrost Go sidecar that offloads the hottest relay path (`BIFROST_BASE_URL`, with automatic fallback to the TypeScript path on timeout). → [Environment](docs/reference/ENVIRONMENT.md)
@@ -317,7 +317,6 @@ Result: 4 layers of fallback = zero downtime
- @@ -349,7 +348,7 @@ Result: 4 layers of fallback = zero downtime -> The most complete catalog of any open-source router: **231 providers**, **50+ with a free tier**, **11 free forever**. +> The most complete catalog of any open-source router: **237 providers**, **50+ with a free tier**, **11 free forever**.
@@ -364,7 +363,6 @@ Result: 4 layers of fallback = zero downtime
- @@ -383,16 +381,16 @@ Result: 4 layers of fallback = zero downtime > Same app, your machine, your rules. From a global npm install to **your phone** via Termux. -| Platform | Install | Highlights | -| ------------------------- | -------------------------------------------- | --------------------------------------------------------- | -| 📦 **npm (global)** | `npm install -g omniroute` | One command, any OS | -| 🐳 **Docker** | `docker run … diegosouzapw/omniroute` | Multi-arch **AMD64 + ARM64** | -| 🖥️ **Desktop (Electron)** | `npm run electron:build` | Native window + system tray — **Windows / macOS / Linux** | -| 💪 **ARM** | native `arm64` | Raspberry Pi, ARM servers, Apple Silicon | -| 📱 **Android (Termux)** | `pkg install nodejs-lts && npx -y omniroute` | Runs **on your phone**, 24/7, no root | -| 📲 **PWA** | "Add to Home Screen" | Fullscreen, offline, installable from browser | -| 🧩 **OpenCode plugin** | `@omniroute/opencode-provider` | Native OpenCode integration | -| 🛠️ **From source** | `npm install && npm run dev` | Hack on it, contribute | +| Platform | Install | Highlights | +| ------------------------- | ---------------------------------------- | --------------------------------------------------------- | +| 📦 **npm (global)** | `npm install -g omniroute` | One command, any OS | +| 🐳 **Docker** | `docker run … diegosouzapw/omniroute` | Multi-arch **AMD64 + ARM64** | +| 🖥️ **Desktop (Electron)** | `npm run electron:build` | Native window + system tray — **Windows / macOS / Linux** | +| 💪 **ARM** | native `arm64` | Raspberry Pi, ARM servers, Apple Silicon | +| 📱 **Android (Termux)** | `pkg install nodejs && npx -y omniroute` | Runs **on your phone**, 24/7, no root | +| 📲 **PWA** | "Add to Home Screen" | Fullscreen, offline, installable from browser | +| 🧩 **OpenCode plugin** | `@omniroute/opencode-provider` | Native OpenCode integration | +| 🛠️ **From source** | `npm install && npm run dev` | Hack on it, contribute | 📖 [Docker Guide](docs/guides/DOCKER_GUIDE.md) · [Desktop](electron/README.md) · [Termux](docs/guides/TERMUX_GUIDE.md) · [PWA](docs/guides/PWA_GUIDE.md) · [OpenCode](docs/frameworks/OPENCODE.md) @@ -806,7 +804,7 @@ Compression: aggressive (~50%) → double your free quota · Cost: $0/mo **Will I be charged by OmniRoute?** No — it's free, open-source software on your machine. You only pay paid providers directly. OmniRoute has no billing system. **Are FREE providers really unlimited?** Mostly — Qoder, Pollinations, LongCat, and Cloudflare are free with no per-account credit cap. Kiro is free too but capped at ~50 credits/month per account. Stack multiple free providers in a combo and auto-fallback keeps you serving for $0. **Will compression hurt quality?** No — it only compresses the **input**; code, URLs, JSON are always protected. -**Does it work where AI is blocked?** Yes — 3-level proxy + 1proxy marketplace reach all 231 providers. +**Does it work where AI is blocked?** Yes — 3-level proxy + 1proxy marketplace reach all 237 providers. 📖 [User Guide](docs/guides/USER_GUIDE.md) · [API Reference](docs/reference/API_REFERENCE.md) · [Environment Config](docs/reference/ENVIRONMENT.md) diff --git a/bin/_ops-common.sh b/bin/_ops-common.sh index 820efccc30..96f0c1b78c 100644 --- a/bin/_ops-common.sh +++ b/bin/_ops-common.sh @@ -1,8 +1,8 @@ # bin/_ops-common.sh — shared helpers for the OmniRoute ops runbook scripts. # # Sourced (not executed) by rollback.sh / snapshot-data.sh / restore-data.sh / -# restore-policies.sh / cold-start-bench.sh. The runbook context lives in -# docs/INCIDENT_RESPONSE.md and docs/PERF_BUDGETS.md. +# restore-policies.sh / cold-start-bench.sh — the self-hoster incident-recovery +# and cold-start ops tooling. Each script documents its own contract via --help. # # Path resolution mirrors the app (src/lib/db/core.ts): the SQLite store is # $DATA_DIR/storage.sqlite and managed backups go to $DATA_DIR/db_backups diff --git a/bin/cli/commands/registry.mjs b/bin/cli/commands/registry.mjs index c06827219b..a2a91bb8c8 100644 --- a/bin/cli/commands/registry.mjs +++ b/bin/cli/commands/registry.mjs @@ -71,7 +71,6 @@ import { registerSetupCrush } from "./setup-crush.mjs"; import { registerSetupGoose } from "./setup-goose.mjs"; import { registerSetupQwen } from "./setup-qwen.mjs"; import { registerSetupAider } from "./setup-aider.mjs"; -import { registerSetupGemini } from "./setup-gemini.mjs"; import { registerConnect } from "./connect.mjs"; import { registerContexts } from "./contexts.mjs"; import { registerTokens } from "./tokens.mjs"; @@ -154,7 +153,6 @@ export function registerCommands(program) { registerSetupGoose(program); registerSetupQwen(program); registerSetupAider(program); - registerSetupGemini(program); registerConnect(program); registerContexts(program); registerTokens(program); diff --git a/bin/cli/commands/serve.mjs b/bin/cli/commands/serve.mjs index 45c6a06a30..eed761c693 100644 --- a/bin/cli/commands/serve.mjs +++ b/bin/cli/commands/serve.mjs @@ -313,7 +313,7 @@ async function maybeStartTray(port, apiPort, supervisor) { if (!isTraySupported()) return; const { default: open } = await import("open").catch(() => ({ default: null })); const dashboardUrl = `http://localhost:${port}`; - const tray = initTray({ + const tray = await initTray({ port, onQuit: () => { killTrayIfActive(); @@ -329,8 +329,12 @@ async function maybeStartTray(port, apiPort, supervisor) { const { killTray } = await import("../tray/index.mjs"); _killTray = killTray; } - } catch { - // tray is optional — do not fail the server + } catch (err) { + // tray is optional — do not fail the server, but surface why it failed so + // "--tray shows nothing" is diagnosable instead of silent (#4605). + process.stderr.write( + `[omniroute][tray] failed to start: ${err?.message ?? String(err)}\n` + ); } } diff --git a/bin/cli/commands/setup-gemini.mjs b/bin/cli/commands/setup-gemini.mjs deleted file mode 100644 index 3db242336c..0000000000 --- a/bin/cli/commands/setup-gemini.mjs +++ /dev/null @@ -1,148 +0,0 @@ -/** - * omniroute setup-gemini — point the Gemini CLI at OmniRoute's Gemini endpoint. - * - * The Gemini CLI is NOT OpenAI-compatible — it speaks the native Gemini API. - * OmniRoute exposes a Gemini-native surface at /v1beta (e.g. - * /v1beta/models/:generateContent), so the CLI can target it via the - * @google/genai SDK env `GOOGLE_GEMINI_BASE_URL` (ROOT — the SDK appends /v1beta) - * + `GEMINI_API_KEY`. There is no settings.json key for the base URL, so this is - * primarily an env recipe; we optionally write ~/.gemini/settings.json `model`. - * - * ⚠ Known Gemini CLI caveat: it may ignore GOOGLE_GEMINI_BASE_URL if a cached - * Google login exists — run `gemini` logged-out / API-key-only for it to take. - */ - -import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs"; -import { join } from "node:path"; -import os from "node:os"; -import { printHeading, printInfo, printSuccess, printError, createPrompt } from "../io.mjs"; -import { resolveActiveContext } from "../contexts.mjs"; - -function stripToRoot(url) { - const s = String(url || "").replace(/\/+$/, ""); - return s.endsWith("/v1beta") ? s.slice(0, -7) : s.endsWith("/v1") ? s.slice(0, -3) : s; -} - -/** Resolve GOOGLE_GEMINI_BASE_URL (ROOT — SDK appends /v1beta) + apiKey. */ -export function resolveGeminiTarget(opts = {}) { - let root; - if (opts.remote) root = stripToRoot(opts.remote); - else { - try { - root = stripToRoot(resolveActiveContext(opts.context ?? process.env.OMNIROUTE_CONTEXT)?.baseUrl); - } catch { - /* none */ - } - if (!root) root = `http://localhost:${Number(opts.port ?? process.env.PORT ?? 20128) || 20128}`; - } - let apiKey = opts.apiKey ?? opts["api-key"]; - if (!apiKey) { - try { - const c = resolveActiveContext(opts.context ?? process.env.OMNIROUTE_CONTEXT); - apiKey = c?.accessToken || c?.apiKey; - } catch { - /* none */ - } - } - if (!apiKey) apiKey = process.env.OMNIROUTE_API_KEY || ""; - return { baseUrl: root, apiKey }; -} - -/** The guaranteed env recipe (pure → testable). */ -export function buildGeminiRecipe({ baseUrl, model }) { - return [ - `export GOOGLE_GEMINI_BASE_URL=${baseUrl}`, - "export GEMINI_API_KEY=$OMNIROUTE_API_KEY", - `export GEMINI_MODEL=${model}`, - `gemini -p "reply OK" # or: gemini (interactive)`, - ].join("\n"); -} - -/** Merge the model into ~/.gemini/settings.json (base URL is env-only). */ -export function buildGeminiSettings(existing, { model }) { - const s = existing && typeof existing === "object" ? { ...existing } : {}; - if (model) s.model = model; - return s; -} - -function readJson(path) { - try { - if (existsSync(path)) return JSON.parse(readFileSync(path, "utf8")); - } catch { - /* corrupt/missing */ - } - return {}; -} - -async function fetchGeminiModelIds(baseUrl, apiKey) { - try { - const res = await fetch(`${baseUrl}/v1beta/models`, { - headers: { "x-goog-api-key": apiKey || "" }, - signal: AbortSignal.timeout(8000), - }); - if (!res.ok) return []; - const body = await res.json(); - return (body.models || []).map((m) => String(m.name || "").replace(/^models\//, "")).filter(Boolean); - } catch { - return []; - } -} - -export async function runSetupGeminiCommand(opts = {}) { - const { baseUrl, apiKey } = resolveGeminiTarget(opts); - const dryRun = Boolean(opts.dryRun ?? opts["dry-run"]); - const configPath = opts.configPath ?? opts["config-path"] ?? join(os.homedir(), ".gemini", "settings.json"); - - printHeading("OmniRoute → Gemini CLI (native Gemini /v1beta endpoint)"); - printInfo(`GOOGLE_GEMINI_BASE_URL: ${baseUrl} (root — SDK appends /v1beta)`); - - let model = opts.model; - if (!model) { - const ids = await fetchGeminiModelIds(baseUrl, apiKey); - if (ids.length && !opts.yes) { - printInfo(`Examples: ${ids.slice(0, 20).join(", ")}${ids.length > 20 ? " …" : ""}`); - const prompt = createPrompt(); - try { - model = await prompt.ask("Model id for Gemini CLI"); - } finally { - prompt.close(); - } - } - } - if (!model) { - printError("A model is required. Pass --model ."); - return 2; - } - - if (dryRun) { - console.log(`\n── [dry-run] ${configPath} ── { "model": "${model}" }`); - } else { - const merged = buildGeminiSettings(readJson(configPath), { model }); - mkdirSync(join(configPath, ".."), { recursive: true }); - writeFileSync(configPath, JSON.stringify(merged, null, 2) + "\n", "utf8"); - printSuccess(`Wrote ${configPath} (model)`); - } - - printInfo("\nThe base URL is env-only for Gemini CLI — export these:"); - console.log(buildGeminiRecipe({ baseUrl, model })); - printInfo("\n⚠ If Gemini CLI ignores the base URL, you have a cached Google login —"); - printInfo(" run logged-out (API-key only) so GOOGLE_GEMINI_BASE_URL takes effect."); - return 0; -} - -export function registerSetupGemini(program) { - program - .command("setup-gemini") - .description("Point the Gemini CLI at OmniRoute's native Gemini /v1beta endpoint (env recipe + settings model)") - .option("--port ", "Local OmniRoute port (ignored when --remote is set)", "20128") - .option("--remote ", "Remote OmniRoute URL, e.g. http://192.168.0.15:20128") - .option("--api-key ", "OmniRoute API key (defaults to OMNIROUTE_API_KEY env var)") - .option("--model ", "Model id for Gemini CLI (required unless picked interactively)") - .option("--config-path ", "settings.json path (default: ~/.gemini/settings.json)") - .option("--yes", "Non-interactive: do not prompt (requires --model)") - .option("--dry-run", "Print what would be written without touching the filesystem") - .action(async (opts) => { - const code = await runSetupGeminiCommand(opts); - if (code !== 0) process.exit(code); - }); -} diff --git a/bin/cli/commands/setup-qwen.mjs b/bin/cli/commands/setup-qwen.mjs index 837b6f9c93..dd34300560 100644 --- a/bin/cli/commands/setup-qwen.mjs +++ b/bin/cli/commands/setup-qwen.mjs @@ -1,7 +1,7 @@ /** * omniroute setup-qwen — configure Qwen Code (QwenLM/qwen-code) for OmniRoute. * - * Qwen Code is a terminal AI agent (gemini-cli fork) with a file-based config at + * Qwen Code is a terminal AI agent with a file-based config at * ~/.qwen/settings.json. For a custom OpenAI-compatible endpoint it uses a * `modelProviders` entry with authType "openai", baseUrl WITH /v1, and an * `envKey` naming the env var holding the key (secret stays in the env, never the @@ -47,7 +47,9 @@ export function resolveQwenTarget(opts = {}) { /** Merge the OmniRoute modelProvider into Qwen's settings.json (preserve rest). */ export function buildQwenSettings(existing, { baseUrl, model }) { const s = existing && typeof existing === "object" ? { ...existing } : {}; - const providers = Array.isArray(s.modelProviders) ? s.modelProviders.filter((p) => p?.id !== "omniroute") : []; + const providers = Array.isArray(s.modelProviders) + ? s.modelProviders.filter((p) => p?.id !== "omniroute") + : []; providers.push({ id: "omniroute", name: "OmniRoute", @@ -82,7 +84,7 @@ async function fetchModelIds(baseUrl, apiKey) { }); if (!res.ok) return []; const body = await res.json(); - const list = Array.isArray(body) ? body : body.data ?? body.models ?? []; + const list = Array.isArray(body) ? body : (body.data ?? body.models ?? []); return list.map((m) => (typeof m === "string" ? m : m?.id)).filter(Boolean); } catch { return []; @@ -92,7 +94,8 @@ async function fetchModelIds(baseUrl, apiKey) { export async function runSetupQwenCommand(opts = {}) { const { baseUrl, apiKey } = resolveQwenTarget(opts); const dryRun = Boolean(opts.dryRun ?? opts["dry-run"]); - const configPath = opts.configPath ?? opts["config-path"] ?? join(os.homedir(), ".qwen", "settings.json"); + const configPath = + opts.configPath ?? opts["config-path"] ?? join(os.homedir(), ".qwen", "settings.json"); printHeading("OmniRoute → Qwen Code (openai-compatible)"); printInfo(`baseUrl: ${baseUrl}`); @@ -126,7 +129,9 @@ export async function runSetupQwenCommand(opts = {}) { writeFileSync(configPath, out, "utf8"); printSuccess(`Wrote ${configPath}`); } - printInfo("\nProvide the key (settings reference OMNIROUTE_API_KEY): export OMNIROUTE_API_KEY=..."); + printInfo( + "\nProvide the key (settings reference OMNIROUTE_API_KEY): export OMNIROUTE_API_KEY=..." + ); printInfo('Then run: qwen (or headless: qwen -p "reply OK")'); return 0; } @@ -134,7 +139,9 @@ export async function runSetupQwenCommand(opts = {}) { export function registerSetupQwen(program) { program .command("setup-qwen") - .description("Configure Qwen Code for OmniRoute: write ~/.qwen/settings.json (openai modelProvider)") + .description( + "Configure Qwen Code for OmniRoute: write ~/.qwen/settings.json (openai modelProvider)" + ) .option("--port ", "Local OmniRoute port (ignored when --remote is set)", "20128") .option("--remote ", "Remote OmniRoute URL, e.g. http://192.168.0.15:20128") .option("--api-key ", "OmniRoute API key (defaults to OMNIROUTE_API_KEY env var)") diff --git a/bin/cli/tray/index.mjs b/bin/cli/tray/index.mjs index ab68812591..5745062e66 100644 --- a/bin/cli/tray/index.mjs +++ b/bin/cli/tray/index.mjs @@ -5,10 +5,12 @@ let active = null; export { isTraySupported }; -export function initTray({ port, onQuit, onOpenDashboard, onShowLogs }) { +export async function initTray({ port, onQuit, onOpenDashboard, onShowLogs }) { if (!isTraySupported()) return null; const ctx = { port, onQuit, onOpenDashboard, onShowLogs }; - active = process.platform === "win32" ? initWinTray(ctx) : initSystrayUnix(ctx); + // initSystrayUnix is async: it lazily installs/loads systray2 from the runtime + // dir (trayRuntime.ts) rather than from node_modules. (#4605) + active = process.platform === "win32" ? initWinTray(ctx) : await initSystrayUnix(ctx); return active; } diff --git a/bin/cli/tray/traySystray.mjs b/bin/cli/tray/traySystray.mjs index 4c5bcb3c5e..c7916a4108 100644 --- a/bin/cli/tray/traySystray.mjs +++ b/bin/cli/tray/traySystray.mjs @@ -14,30 +14,31 @@ export function isTraySupported() { return true; } -function loadSystray2() { - const candidates = [ - () => { - const { createRequire } = require("module"); - const req = createRequire(import.meta.url); - return req("systray2").default; - }, - ]; - for (const attempt of candidates) { - try { - return attempt(); - } catch {} - } - return null; +// systray2 is NOT a static dependency — it is lazily installed into +// ~/.omniroute/runtime by trayRuntime.ts (loadSystray). The previous inline +// loader called `require("module")`, which throws `ReferenceError: require is +// not defined` in this ESM file (package "type":"module"); the throw was +// silently swallowed, so the tray never appeared on macOS/Linux with no error +// printed (#4605, regressed in v3.8.34). Delegate to the runtime loader, which +// resolves systray2 from the runtime dir and surfaces install/import failures. +async function loadSystray2() { + const { loadSystray } = await import("../runtime/trayRuntime.ts"); + return loadSystray(); } function getIconBase64() { - const iconPath = join(__dirname, "icons", "icon.png"); + // Icon ships at bin/cli/tray/icon.png — the previous "icons/icon.png" path + // never existed, so the tray was created with an empty icon (#4605). + const iconPath = join(__dirname, "icon.png"); if (existsSync(iconPath)) return readFileSync(iconPath).toString("base64"); return ""; } -export function initSystrayUnix({ port, onQuit, onOpenDashboard, onShowLogs }) { - const SysTray = loadSystray2(); +export async function initSystrayUnix( + { port, onQuit, onOpenDashboard, onShowLogs }, + loadCtor = loadSystray2 +) { + const SysTray = await loadCtor(); if (!SysTray) return null; const autostartEnabled = isAutostartEnabled(); @@ -57,7 +58,10 @@ export function initSystrayUnix({ port, onQuit, onOpenDashboard, onShowLogs }) { tray = new SysTray({ menu: { icon: getIconBase64(), - isTemplateIcon: process.platform === "darwin", + // isTemplateIcon must be false: icon.png is a full-color RGBA logo, and + // macOS template mode uses only the alpha channel → a solid white square + // (the icon looked "missing" even when the tray loaded). (PR #1080) + isTemplateIcon: false, title: "", tooltip: `OmniRoute — port ${port}`, items, diff --git a/bin/cold-start-bench.sh b/bin/cold-start-bench.sh index f879610712..d21285b3ca 100755 --- a/bin/cold-start-bench.sh +++ b/bin/cold-start-bench.sh @@ -1,6 +1,6 @@ #!/usr/bin/env bash -# bin/cold-start-bench.sh — measure OmniRoute cold-start against the budgets in -# docs/PERF_BUDGETS.md §5 (container start → HTTP listening ≤ 800 ms; first warm +# bin/cold-start-bench.sh — measure OmniRoute cold-start against the target +# budgets (container start → HTTP listening ≤ 800 ms; first warm # TTFB ≤ 200 ms). Boots the server on a throwaway port, times until # /api/health/ping answers 200, measures a warm request, and reports PASS/FAIL. set -euo pipefail @@ -13,7 +13,7 @@ Usage: bin/cold-start-bench.sh [--port ] [--start-cmd ""] [--url ] [--listen-budget-ms ] [--ttfb-budget-ms ] [-h|--help] Boots OmniRoute, times cold-start to the first /api/health/ping 200, measures -warm TTFB, and compares against the PERF_BUDGETS.md §5 budgets +warm TTFB, and compares against the cold-start budgets (listen ≤ 800 ms, TTFB ≤ 200 ms). Exits non-zero if a budget is exceeded. --url benches an already-running server instead of booting one (skips the boot diff --git a/bin/nodeRuntimeSupport.mjs b/bin/nodeRuntimeSupport.mjs index 0dfdd81424..47905f0e4f 100644 --- a/bin/nodeRuntimeSupport.mjs +++ b/bin/nodeRuntimeSupport.mjs @@ -1,7 +1,6 @@ #!/usr/bin/env node export const SECURE_NODE_LINES = Object.freeze([ - Object.freeze({ major: 20, minor: 20, patch: 2 }), Object.freeze({ major: 22, minor: 22, patch: 2 }), Object.freeze({ major: 24, minor: 0, patch: 0 }), Object.freeze({ major: 25, minor: 0, patch: 0 }), @@ -9,9 +8,9 @@ export const SECURE_NODE_LINES = Object.freeze([ ]); export const RECOMMENDED_NODE_VERSION = "24.14.1"; -export const SUPPORTED_NODE_RANGE = ">=20.20.2 <21 || >=22.22.2 <23 || >=24.0.0 <27"; +export const SUPPORTED_NODE_RANGE = ">=22.22.2 <23 || >=24.0.0 <27"; export const SUPPORTED_NODE_DISPLAY = - "Node.js 20.20.2+ (20.x LTS), 22.22.2+ (22.x LTS), 24.0.0+ (24.x LTS), 25.0.0+ (25.x), or 26.0.0+ (26.x)"; + "Node.js 22.22.2+ (22.x LTS), 24.0.0+ (24.x LTS), 25.0.0+ (25.x), or 26.0.0+ (26.x)"; function formatVersion(version) { return `${version.major}.${version.minor}.${version.patch}`; @@ -78,7 +77,7 @@ export function getNodeRuntimeWarning(version = process.versions.node) { } if (support.reason === "unreleased-major") { - return `Node.js ${support.nodeVersion} is outside the supported LTS lines. OmniRoute currently supports Node.js 20.x, 22.x, 24.x, 25.x, and 26.x.`; + return `Node.js ${support.nodeVersion} is outside the supported LTS lines. OmniRoute currently supports Node.js 22.x, 24.x, 25.x, and 26.x.`; } return `Node.js ${support.nodeVersion} is outside OmniRoute's approved secure runtime policy.`; diff --git a/bin/restore-data.sh b/bin/restore-data.sh index df907fb732..c402a32abf 100755 --- a/bin/restore-data.sh +++ b/bin/restore-data.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # bin/restore-data.sh — restore the OmniRoute SQLite data volume from a snapshot -# created by bin/snapshot-data.sh. Used by the data-layer incident runbook -# (docs/INCIDENT_RESPONSE.md §4.4) after stopping writers. +# created by bin/snapshot-data.sh. Used by the data-layer incident-recovery +# flow after stopping writers. # # Safety: takes a pre-restore snapshot of the CURRENT data, refuses to run # unattended without --yes, and verifies the snapshot before overwriting. diff --git a/bin/restore-policies.sh b/bin/restore-policies.sh index 9cf4da03ef..de1c2608aa 100755 --- a/bin/restore-policies.sh +++ b/bin/restore-policies.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # bin/restore-policies.sh — restore ONLY the API-key policy tables from a # snapshot, leaving request/audit/runtime state intact. Used by the auth-layer -# incident runbook (docs/INCIDENT_RESPONSE.md §4.3) when policies_active is empty +# incident-recovery flow when policies_active is empty # but the rest of the database is healthy (so a full restore-data is overkill). # # "Policy" tables = api_key* definition tables (the key + its limits/allowed diff --git a/bin/rollback.sh b/bin/rollback.sh index 41d7bfcd43..b2184fa50b 100755 --- a/bin/rollback.sh +++ b/bin/rollback.sh @@ -1,6 +1,6 @@ #!/usr/bin/env bash # bin/rollback.sh — roll OmniRoute back to a previous release to mitigate a bad -# deploy. Used by the incident runbook (docs/INCIDENT_RESPONSE.md §3 / §4). +# deploy. Part of the deploy-rollback incident-recovery flow. # # Methods (auto-detected; override with --method): # • npm — `npm install -g omniroute@` and, if PM2 manages it, diff --git a/bin/snapshot-data.sh b/bin/snapshot-data.sh index f312a46b9a..125f4e59d6 100755 --- a/bin/snapshot-data.sh +++ b/bin/snapshot-data.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # bin/snapshot-data.sh — consistent point-in-time snapshot of the OmniRoute data -# volume (the SQLite store under $DATA_DIR). Used by the data-layer incident -# runbook (docs/INCIDENT_RESPONSE.md §4.4) before any restore. +# volume (the SQLite store under $DATA_DIR). Used by the data-layer +# incident-recovery flow before any restore. # # Output: a directory $DB_BACKUPS_DIR/snapshot_[_`. diff --git a/open-sse/services/geminiCliHeaders.ts b/open-sse/services/geminiCliHeaders.ts deleted file mode 100644 index b958a0e9dd..0000000000 --- a/open-sse/services/geminiCliHeaders.ts +++ /dev/null @@ -1,43 +0,0 @@ -import { - getCloudCodeNodeApiClientHeader, - normalizeCloudCodeArch, - normalizeCloudCodePlatform, -} from "./cloudCodeHeaders.ts"; - -export const GEMINI_CLI_VERSION = "0.42.0"; -export const GEMINI_CLI_GOOGLE_API_NODE_CLIENT_VERSION = "10.3.0"; - -const GEMINI_CLI_LOAD_CODE_ASSIST_METADATA = Object.freeze({ - ideType: "TERMINAL", - platform: "PLATFORM_UNSPECIFIED", - pluginType: "GEMINI", -}); - -export function getGeminiCliLoadCodeAssistMetadata(): Record { - return { ...GEMINI_CLI_LOAD_CODE_ASSIST_METADATA }; -} - -export function geminiCliUserAgent(model: string): string { - const normalizedModel = model || "unknown"; - return `GeminiCLI/${GEMINI_CLI_VERSION}/${normalizedModel} (${normalizeCloudCodePlatform()}; ${normalizeCloudCodeArch()}; terminal) google-api-nodejs-client/${GEMINI_CLI_GOOGLE_API_NODE_CLIENT_VERSION}`; -} - -export function geminiCliApiClientHeader(): string { - return getCloudCodeNodeApiClientHeader(); -} - -export function getGeminiCliHeaders( - model: string, - accessToken: string, - accept: "application/json" | "*/*" -): Record { - // Order matches the native Gemini CLI fingerprint: Authorization is sent - // last so the request is indistinguishable from the official client. - return { - "Content-Type": "application/json", - "User-Agent": geminiCliUserAgent(model), - "X-Goog-Api-Client": geminiCliApiClientHeader(), - Accept: accept, - Authorization: `Bearer ${accessToken}`, - }; -} diff --git a/open-sse/services/grokTlsClient.ts b/open-sse/services/grokTlsClient.ts index 198ed9e479..cdc888f896 100644 --- a/open-sse/services/grokTlsClient.ts +++ b/open-sse/services/grokTlsClient.ts @@ -25,7 +25,7 @@ import { randomUUID } from "node:crypto"; let clientPromise: Promise | null = null; let exitHookInstalled = false; -const GROK_PROFILE = "chrome_146"; // closest Chrome profile to the UA we send +const GROK_PROFILE = "chrome_149"; // closest Chrome profile to the UA we send const DEFAULT_TIMEOUT_MS = Number.parseInt(process.env.OMNIROUTE_GROK_TLS_TIMEOUT_MS || "", 10) || 60_000; // Grace period added to the binding's wire-level timeout before our JS-level diff --git a/open-sse/services/model.ts b/open-sse/services/model.ts index 674ddbe2ce..3150750a56 100644 --- a/open-sse/services/model.ts +++ b/open-sse/services/model.ts @@ -65,10 +65,6 @@ const PROVIDER_MODEL_ALIASES: ProviderModelAliasMap = { "gemini-3.1-pro": "gemini-3.1-pro-preview", "gemini-3-1-pro": "gemini-3.1-pro-preview", }, - "gemini-cli": { - "gemini-3.1-pro": "gemini-3.1-pro-preview", - "gemini-3-1-pro": "gemini-3.1-pro-preview", - }, nvidia: { "gpt-oss-120b": "openai/gpt-oss-120b", "nvidia/gpt-oss-120b": "openai/gpt-oss-120b", diff --git a/open-sse/services/opencodeOllamaUsage.ts b/open-sse/services/opencodeOllamaUsage.ts index 7a8befa01c..25cd3c3cc9 100644 --- a/open-sse/services/opencodeOllamaUsage.ts +++ b/open-sse/services/opencodeOllamaUsage.ts @@ -283,7 +283,7 @@ async function fetchOpenCodeGoDashboardUsage( headers: { Accept: "text/html", Cookie: `auth=${normalizeOpenCodeGoAuthCookie(config.authCookie)}`, - "User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) Gecko/20100101 Firefox/148.0", + "User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) Gecko/20100101 Firefox/152.0", }, signal: AbortSignal.timeout(10_000), }); @@ -508,7 +508,7 @@ async function fetchOllamaCloudUsageFromSettings( headers: { Accept: "text/html", Cookie: `${OLLAMA_CLOUD_SESSION_COOKIE}=${normalizeOllamaCloudCookie(config.cookie)}`, - "User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) Gecko/20100101 Firefox/148.0", + "User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) Gecko/20100101 Firefox/152.0", }, signal: AbortSignal.timeout(10_000), }); diff --git a/open-sse/services/payloadRules.ts b/open-sse/services/payloadRules.ts index 98ad426ee1..12bb32151d 100644 --- a/open-sse/services/payloadRules.ts +++ b/open-sse/services/payloadRules.ts @@ -399,7 +399,7 @@ export function resolvePayloadRuleProtocols({ if (targetFormat === "openai-responses" || targetFormat === "openai-response") { protocols.add("openai"); } - if (targetFormat === "gemini-cli" || targetFormat === "antigravity") { + if (targetFormat === "antigravity") { protocols.add("gemini"); } diff --git a/open-sse/services/provider.ts b/open-sse/services/provider.ts index fddf0daae8..9149c7e70b 100644 --- a/open-sse/services/provider.ts +++ b/open-sse/services/provider.ts @@ -285,7 +285,7 @@ export function buildProviderUrl( if (entry.urlBuilder) return entry.urlBuilder(baseUrl, model, stream); return baseUrl; } - // Custom URL builder (e.g. gemini, gemini-cli) + // Custom URL builder (e.g. gemini, antigravity) if (entry.urlBuilder) { const baseUrl = entry.baseUrl || config.baseUrl; if (baseUrl) { diff --git a/open-sse/services/qoderCli.ts b/open-sse/services/qoderCli.ts index 507e92af8c..b8674483d0 100644 --- a/open-sse/services/qoderCli.ts +++ b/open-sse/services/qoderCli.ts @@ -407,6 +407,10 @@ const QODER_JOB_TOKEN_MIN_TTL_MS = 60 * 1000; type QoderJobTokenCacheEntry = { jobToken: string; expiresAt: number }; const qoderJobTokenCache = new Map(); +const qoderJobTokenPending = new Map< + string, + Promise<{ jobToken: string; expiresInMs: number } | null> +>(); type FetchLike = (input: string, init?: Record) => Promise; @@ -482,7 +486,14 @@ export async function resolveQoderJobToken( const cached = qoderJobTokenCache.get(trimmed); if (cached && cached.expiresAt > now) return cached.jobToken; - const exchanged = await exchangeQoderJobToken(trimmed, options); + let pending = qoderJobTokenPending.get(trimmed); + if (!pending) { + pending = exchangeQoderJobToken(trimmed, { fetchImpl: options.fetchImpl }).finally(() => { + qoderJobTokenPending.delete(trimmed); + }); + qoderJobTokenPending.set(trimmed, pending); + } + const exchanged = await pending; if (!exchanged) return trimmed; // graceful fallback — keep prior behavior qoderJobTokenCache.set(trimmed, { jobToken: exchanged.jobToken, @@ -494,6 +505,7 @@ export async function resolveQoderJobToken( /** Test-only: clear the job-token cache so unit tests don't leak state. */ export function __clearQoderJobTokenCache(): void { qoderJobTokenCache.clear(); + qoderJobTokenPending.clear(); } export async function validateQoderCliPat({ diff --git a/open-sse/services/sessionPool/fingerprintRotator.ts b/open-sse/services/sessionPool/fingerprintRotator.ts index bc3dc248c5..479a3b808a 100644 --- a/open-sse/services/sessionPool/fingerprintRotator.ts +++ b/open-sse/services/sessionPool/fingerprintRotator.ts @@ -14,40 +14,39 @@ const PROFILES: Fingerprint[] = [ { id: "chrome-mac", userAgent: - "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/135.0.0.0 Safari/537.36", + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36", acceptLanguage: "en-US,en;q=0.9", - secChUa: '"Not-A.Brand";v="99", "Chromium";v="135", "Google Chrome";v="135"', + secChUa: '"Not-A.Brand";v="99", "Chromium";v="149", "Google Chrome";v="149"', secChUaPlatform: '"macOS"', secChUaMobile: "?0", }, { id: "chrome-linux", userAgent: - "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/135.0.0.0 Safari/537.36", + "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36", acceptLanguage: "en-US,en;q=0.9", - secChUa: '"Not-A.Brand";v="99", "Chromium";v="135", "Google Chrome";v="135"', + secChUa: '"Not-A.Brand";v="99", "Chromium";v="149", "Google Chrome";v="149"', secChUaPlatform: '"Linux"', secChUaMobile: "?0", }, { id: "chrome-win", userAgent: - "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/135.0.0.0 Safari/537.36", + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36", acceptLanguage: "en-US,en;q=0.9", - secChUa: '"Not-A.Brand";v="99", "Chromium";v="135", "Google Chrome";v="135"', + secChUa: '"Not-A.Brand";v="99", "Chromium";v="149", "Google Chrome";v="149"', secChUaPlatform: '"Windows"', secChUaMobile: "?0", }, { id: "firefox-mac", userAgent: - "Mozilla/5.0 (Macintosh; Intel Mac OS X 10.15; rv:148.0) Gecko/20100101 Firefox/148.0", + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10.15; rv:152.0) Gecko/20100101 Firefox/152.0", acceptLanguage: "en-US,en;q=0.9", }, { id: "firefox-linux", - userAgent: - "Mozilla/5.0 (X11; Linux x86_64; rv:148.0) Gecko/20100101 Firefox/148.0", + userAgent: "Mozilla/5.0 (X11; Linux x86_64; rv:152.0) Gecko/20100101 Firefox/152.0", acceptLanguage: "en-US,en;q=0.9", }, { @@ -59,18 +58,18 @@ const PROFILES: Fingerprint[] = [ { id: "edge-mac", userAgent: - "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/135.0.0.0 Safari/537.36 Edg/135.0.0.0", + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36 Edg/149.0.0.0", acceptLanguage: "en-US,en;q=0.9", - secChUa: '"Not-A.Brand";v="99", "Chromium";v="135", "Microsoft Edge";v="135"', + secChUa: '"Not-A.Brand";v="99", "Chromium";v="149", "Microsoft Edge";v="149"', secChUaPlatform: '"macOS"', secChUaMobile: "?0", }, { id: "edge-win", userAgent: - "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/135.0.0.0 Safari/537.36 Edg/135.0.0.0", + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36 Edg/149.0.0.0", acceptLanguage: "en-US,en;q=0.9", - secChUa: '"Not-A.Brand";v="99", "Chromium";v="135", "Microsoft Edge";v="135"', + secChUa: '"Not-A.Brand";v="99", "Chromium";v="149", "Microsoft Edge";v="149"', secChUaPlatform: '"Windows"', secChUaMobile: "?0", }, @@ -104,10 +103,7 @@ export class FingerprintRotator { } /** Build a Headers object from a fingerprint (with optional extras) */ - buildHeaders( - fingerprint: Fingerprint, - extra?: Record, - ): Record { + buildHeaders(fingerprint: Fingerprint, extra?: Record): Record { const headers: Record = { "Content-Type": "application/json", Accept: "application/json, text/plain, */*", diff --git a/open-sse/services/tierConfig.ts b/open-sse/services/tierConfig.ts index 7bc5945fe8..9bd1939aa6 100644 --- a/open-sse/services/tierConfig.ts +++ b/open-sse/services/tierConfig.ts @@ -52,7 +52,6 @@ export const LEGACY_FREE_PROVIDERS: readonly string[] = [ "longcat", "cloudflare-ai", "qwen", - "gemini-cli", "nvidia-nim", "cerebras", "groq", diff --git a/open-sse/services/tierDefaults.json b/open-sse/services/tierDefaults.json index 7cd15670bd..49be7b08d3 100644 --- a/open-sse/services/tierDefaults.json +++ b/open-sse/services/tierDefaults.json @@ -19,7 +19,6 @@ "longcat", "cloudflare-ai", "qwen", - "gemini-cli", "nvidia-nim", "cerebras", "groq" diff --git a/open-sse/services/tokenRefresh.ts b/open-sse/services/tokenRefresh.ts index 2528c167bc..338ddca604 100755 --- a/open-sse/services/tokenRefresh.ts +++ b/open-sse/services/tokenRefresh.ts @@ -4,7 +4,7 @@ import { PROVIDERS, OAUTH_ENDPOINTS } from "../config/constants.ts"; import { getGitHubCopilotRefreshHeaders } from "../config/providerHeaderProfiles.ts"; import { pbkdf2Sync } from "node:crypto"; import { runWithProxyContext } from "../utils/proxyFetch.ts"; -import { serializeRefresh } from "./refreshSerializer.ts"; +import { serializeRefresh, wasRefreshTokenRotated } from "./refreshSerializer.ts"; import { WINDSURF_CONFIG } from "@/lib/oauth/constants/oauth"; import { buildGitLabOAuthEndpoints, resolveGitLabOAuthBaseUrl } from "@/lib/oauth/gitlab"; @@ -43,7 +43,6 @@ export const REFRESH_LEAD_MS: Record = { iflow: 24 * 60 * 60 * 1000, // 24 hours // Google OAuth refresh_tokens are permanent (non-rotating) — longer lead // is safe and reduces unnecessary upstream chatter. - "gemini-cli": 15 * 60 * 1000, antigravity: 15 * 60 * 1000, agy: 15 * 60 * 1000, // same Google backend as antigravity (non-rotating refresh tokens) }; @@ -172,6 +171,81 @@ export function getActiveOnPersist(): RefreshPersistFn | undefined { return onPersistStore.getStore(); } +// ── #4038: compare-and-swap (CAS) guard on the refresh persist ─────────────── +// Fix A makes [network refresh + DB write] atomic *for a single connection's +// mutex*. It does NOT protect against a THIRD writer (a sibling process, a +// concurrent HealthCheck, or a replica) landing a fresher rotation on the same +// `connection_id` between the moment the caller read the row and the moment this +// persist runs. Overwriting that fresher row reverts the sibling's rotation, the +// next caller loads the reverted (now-consumed) refresh_token, and Auth0/Anthropic +// revoke the whole token family (the 1352× claude/aa5dd5cf invalidation storm). +// +// The CAS guard carries the refresh_token the caller PRESENTED (the version token, +// since refresh_tokens rotate on every refresh) plus a `reread` of the row's +// current refresh_token. Right before persisting, `getAccessToken` re-reads and, if +// a concurrent writer already rotated the row past the presented token, SKIPS the +// persist so the DB stays at the fresher state. The caller still receives the new +// accessToken — upstream already authenticated the request; only the DB write is +// skipped. No active guard ⇒ behavior is byte-identical to before (opt-in). +type CasGuard = { + /** The refresh_token the caller presented for this refresh (CAS version token). */ + expectedRefreshToken: string | null; + /** Re-reads the CURRENT persisted refresh_token for this connection (decrypted). */ + reread: () => Promise; +}; +const casGuardStore = new AsyncLocalStorage(); +const casGuardStats = { skipped: 0, persisted: 0 }; + +export function runWithCasGuard( + guard: CasGuard | undefined | null, + fn: () => Promise +): Promise { + if (!guard) return fn(); + return casGuardStore.run(guard, fn); +} + +export function getActiveCasGuard(): CasGuard | undefined { + return casGuardStore.getStore(); +} + +/** Skip/persist counters for observability + tests. */ +export function getCasGuardStats(): { skipped: number; persisted: number } { + return { ...casGuardStats }; +} + +/** Test-only: reset the CAS counters between cases. */ +export function _resetCasGuardStats(): void { + casGuardStats.skipped = 0; + casGuardStats.persisted = 0; +} + +/** + * Returns true when the persist should be SKIPPED because a concurrent writer + * already rotated the row's refresh_token past the one we presented (CAS mismatch). + * Best-effort: any reread failure falls through to persist (never blocks recovery). + */ +async function casGuardShouldSkipPersist(log?: RefreshLogger): Promise { + const guard = getActiveCasGuard(); + if (!guard || !guard.expectedRefreshToken) return false; + let current: string | null | undefined; + try { + current = await guard.reread(); + } catch { + return false; // reread failed — fall through to persist (best-effort) + } + // wasRefreshTokenRotated is true iff both are non-empty AND current !== expected. + if (wasRefreshTokenRotated(guard.expectedRefreshToken, current)) { + casGuardStats.skipped++; + log?.warn?.( + "TOKEN_REFRESH", + "CAS guard: skipping persist — a concurrent writer already rotated the refresh_token (#4038)" + ); + return true; + } + casGuardStats.persisted++; + return false; +} + type RefreshLogger = { info?: (tag: string, message: string, data?: Record) => void; warn?: (tag: string, message: string, data?: Record) => void; @@ -537,10 +611,7 @@ export async function refreshCodebuddyCnToken( expiresIn: data.data.expiresIn, }; } catch (error) { - log?.error?.( - "TOKEN_REFRESH", - `Network error refreshing CodeBuddy CN token: ${error?.message}` - ); + log?.error?.("TOKEN_REFRESH", `Network error refreshing CodeBuddy CN token: ${error?.message}`); return null; } } @@ -1496,7 +1567,6 @@ export async function refreshCopilotToken(githubAccessToken, log, proxyConfig: u async function _getAccessTokenInternal(provider, credentials, log, proxyConfig: unknown = null) { switch (provider) { case "gemini": - case "gemini-cli": case "antigravity": case "agy": return await refreshGoogleToken( @@ -1574,7 +1644,6 @@ async function _getAccessTokenInternal(provider, credentials, log, proxyConfig: export function supportsTokenRefresh(provider) { const explicitlySupported = new Set([ "gemini", - "gemini-cli", "antigravity", "agy", "claude", @@ -1676,6 +1745,11 @@ export async function getAccessToken( // Invoke onPersist INSIDE the mutex so [network call + DB write] are one atomic step. // This prevents a concurrent waiter from reading stale credentials before the DB is updated. if (result?.accessToken && effectiveOnPersist) { + // #4038: skip the persist if a concurrent writer already rotated this row past the + // refresh_token we presented (compare-and-swap) — overwriting would revert it. + if (await casGuardShouldSkipPersist(log)) { + return result; + } try { await effectiveOnPersist(result); } catch (persistErr) { @@ -1713,6 +1787,11 @@ export async function getAccessToken( ) .then(async (result) => { if (result?.accessToken && effectiveOnPersist) { + // #4038: same compare-and-swap guard as Layer 1 — skip the persist if a concurrent + // writer already rotated this row past the refresh_token we presented. + if (await casGuardShouldSkipPersist(log)) { + return result; + } try { await effectiveOnPersist(result); } catch (persistErr) { @@ -1890,7 +1969,6 @@ export function formatProviderCredentials(provider, credentials, log) { case "antigravity": case "agy": - case "gemini-cli": return { accessToken: credentials.accessToken, refreshToken: credentials.refreshToken, diff --git a/open-sse/services/usage.ts b/open-sse/services/usage.ts index 76ac708a50..3aae5f02c6 100644 --- a/open-sse/services/usage.ts +++ b/open-sse/services/usage.ts @@ -58,9 +58,6 @@ import { // Quota / usage upstream URLs (overridable for testing or relays). const CROF_USAGE_URL = process.env.OMNIROUTE_CROF_USAGE_URL ?? "https://crof.ai/usage_api/"; -const GEMINI_CLI_USAGE_URL = - process.env.OMNIROUTE_GEMINI_CLI_USAGE_URL ?? - "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist"; const CODEWHISPERER_BASE_URL = process.env.OMNIROUTE_CODEWHISPERER_BASE_URL ?? "https://codewhisperer.us-east-1.amazonaws.com"; @@ -97,7 +94,7 @@ const CURSOR_USAGE_CONFIG = { origin: "https://cursor.com", referer: "https://cursor.com/dashboard/spending", userAgent: - "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36", }; type JsonRecord = Record; @@ -110,10 +107,6 @@ type UsageProviderConnection = JsonRecord & { projectId?: string; email?: string; }; -type SubscriptionCacheEntry = { - data: unknown; - fetchedAt: number; -}; function shouldDisplayGitHubQuota(quota: UsageQuota | null): quota is UsageQuota { if (!quota) return false; @@ -698,7 +691,6 @@ async function getCursorUsage(accessToken: string, providerSpecificData?: unknow */ export const USAGE_FETCHER_PROVIDERS = [ "github", - "gemini-cli", "antigravity", "agy", "claude", @@ -745,8 +737,6 @@ export async function getUsageForProvider( switch (provider) { case "github": return await getGitHubUsage(accessToken, providerSpecificData); - case "gemini-cli": - return await getGeminiUsage(accessToken, providerSpecificData, projectId); case "antigravity": case "agy": return await getAntigravityUsage( @@ -1021,187 +1011,6 @@ function inferGitHubPlanName(data: JsonRecord, premiumQuota: UsageQuota | null): return "GitHub Copilot"; } -// ── Gemini CLI subscription info cache ────────────────────────────────────── -// Prevents duplicate loadCodeAssist calls within the same quota cycle. -// Key: accessToken → { data, fetchedAt } -const _geminiCliSubCache = new Map(); -const GEMINI_CLI_CACHE_TTL_MS = 5 * 60 * 1000; // 5 minutes - -/** - * Normalize a Cloud Code project value into a trimmed string (or null). - * The upstream `loadCodeAssist` endpoint returns the project either as a bare - * string or as an object of the form `{ id: "..." }`, and stored connection - * project ids can carry stray whitespace. Centralized here so the Gemini CLI - * usage path matches the executor/oauth normalization already shipped in - * `open-sse/executors/gemini-cli.ts` and `src/lib/oauth/services/gemini.ts`. - */ -function normalizeCloudCodeProjectId(project: unknown): string | null { - if (typeof project === "string") return project.trim() || null; - if (project && typeof project === "object") { - const candidate = (project as { id?: unknown }).id; - if (typeof candidate === "string") return candidate.trim() || null; - } - return null; -} - -/** - * Gemini CLI Usage — fetch per-model quota from Cloud Code Assist API. - * Gemini CLI and Antigravity share the same upstream (cloudcode-pa.googleapis.com), - * so this follows the same pattern as getAntigravityUsage(). - */ -async function getGeminiUsage( - accessToken?: string, - providerSpecificData?: JsonRecord, - connectionProjectId?: string -) { - if (!accessToken) { - return { plan: "Free", message: "Gemini CLI access token not available." }; - } - - try { - // #1271: the OAuth save path stores `projectId` on the connection (not always in - // `providerSpecificData`), and `loadCodeAssist` may return the project either as a - // bare string or wrapped in `{ id: "..." }`. Normalize both so the quota lookup - // reuses the stored project id and skips a redundant `loadCodeAssist` round-trip - // when it is already known. - let projectId = - normalizeCloudCodeProjectId(connectionProjectId) || - normalizeCloudCodeProjectId(providerSpecificData?.projectId); - let plan = "Free"; - - if (!projectId) { - const subscriptionInfo = await getGeminiCliSubscriptionInfoCached(accessToken); - projectId = normalizeCloudCodeProjectId(toRecord(subscriptionInfo).cloudaicompanionProject); - plan = getGeminiCliPlanLabel(subscriptionInfo); - } - - if (!projectId) { - return { - plan, - message: - "Gemini CLI project ID not available. Reconnect Gemini CLI, or configure a Google Cloud project with Gemini Code Assist access before checking quota.", - }; - } - - // Use retrieveUserQuota (same endpoint as Gemini CLI /stats command). - // Returns per-model buckets with remainingFraction and resetTime. - const response = await fetch( - "https://cloudcode-pa.googleapis.com/v1internal:retrieveUserQuota", - { - method: "POST", - headers: { - Authorization: `Bearer ${accessToken}`, - "Content-Type": "application/json", - }, - body: JSON.stringify({ project: projectId }), - signal: AbortSignal.timeout(10000), - } - ); - - if (!response.ok) { - return { plan, message: `Gemini CLI quota error (${response.status}).` }; - } - - const data = await response.json(); - const quotas: Record = {}; - - const dataRecord = toRecord(data); - if (Array.isArray(dataRecord.buckets)) { - for (const bucketValue of dataRecord.buckets) { - const bucket = toRecord(bucketValue); - if (!bucket.modelId || bucket.remainingFraction == null) continue; - - const remainingFraction = toNumber(bucket.remainingFraction, 0); - const remainingPercentage = remainingFraction * 100; - const QUOTA_NORMALIZED_BASE = 1000; - const total = QUOTA_NORMALIZED_BASE; - const remaining = Math.round(total * remainingFraction); - const used = Math.max(0, total - remaining); - - quotas[String(bucket.modelId)] = { - used, - total, - resetAt: parseResetTime(bucket.resetTime), - remainingPercentage, - unlimited: false, - }; - } - } - - return { plan, quotas }; - } catch (error) { - return { message: `Gemini CLI error: ${(error as Error).message}` }; - } -} - -/** - * Get Gemini CLI subscription info (cached, 5 min TTL) - */ -async function getGeminiCliSubscriptionInfoCached(accessToken: string): Promise { - const cacheKey = accessToken; - const cached = _geminiCliSubCache.get(cacheKey); - - if (cached && Date.now() - cached.fetchedAt < GEMINI_CLI_CACHE_TTL_MS) { - return cached.data; - } - - const data = await getGeminiCliSubscriptionInfo(accessToken); - _geminiCliSubCache.set(cacheKey, { data, fetchedAt: Date.now() }); - return data; -} - -/** - * Get Gemini CLI subscription info using correct headers. - */ -async function getGeminiCliSubscriptionInfo(accessToken: string): Promise { - try { - const response = await fetch(GEMINI_CLI_USAGE_URL, { - method: "POST", - headers: { - Authorization: `Bearer ${accessToken}`, - "Content-Type": "application/json", - }, - body: JSON.stringify({ - metadata: { - ideType: "IDE_UNSPECIFIED", - platform: "PLATFORM_UNSPECIFIED", - pluginType: "GEMINI", - }, - }), - }); - - if (!response.ok) return null; - - return await response.json(); - } catch { - return null; - } -} - -/** - * Map Gemini CLI subscription tier to display label (same tiers as Antigravity). - */ -function getGeminiCliPlanLabel(subscriptionInfo: unknown): string { - return mapCodeAssistSubscriptionToPlanLabel(subscriptionInfo); -} - -// ── Antigravity subscription info cache ────────────────────────────────────── -// ── Proactive TTL purging for the Gemini CLI subscription cache ──────────── -// Passive TTL evicts on read; this interval proactively purges stale entries so -// keys accessed once and never again don't leak memory. The Antigravity caches -// + their own purge timer were split out into ./usage/antigravity.ts (god-file -// decomposition), so each module now owns its caches and their cleanup. -const _geminiCacheCleanupTimer = setInterval( - () => { - const now = Date.now(); - for (const [key, entry] of _geminiCliSubCache) { - if (now - entry.fetchedAt > GEMINI_CLI_CACHE_TTL_MS) _geminiCliSubCache.delete(key); - } - }, - 5 * 60 * 1000 -); // every 5 minutes -_geminiCacheCleanupTimer.unref?.(); // Don't prevent process exit - /** * Claude Usage - Try to fetch from Anthropic API */ @@ -1886,7 +1695,6 @@ export const __testing = { parseResetTime, formatGitHubQuotaSnapshot, inferGitHubPlanName, - getGeminiCliPlanLabel, getAntigravityPlanLabel, extractCodeAssistSubscriptionTier, extractCodeAssistOnboardTierId, diff --git a/open-sse/services/usage/antigravity.ts b/open-sse/services/usage/antigravity.ts index 02f98b010d..99f246155c 100644 --- a/open-sse/services/usage/antigravity.ts +++ b/open-sse/services/usage/antigravity.ts @@ -5,7 +5,7 @@ * local-usage fallback, code-assist tier/plan mapping, credit-balance probing, the user-quota * + available-models fetchers (with their module-level caches), and getAntigravityUsage. The * 4 data caches + their proactive TTL-purge setInterval move here as a self-contained unit - * (previously the purge timer was shared with the Gemini CLI cache in usage.ts; that timer was + * (previously the purge timer lived in usage.ts; that timer was * split so each module owns its own caches + cleanup). usage.ts imports getAntigravityUsage * (dispatcher) + getAntigravityPlanLabel/mapCodeAssist* (__testing). Behavior-preserving move. */ diff --git a/open-sse/transformer/responsesTransformer.ts b/open-sse/transformer/responsesTransformer.ts index 0bb9156a73..afe245a019 100644 --- a/open-sse/transformer/responsesTransformer.ts +++ b/open-sse/transformer/responsesTransformer.ts @@ -1,4 +1,5 @@ import { appendToolCallArgumentDelta } from "../utils/toolCallArguments.ts"; +import { shouldParseTextualReasoningTags } from "../handlers/responseSanitizer.ts"; import * as fs from "fs"; import * as path from "path"; /** @@ -92,6 +93,7 @@ export function createResponsesApiTransformStream(logger = null, keepaliveInterv reasoningPartAdded: false, reasoningDone: false, inThinking: false, + parseTextualReasoningTags: false, funcArgsBuf: {}, funcNames: {}, funcCallIds: {}, @@ -407,6 +409,13 @@ export function createResponsesApiTransformStream(logger = null, keepaliveInterv const choice = parsed.choices[0]; const idx = choice.index || 0; const delta = choice.delta || {}; + if (state.parseTextualReasoningTags !== true && typeof parsed.model === "string") { + state.parseTextualReasoningTags = shouldParseTextualReasoningTags( + undefined, + parsed.model + ); + } + const parseTextualReasoningTags = state.parseTextualReasoningTags === true; // Emit initial events if (!state.started) { @@ -443,41 +452,45 @@ export function createResponsesApiTransformStream(logger = null, keepaliveInterv emitReasoningDelta(controller, delta.reasoning_content); } - // Handle text content (may contain tags) + // Handle text content. Generic prompt-format tags are visible text; + // only tag-native models opt into textual reasoning extraction. if (delta.content) { // Close reasoning if it was opened via native reasoning_content // and is still open, before emitting message content. Without this // the reasoning item is never closed and the message reuses the // reasoning output_index, producing a protocol-invalid stream. - // Guard on !inThinking: reasoning opened via tags is closed by - // its matching below — force-closing it here would snapshot a - // partial buffer (dense output records the item at close time). (#4848 + #4906) - if (state.reasoningId && !state.reasoningDone && !state.inThinking) { + if ( + state.reasoningId && + !state.reasoningDone && + (!parseTextualReasoningTags || !state.inThinking) + ) { closeReasoning(controller); } let content = delta.content; - if (content.includes("")) { - state.inThinking = true; - content = content.replaceAll("", ""); - startReasoning(controller, idx); - } + if (parseTextualReasoningTags) { + if (content.includes("")) { + state.inThinking = true; + content = content.replaceAll("", ""); + startReasoning(controller, idx); + } - if (content.includes("")) { - const parts = content.split(""); - const thinkPart = parts[0]; - const textPart = parts.slice(1).join(""); + if (content.includes("")) { + const parts = content.split(""); + const thinkPart = parts[0]; + const textPart = parts.slice(1).join(""); - if (thinkPart) emitReasoningDelta(controller, thinkPart); - closeReasoning(controller); - state.inThinking = false; - content = textPart; - } + if (thinkPart) emitReasoningDelta(controller, thinkPart); + closeReasoning(controller); + state.inThinking = false; + content = textPart; + } - if (state.inThinking && content) { - emitReasoningDelta(controller, content); - continue; + if (state.inThinking && content) { + emitReasoningDelta(controller, content); + continue; + } } // Regular text content diff --git a/open-sse/translator/formats.ts b/open-sse/translator/formats.ts index 5ff2098b9d..4e0bd391f0 100644 --- a/open-sse/translator/formats.ts +++ b/open-sse/translator/formats.ts @@ -5,7 +5,6 @@ export const FORMATS = { OPENAI_RESPONSE: "openai-response", CLAUDE: "claude", GEMINI: "gemini", - GEMINI_CLI: "gemini-cli", CODEX: "codex", ANTIGRAVITY: "antigravity", KIRO: "kiro", diff --git a/open-sse/translator/index.ts b/open-sse/translator/index.ts index 16b6fdf1a3..6933741489 100644 --- a/open-sse/translator/index.ts +++ b/open-sse/translator/index.ts @@ -289,8 +289,7 @@ export function translateRequest( // requested upstream; generic/implicit-cache OpenAI providers stay stripped. result = filterToOpenAIFormat(result, { preserveCacheControl: - options?.preserveCacheControl === true && - providerHonorsOpenAIFormatCacheControl(provider), + options?.preserveCacheControl === true && providerHonorsOpenAIFormatCacheControl(provider), // #4849 regression guard: keep client reasoning_content for replay providers. preserveReasoningContent: isReasoner, }); @@ -587,6 +586,7 @@ export function initState(sourceFormat) { reasoningPartAdded: false, reasoningDone: false, inThinking: false, + parseTextualReasoningTags: false, funcArgsBuf: {}, funcNames: {}, funcCallIds: {}, diff --git a/open-sse/translator/request/claude-to-gemini.ts b/open-sse/translator/request/claude-to-gemini.ts index fadc428731..f7af11ad30 100644 --- a/open-sse/translator/request/claude-to-gemini.ts +++ b/open-sse/translator/request/claude-to-gemini.ts @@ -231,6 +231,6 @@ export function claudeToGeminiRequest(model, body, stream, credentials = null) { } // Register direct path only for plain Gemini API. -// Gemini CLI / Antigravity require Cloud Code envelope wrapping, +// Antigravity requires Cloud Code envelope wrapping, // so they must use the existing hub path (Claude -> OpenAI -> target). register(FORMATS.CLAUDE, FORMATS.GEMINI, claudeToGeminiRequest, null); diff --git a/open-sse/translator/request/gemini-to-openai.ts b/open-sse/translator/request/gemini-to-openai.ts index 8be1012e74..4e968e2671 100644 --- a/open-sse/translator/request/gemini-to-openai.ts +++ b/open-sse/translator/request/gemini-to-openai.ts @@ -159,4 +159,3 @@ function extractGeminiText(content) { // Register register(FORMATS.GEMINI, FORMATS.OPENAI, geminiToOpenAIRequest, null); -register(FORMATS.GEMINI_CLI, FORMATS.OPENAI, geminiToOpenAIRequest, null); diff --git a/open-sse/translator/request/openai-responses.ts b/open-sse/translator/request/openai-responses.ts index 004d204819..a79259bd67 100644 --- a/open-sse/translator/request/openai-responses.ts +++ b/open-sse/translator/request/openai-responses.ts @@ -804,9 +804,6 @@ export function openaiToOpenAIResponsesRequest( if (tool.type === "function") { const fn = toRecord(tool.function); const name = toString(fn.name); - if (name === "shell") { - return { type: "local_shell" }; - } return { type: "function", name, @@ -827,11 +824,7 @@ export function openaiToOpenAIResponsesRequest( const tc = toRecord(root.tool_choice); if (tc.type === "function" && tc.function) { const fn = toRecord(tc.function); - if (toString(fn.name) === "shell") { - result.tool_choice = { type: "local_shell" }; - } else { - result.tool_choice = { type: "function", name: fn.name }; - } + result.tool_choice = { type: "function", name: fn.name }; } else { result.tool_choice = root.tool_choice; } diff --git a/open-sse/translator/request/openai-to-gemini.ts b/open-sse/translator/request/openai-to-gemini.ts index 09f1001a16..39a7e16018 100644 --- a/open-sse/translator/request/openai-to-gemini.ts +++ b/open-sse/translator/request/openai-to-gemini.ts @@ -16,12 +16,6 @@ import { getDefaultThinkingBudget, } from "../../../src/lib/modelCapabilities.ts"; -import * as crypto from "node:crypto"; - -function generateUUID() { - return crypto.randomUUID(); -} - import { DEFAULT_SAFETY_SETTINGS, convertOpenAIContentToParts, @@ -113,7 +107,6 @@ type CloudCodeEnvelope = { type GeminiToolNameOptions = { stripNamespace?: boolean; - functionResponseShape?: "result" | "output"; signatureNamespace?: string | null; signaturelessToolCallMode?: "native" | "text" | "context"; // Vertex AI's FunctionCall/FunctionResponse protos have no `id` field; emitting it @@ -121,7 +114,7 @@ type GeminiToolNameOptions = { // Gemini API DOES use `id` for Gemini 3+ signature matching, so this is scoped to // the vertex provider only. stripFunctionCallId?: boolean; - /** Only Antigravity/Gemini CLI support the thoughtSignature field. Standard Gemini rejects it with 400. */ + /** Antigravity supports the thoughtSignature field. Standard Gemini rejects it with 400. */ supportsSignatureBypass?: boolean; }; @@ -549,10 +542,7 @@ function openaiToGeminiBase( functionResponse: { ...(toolNameOptions.stripFunctionCallId ? {} : { id: fid }), name: name, - response: - toolNameOptions.functionResponseShape === "output" - ? { output: typeof resp === "string" ? resp : JSON.stringify(resp) } - : { result: parsedResp }, + response: { result: parsedResp }, }, }); } @@ -677,31 +667,25 @@ export function openaiToGeminiRequest( }); } -// OpenAI -> Gemini CLI (Cloud Code Assist) -export function openaiToGeminiCLIRequest( +// OpenAI -> Cloud Code Gemini payload used by Antigravity. +export function openaiToCloudCodeGeminiRequest( model: string, body: Record, stream: boolean, options: { - functionResponseShape?: "result" | "output"; signatureNamespace?: string | null; signaturelessToolCallMode?: "native" | "text" | "context"; } = {} ) { return openaiToGeminiBase(model, body, stream, { stripNamespace: true, - functionResponseShape: options.functionResponseShape, signatureNamespace: options.signatureNamespace, signaturelessToolCallMode: options.signaturelessToolCallMode, supportsSignatureBypass: true, }); } -// Wrap Gemini CLI format in Cloud Code wrapper -function wrapInCloudCodeEnvelope(model, geminiCLI, credentials = null, isAntigravity = false) { - // Both Antigravity and Gemini CLI need the project field for the Cloud Code API. - // For Gemini CLI, the stored projectId may be stale; the executor's transformRequest - // refreshes it via loadCodeAssist before the request is sent to the API. +function wrapInCloudCodeEnvelope(model, cloudCodeRequest, credentials = null) { // Fall back to providerSpecificData.projectId — some connections (and post-refresh // credentials) store it there rather than at the top level, which otherwise produced a // spurious 422 "Missing Google projectId" on the Antigravity /v1beta path (#2480). @@ -714,7 +698,7 @@ function wrapInCloudCodeEnvelope(model, geminiCLI, credentials = null, isAntigra if (!projectId) { console.warn( - `[OmniRoute] ${isAntigravity ? "Antigravity" : "GeminiCLI"} account is missing projectId. ` + + `[OmniRoute] Antigravity account is missing projectId. ` + `Attempting request with empty project — reconnect OAuth to resolve.` ); projectId = ""; @@ -722,83 +706,61 @@ function wrapInCloudCodeEnvelope(model, geminiCLI, credentials = null, isAntigra const cleanModel = model.includes("/") ? model.split("/").pop()! : model; - const envelope: CloudCodeEnvelope = isAntigravity - ? { - project: projectId, - requestId: generateAntigravityRequestId(), - request: { - sessionId: getAntigravitySessionId(credentials), - contents: geminiCLI.contents, - systemInstruction: geminiCLI.systemInstruction, - generationConfig: applyAntigravityGenerationDefaults(geminiCLI.generationConfig), - tools: geminiCLI.tools, - }, - model: cleanModel, - userAgent: getAntigravityEnvelopeUserAgent(credentials), - requestType: "agent", - enabledCreditTypes: ["GOOGLE_ONE_AI"], - } - : { - model: cleanModel, - project: projectId, - user_prompt_id: generateUUID(), - request: { - contents: geminiCLI.contents, - systemInstruction: geminiCLI.systemInstruction, - generationConfig: geminiCLI.generationConfig, - tools: geminiCLI.tools, - }, - }; - if (geminiCLI._toolNameMap instanceof Map && geminiCLI._toolNameMap.size > 0) { - envelope._toolNameMap = geminiCLI._toolNameMap; + const envelope: CloudCodeEnvelope = { + project: projectId, + requestId: generateAntigravityRequestId(), + request: { + sessionId: getAntigravitySessionId(credentials), + contents: cloudCodeRequest.contents, + systemInstruction: cloudCodeRequest.systemInstruction, + generationConfig: applyAntigravityGenerationDefaults(cloudCodeRequest.generationConfig), + tools: cloudCodeRequest.tools, + }, + model: cleanModel, + userAgent: getAntigravityEnvelopeUserAgent(credentials), + requestType: "agent", + enabledCreditTypes: ["GOOGLE_ONE_AI"], + }; + if (cloudCodeRequest._toolNameMap instanceof Map && cloudCodeRequest._toolNameMap.size > 0) { + envelope._toolNameMap = cloudCodeRequest._toolNameMap; } - // Antigravity specific fields - if (isAntigravity) { - // Inject required default system prompt for Antigravity - const defaultPart: GeminiPart = { text: ANTIGRAVITY_DEFAULT_SYSTEM }; - if (envelope.request.systemInstruction?.parts) { - envelope.request.systemInstruction.parts.unshift(defaultPart); - } else { - envelope.request.systemInstruction = { role: "system", parts: [defaultPart] }; - } - - // Strip Gemini built-in tool *names* out of functionDeclarations: Antigravity's - // v1internal endpoint returns 400 when a built-in tool (google_search etc.) is - // mixed with functionDeclarations in the same request. Native grounding entries - // (e.g. `{ googleSearch: {} }`) are left intact; only the functionDeclarations - // arrays are cleaned, and a declarations entry that becomes empty is dropped. - if (envelope.request.tools && envelope.request.tools.length > 0) { - const cleanedTools = envelope.request.tools - .map((tool) => { - if (!Array.isArray(tool.functionDeclarations)) { - return tool; - } - const customDecls = tool.functionDeclarations.filter( - (fn) => !GEMINI_BUILTIN_TOOL_NAMES.has(fn.name) - ); - return { ...tool, functionDeclarations: customDecls }; - }) - .filter( - (tool) => - !Array.isArray(tool.functionDeclarations) || tool.functionDeclarations.length > 0 - ); - envelope.request.tools = cleanedTools.length > 0 ? cleanedTools : undefined; - } - - // Add toolConfig for Antigravity only when custom functionDeclarations remain. - const hasCustomTools = envelope.request.tools?.some( - (tool) => (tool.functionDeclarations?.length ?? 0) > 0 - ); - if (hasCustomTools) { - envelope.request.toolConfig = { - functionCallingConfig: { mode: "VALIDATED" }, - }; - } + const defaultPart: GeminiPart = { text: ANTIGRAVITY_DEFAULT_SYSTEM }; + if (envelope.request.systemInstruction?.parts) { + envelope.request.systemInstruction.parts.unshift(defaultPart); } else { - // Gemini CLI's native Cloud Code envelope uses snake_case identifiers. - envelope.request.session_id = envelope.user_prompt_id; - envelope.request.safetySettings = geminiCLI.safetySettings; + envelope.request.systemInstruction = { role: "system", parts: [defaultPart] }; + } + + // Strip Gemini built-in tool *names* out of functionDeclarations: Antigravity's + // v1internal endpoint returns 400 when a built-in tool (google_search etc.) is + // mixed with functionDeclarations in the same request. Native grounding entries + // (e.g. `{ googleSearch: {} }`) are left intact; only the functionDeclarations + // arrays are cleaned, and a declarations entry that becomes empty is dropped. + if (envelope.request.tools && envelope.request.tools.length > 0) { + const cleanedTools = envelope.request.tools + .map((tool) => { + if (!Array.isArray(tool.functionDeclarations)) { + return tool; + } + const customDecls = tool.functionDeclarations.filter( + (fn) => !GEMINI_BUILTIN_TOOL_NAMES.has(fn.name) + ); + return { ...tool, functionDeclarations: customDecls }; + }) + .filter( + (tool) => !Array.isArray(tool.functionDeclarations) || tool.functionDeclarations.length > 0 + ); + envelope.request.tools = cleanedTools.length > 0 ? cleanedTools : undefined; + } + + const hasCustomTools = envelope.request.tools?.some( + (tool) => (tool.functionDeclarations?.length ?? 0) > 0 + ); + if (hasCustomTools) { + envelope.request.toolConfig = { + functionCallingConfig: { mode: "VALIDATED" }, + }; } return envelope; @@ -831,16 +793,16 @@ export function openaiToAntigravityRequest(model, body, stream, credentials = nu typeof (credentials as Record)._signatureNamespace === "string" ? ((credentials as Record)._signatureNamespace as string) : null; - const geminiCLI = openaiToGeminiCLIRequest(model, body, stream, { + const cloudCodeRequest = openaiToCloudCodeGeminiRequest(model, body, stream, { signatureNamespace, signaturelessToolCallMode: isThinkingGemini ? "context" : "native", }); if (isClaude) { - geminiCLI.generationConfig.maxOutputTokens = getAntigravityClaudeOutputTokens(body); + cloudCodeRequest.generationConfig.maxOutputTokens = getAntigravityClaudeOutputTokens(body); } - const envelope = wrapInCloudCodeEnvelope(model, geminiCLI, credentials, true); + const envelope = wrapInCloudCodeEnvelope(model, cloudCodeRequest, credentials); // Match real Antigravity client: don't send maxOutputTokens when the user // hasn't explicitly specified max_tokens / max_completion_tokens. @@ -882,24 +844,4 @@ register( }), null ); -register( - FORMATS.OPENAI, - FORMATS.GEMINI_CLI, - (model, body, stream, credentials) => - wrapInCloudCodeEnvelope( - model, - openaiToGeminiCLIRequest(model, body, stream, { - functionResponseShape: "output", - // Forward the signature namespace so streaming thoughtSignatures round-trip (#2504). - signatureNamespace: - credentials && - typeof credentials === "object" && - typeof credentials["_signatureNamespace"] === "string" - ? (credentials["_signatureNamespace"] as string) - : null, - }), - credentials - ), - null -); register(FORMATS.OPENAI, FORMATS.ANTIGRAVITY, openaiToAntigravityRequest, null); diff --git a/open-sse/translator/request/openai-to-kiro.ts b/open-sse/translator/request/openai-to-kiro.ts index 9411b33dbd..194484ded8 100644 --- a/open-sse/translator/request/openai-to-kiro.ts +++ b/open-sse/translator/request/openai-to-kiro.ts @@ -405,14 +405,14 @@ function convertMessages(messages, tools, model) { // Kiro requires currentMessage to be a user turn. If the request ends with a // user turn, move that final turn into currentMessage. If it ends with an - // assistant/tool turn, keep chronological history intact and ask Kiro to - // continue instead of reordering prior turns. + // assistant/tool turn, synthesize a neutral filler ("...") instead of the + // literal "Continue", which Kiro can read as a real instruction (#5231). if (history.length > 0 && history[history.length - 1].userInputMessage) { currentMessage = history.pop(); } else { currentMessage = { userInputMessage: { - content: "Continue", + content: "...", modelId: model, }, }; @@ -435,7 +435,7 @@ function convertMessages(messages, tools, model) { // Fallback: if the schema was never attached to any user turn (e.g. the // input contained no user messages and currentMessage is a synthesized - // "Continue" turn), attach the provided tools directly to currentMessage so + // neutral-filler turn), attach the provided tools directly to currentMessage so // Kiro still sees the schema it needs to validate assistant.toolUses in // history. if ( diff --git a/open-sse/translator/response/gemini-to-claude.ts b/open-sse/translator/response/gemini-to-claude.ts index 280d1f85cb..2019c46100 100644 --- a/open-sse/translator/response/gemini-to-claude.ts +++ b/open-sse/translator/response/gemini-to-claude.ts @@ -196,5 +196,4 @@ export function geminiToClaudeResponse(chunk, state) { // Register as direct path: Gemini → Claude register(FORMATS.GEMINI, FORMATS.CLAUDE, null, geminiToClaudeResponse); -register(FORMATS.GEMINI_CLI, FORMATS.CLAUDE, null, geminiToClaudeResponse); register(FORMATS.ANTIGRAVITY, FORMATS.CLAUDE, null, geminiToClaudeResponse); diff --git a/open-sse/translator/response/gemini-to-openai.ts b/open-sse/translator/response/gemini-to-openai.ts index ba23d68819..dbf608b148 100644 --- a/open-sse/translator/response/gemini-to-openai.ts +++ b/open-sse/translator/response/gemini-to-openai.ts @@ -297,6 +297,9 @@ export function geminiToOpenAIResponse(chunk, state) { const response = chunk.response || chunk; if (!response) return null; + const modelVersion = + typeof response.modelVersion === "string" ? response.modelVersion.toLowerCase() : ""; + const parseTextualReasoningTags = !chunk.response && !modelVersion.startsWith("antigravity/"); const results = []; const candidate = response.candidates?.[0]; @@ -438,16 +441,18 @@ export function geminiToOpenAIResponse(chunk, state) { } if (hasFunctionCall) { - // Flush any still-open textual reasoning wrapper as reasoning_content BEFORE - // the tool call. A signed native functionCall arriving while a `` - // (etc.) tag opened in an earlier chunk is still buffered must not silently - // drop that buffered reasoning — flushOpenTextualReasoning emits it and clears - // the active-tag/content buffers. (LEDGER-4 / #3821-review) - flushOpenTextualReasoning(state, results); - // Also drop any partial open-tag fragment buffered at a chunk boundary - // (flushOpenTextualReasoning early-returns when only this is set), matching the - // pre-fix branch which cleared all three buffers. (#3821-review convergence) - state.textualReasoningTagBuffer = undefined; + if (parseTextualReasoningTags) { + // Flush any still-open textual reasoning wrapper as reasoning_content BEFORE + // the tool call. A signed native functionCall arriving while a `` + // (etc.) tag opened in an earlier chunk is still buffered must not silently + // drop that buffered reasoning — flushOpenTextualReasoning emits it and clears + // the active-tag/content buffers. (LEDGER-4 / #3821-review) + flushOpenTextualReasoning(state, results); + // Also drop any partial open-tag fragment buffered at a chunk boundary + // (flushOpenTextualReasoning early-returns when only this is set), matching the + // pre-fix branch which cleared all three buffers. (#3821-review convergence) + state.textualReasoningTagBuffer = undefined; + } emitFunctionCallPart(part, state, results); } continue; @@ -459,7 +464,9 @@ export function geminiToOpenAIResponse(chunk, state) { // back to a structured OpenAI tool call so clients/tools do not see it as // assistant prose. if (part.text !== undefined && part.text !== "") { - const afterReasoning = consumeTextualReasoningTags(part.text, state, results); + const afterReasoning = parseTextualReasoningTags + ? consumeTextualReasoningTags(part.text, state, results) + : part.text; if (!afterReasoning) continue; let accumulated = (state.textualToolCallBuffer || "") + afterReasoning; @@ -677,7 +684,9 @@ export function geminiToOpenAIResponse(chunk, state) { // Finish reason - include usage in final chunk if (candidate.finishReason) { - flushOpenTextualReasoning(state, results); + if (parseTextualReasoningTags) { + flushOpenTextualReasoning(state, results); + } if (state.textualToolCallBuffer) { const remainingText = state.textualToolCallBuffer; @@ -748,5 +757,4 @@ export function geminiToOpenAIResponse(chunk, state) { // Register register(FORMATS.GEMINI, FORMATS.OPENAI, null, geminiToOpenAIResponse); -register(FORMATS.GEMINI_CLI, FORMATS.OPENAI, null, geminiToOpenAIResponse); register(FORMATS.ANTIGRAVITY, FORMATS.OPENAI, null, geminiToOpenAIResponse); diff --git a/open-sse/translator/response/kiro-to-openai.ts b/open-sse/translator/response/kiro-to-openai.ts index f58a2aaeaa..694881ade8 100644 --- a/open-sse/translator/response/kiro-to-openai.ts +++ b/open-sse/translator/response/kiro-to-openai.ts @@ -92,7 +92,6 @@ export function convertKiroToOpenAI(chunk, state) { const content = data.reasoningContentEvent?.content || data.content || ""; if (!content) return null; - // Convert to thinking block format (Claude-style) const openaiChunk = { id: state.responseId, object: "chat.completion.chunk", @@ -103,7 +102,7 @@ export function convertKiroToOpenAI(chunk, state) { index: 0, delta: { ...(state.chunkIndex === 0 ? { role: "assistant" } : {}), - content: `${content}`, + reasoning_content: content, }, finish_reason: null, }, diff --git a/open-sse/translator/response/openai-responses.ts b/open-sse/translator/response/openai-responses.ts index 90a6b74d87..d6c397fbad 100644 --- a/open-sse/translator/response/openai-responses.ts +++ b/open-sse/translator/response/openai-responses.ts @@ -6,6 +6,7 @@ import { register } from "../registry.ts"; import { FORMATS } from "../formats.ts"; import { appendToolCallArgumentDelta } from "../../utils/toolCallArguments.ts"; import { fallbackToolCallId } from "../helpers/toolCallHelper.ts"; +import { shouldParseTextualReasoningTags } from "../../handlers/responseSanitizer.ts"; function normalizeToolName(value) { return typeof value === "string" ? value.trim() : ""; @@ -87,6 +88,10 @@ export function openaiToOpenAIResponsesResponse(chunk, state) { const choice = chunk.choices[0]; const idx = choice.index || 0; const delta = choice.delta || {}; + if (state.parseTextualReasoningTags !== true && typeof chunk.model === "string") { + state.parseTextualReasoningTags = shouldParseTextualReasoningTags(undefined, chunk.model); + } + const parseTextualReasoningTags = state.parseTextualReasoningTags === true; // Emit initial events if (!state.started) { @@ -117,50 +122,45 @@ export function openaiToOpenAIResponsesResponse(chunk, state) { }); } - // Handle reasoning_content if (delta.reasoning_content) { startReasoning(state, emit, idx); emitReasoningDelta(state, emit, delta.reasoning_content); } - - // Handle text content if (delta.content) { - // Close reasoning if it was opened via native reasoning_content and is - // still open, before emitting message content. Otherwise the reasoning - // item is never closed and the message reuses its output_index. - // Guard on !inThinking: reasoning opened via tags is closed by its - // matching below — force-closing it here would snapshot a partial - // buffer (dense output records the item at close time). (#4848 + #4906) - if (state.reasoningId && !state.reasoningDone && !state.inThinking) { + if ( + state.reasoningId && + !state.reasoningDone && + (!parseTextualReasoningTags || !state.inThinking) + ) { closeReasoning(state, emit); } let content = delta.content; - if (content.includes("")) { - state.inThinking = true; - content = content.replaceAll("", ""); - startReasoning(state, emit, idx); - } + if (parseTextualReasoningTags) { + if (content.includes("")) { + state.inThinking = true; + content = content.replaceAll("", ""); + startReasoning(state, emit, idx); + } - if (content.includes("")) { - const parts = content.split(""); - const thinkPart = parts[0]; - const textPart = parts.slice(1).join(""); - if (thinkPart) emitReasoningDelta(state, emit, thinkPart); - closeReasoning(state, emit); - state.inThinking = false; - content = textPart; - } + if (content.includes("")) { + const parts = content.split(""); + const thinkPart = parts[0]; + const textPart = parts.slice(1).join(""); + if (thinkPart) emitReasoningDelta(state, emit, thinkPart); + closeReasoning(state, emit); + state.inThinking = false; + content = textPart; + } - if (state.inThinking && content) { - emitReasoningDelta(state, emit, content); - return events; + if (state.inThinking && content) { + emitReasoningDelta(state, emit, content); + return events; + } } if (content) { - // Use a distinct output_index for the message when reasoning was - // emitted, so the message item does not collide with the reasoning item. const msgIdx = state.reasoningId ? state.reasoningIndex + 1 : idx; emitTextContent(state, emit, msgIdx, content); } diff --git a/open-sse/translator/response/openai-to-gemini-sse.ts b/open-sse/translator/response/openai-to-gemini-sse.ts index 087af65ee8..430f385a05 100644 --- a/open-sse/translator/response/openai-to-gemini-sse.ts +++ b/open-sse/translator/response/openai-to-gemini-sse.ts @@ -1,7 +1,7 @@ /** * Convert an OpenAI Chat Completions stream/response into the Gemini * `:streamGenerateContent` / `:generateContent` shape used by the - * `@google/genai` SDK (Gemini CLI). + * `@google/genai` SDK. * * Why this exists * --------------- @@ -20,7 +20,7 @@ * "finishReason":"STOP","index":0}],"usageMetadata":{...},"modelVersion":"..."} * (stream closes — no [DONE]) * - * Forwarding the raw OpenAI SSE to Gemini CLI made it crash with + * Forwarding the raw OpenAI SSE to the Gemini SDK made it crash with * `SyntaxError: Unexpected token 'D', "[DONE]" is not valid JSON`, because * the SDK tries to `JSON.parse("[DONE]")`. * @@ -124,8 +124,7 @@ export function openAIChunkToGeminiChunk( }; if (choice.finish_reason) { - candidate.finishReason = - OPENAI_TO_GEMINI_FINISH_REASON[choice.finish_reason] ?? "STOP"; + candidate.finishReason = OPENAI_TO_GEMINI_FINISH_REASON[choice.finish_reason] ?? "STOP"; } const out: GeminiStreamChunk = { candidates: [candidate] }; @@ -156,10 +155,7 @@ export function openAIChunkToGeminiChunk( * Non-OK / no-body responses are passed through unchanged so that callers * upstream of the route can surface the error to the client untouched. */ -export function transformOpenAISSEToGeminiSSE( - upstreamResponse: Response, - model: string -): Response { +export function transformOpenAISSEToGeminiSSE(upstreamResponse: Response, model: string): Response { if (!upstreamResponse.ok || !upstreamResponse.body) { return upstreamResponse; } @@ -317,8 +313,7 @@ export async function convertOpenAIResponseToGemini( } parts.push({ text: String(message.content ?? "") }); - const finishReason = - OPENAI_TO_GEMINI_FINISH_REASON[finish_reason ?? "stop"] ?? "STOP"; + const finishReason = OPENAI_TO_GEMINI_FINISH_REASON[finish_reason ?? "stop"] ?? "STOP"; const geminiResponse: GeminiNonStreamResponse = { candidates: [ diff --git a/open-sse/types.d.ts b/open-sse/types.d.ts index bfb19f7854..89cb7fef48 100644 --- a/open-sse/types.d.ts +++ b/open-sse/types.d.ts @@ -30,7 +30,7 @@ export interface ProviderCredentials { } export interface ModelInfo { - /** Canonical provider ID (e.g., "claude", "gemini-cli") */ + /** Canonical provider ID (e.g., "claude", "antigravity") */ provider: string; /** Model identifier (e.g., "claude-opus-4-6") */ model: string; diff --git a/open-sse/utils/aiSdkCompat.ts b/open-sse/utils/aiSdkCompat.ts index 088e898b82..fb26394a4b 100644 --- a/open-sse/utils/aiSdkCompat.ts +++ b/open-sse/utils/aiSdkCompat.ts @@ -40,6 +40,22 @@ export function clientWantsJsonResponse(acceptHeader: unknown): boolean { return normalized.includes("application/json") && !normalized.includes("text/event-stream"); } +/** + * Route-level Accept-header streaming opt-in (#302). A client that OMITS `stream` + * in the body but sends `Accept: text/event-stream` is asking for SSE (curl/httpx + * and similar non-SDK clients). But a client that ALSO lists `application/json` + * is using the OpenAI / Vercel AI SDK non-stream signature + * (`Accept: application/json, text/event-stream` with the body omitting `stream`) + * and expects a JSON object — do NOT force SSE for it (#5305). An explicit body + * `stream` value (true or false) always wins and is never overridden. + */ +export function acceptHeaderForcesStream(acceptHeader: unknown, bodyStream: unknown): boolean { + if (bodyStream !== undefined) return false; + if (typeof acceptHeader !== "string") return false; + const normalized = acceptHeader.toLowerCase(); + return normalized.includes("text/event-stream") && !normalized.includes("application/json"); +} + /** * Resolves stream behavior from request body + Accept header. * Priority: explicit `stream: true/false` in body wins, UNLESS the provider @@ -103,6 +119,17 @@ export function resolveStreamFlag( return false; } + // An Accept header that explicitly lists `application/json` is a JSON opt-in, + // even when it ALSO lists `text/event-stream`. That is the OpenAI / Vercel AI + // SDK non-stream signature (`Accept: application/json, text/event-stream` with + // the body omitting `stream`): doGenerate()/generateText() send it and parse + // the response as JSON. Default such requests to non-stream so they don't get + // an SSE body they can't parse (#5305). Pure-SSE clients (text/event-stream + // with no application/json) and clients with no/`*/*` Accept still stream. + if (typeof acceptHeader === "string" && /application\/json/i.test(acceptHeader)) { + return false; + } + // No explicit stream param — preserve OmniRoute's streaming default unless // the client explicitly asks for JSON and does not also accept SSE. return !clientWantsJsonResponse(acceptHeader); diff --git a/open-sse/utils/cursorVersionDetector.ts b/open-sse/utils/cursorVersionDetector.ts index 931f77622b..29a6cdae6d 100644 --- a/open-sse/utils/cursorVersionDetector.ts +++ b/open-sse/utils/cursorVersionDetector.ts @@ -18,7 +18,7 @@ const DB_KEY = "cursorupdate.lastUpdatedAndShown.version"; * `CURSOR_REGISTRY_VERSION` in providerHeaderProfiles.ts. Exported so tests * assert against the single source of truth instead of a drifting literal. */ -export const FALLBACK_VERSION = "3.3"; +export const FALLBACK_VERSION = "3.9"; let cachedVersion: string | null = null; let cachedAt = 0; diff --git a/open-sse/utils/proxyFallback.ts b/open-sse/utils/proxyFallback.ts index 165ce37854..87b4a85d1b 100644 --- a/open-sse/utils/proxyFallback.ts +++ b/open-sse/utils/proxyFallback.ts @@ -4,7 +4,8 @@ * When a direct fetch to a provider fails and no explicit proxy was configured, * this module automatically gathers proxy candidates from all available sources, * tests them in parallel against the provider URL, and returns the first working one. - * Results are cached per hostname to avoid repeated probing. + * Results are cached per target URL to avoid repeated probing without letting + * a failed path poison a different endpoint on the same API host. */ import { fetch as undiciFetch } from "undici"; @@ -36,6 +37,17 @@ interface ProxyShape { const PROXY_FALLBACK_CACHE = new Map(); const CACHE_TTL_MS = 5 * 60 * 1000; // 5 minutes +type ProxyFallbackTestHooks = { + getProxyCandidates?: (targetUrl?: string) => Promise; + testSingleProxy?: ( + proxyUrl: string, + targetUrl: string, + timeoutMs?: number + ) => Promise<{ ok: boolean; latencyMs: number | null }>; +}; + +let proxyFallbackTestHooks: ProxyFallbackTestHooks | null = null; + /** * Clear the in-memory proxy fallback cache. * Useful for testing or admin operations. @@ -44,6 +56,10 @@ export function clearProxyFallbackCache(): void { PROXY_FALLBACK_CACHE.clear(); } +export function __setProxyFallbackTestHooks(hooks: ProxyFallbackTestHooks | null): void { + proxyFallbackTestHooks = hooks; +} + // --------------------------------------------------------------------------- // Helpers // --------------------------------------------------------------------------- @@ -59,6 +75,16 @@ function proxyRecordToUrl(proxy: ProxyShape): string { return `${proxy.type}://${auth}${proxy.host}:${proxy.port}`; } +function cacheKeyForTarget(targetHostname: string, targetUrl: string): string { + try { + const url = new URL(targetUrl); + const normalizedPath = `${url.pathname || "/"}${url.search}`; + return `${url.protocol}//${url.host}${normalizedPath}`; + } catch { + return targetHostname.toLowerCase(); + } +} + /** * Resolve the environment proxy URL (HTTP_PROXY / HTTPS_PROXY / ALL_PROXY) * for the given target URL. Returns null if no env proxy is configured or @@ -266,8 +292,9 @@ export async function testProxiesAgainstTarget( * Find a working proxy for the given target hostname and URL. * * Collects all proxy candidates, tests them in parallel against the provider - * URL, and returns the first one that responds. Results are cached per - * hostname for 5 minutes to avoid repeated probing. + * URL, and returns the first one that responds. Results are cached per target + * URL for 5 minutes to avoid repeated probing while keeping different + * endpoints on a shared host independent. * * @param targetHostname The provider hostname (used as cache key) * @param targetUrl The full provider URL to test against @@ -278,20 +305,23 @@ export async function findWorkingProxy( targetUrl: string ): Promise { if (!targetHostname) return null; + const cacheKey = cacheKeyForTarget(targetHostname, targetUrl); // Check cache first - const cached = PROXY_FALLBACK_CACHE.get(targetHostname); + const cached = PROXY_FALLBACK_CACHE.get(cacheKey); if (cached) { if (cached.expiresAt > Date.now()) { // Cached hit — return the proxy (or null if previously all failed) return cached.proxyUrl || null; } // Expired entry — remove it and re-probe - PROXY_FALLBACK_CACHE.delete(targetHostname); + PROXY_FALLBACK_CACHE.delete(cacheKey); } // Collect candidates - const candidates = await getProxyCandidates(targetUrl); + const candidates = await (proxyFallbackTestHooks?.getProxyCandidates ?? getProxyCandidates)( + targetUrl + ); if (candidates.length === 0) { return null; } @@ -299,7 +329,10 @@ export async function findWorkingProxy( // Test all in parallel, return first that works const results = await Promise.allSettled( candidates.map(async (proxyUrl) => { - const { ok } = await testSingleProxy(proxyUrl, targetUrl); + const { ok } = await (proxyFallbackTestHooks?.testSingleProxy ?? testSingleProxy)( + proxyUrl, + targetUrl + ); return { proxyUrl, ok }; }) ); @@ -311,7 +344,7 @@ export async function findWorkingProxy( if (working && working.status === "fulfilled") { const proxyUrl = working.value.proxyUrl; // Cache the working proxy - PROXY_FALLBACK_CACHE.set(targetHostname, { + PROXY_FALLBACK_CACHE.set(cacheKey, { proxyUrl, expiresAt: Date.now() + CACHE_TTL_MS, }); @@ -319,7 +352,7 @@ export async function findWorkingProxy( } // All failed — cache the negative result to avoid re-probing too often - PROXY_FALLBACK_CACHE.set(targetHostname, { + PROXY_FALLBACK_CACHE.set(cacheKey, { proxyUrl: "", expiresAt: Date.now() + CACHE_TTL_MS, }); diff --git a/open-sse/utils/publicCreds.ts b/open-sse/utils/publicCreds.ts index 4862f34c81..bc4d0993e8 100644 --- a/open-sse/utils/publicCreds.ts +++ b/open-sse/utils/publicCreds.ts @@ -1,7 +1,7 @@ /** * Public credentials decoder. * - * Some upstream providers (Gemini CLI, Antigravity, Windsurf/Devin CLI) ship + * Some upstream providers (Gemini, Antigravity, Windsurf/Devin CLI) ship * OAuth client_id / client_secret / Firebase Web API key values inside their * public binaries or web apps. These are credentials by name only — Google * explicitly documents that: @@ -130,7 +130,7 @@ export function decodePublicCredBytes(bytes: readonly number[]): string { * Or use the helper below `embeddedBytesFor()`. */ const EMBEDDED_DEFAULTS = { - // Gemini CLI / Code Assist — google oauth client (public, PKCE) + // Gemini / Code Assist — google oauth client (public, PKCE) gemini_id: [ 89, 85, 95, 91, 71, 90, 77, 68, 92, 30, 73, 64, 79, 3, 6, 91, 75, 2, 3, 0, 29, 28, 13, 0, 1, 5, 77, 0, 30, 17, 4, 4, 90, 8, 21, 30, 30, 92, 11, 4, 12, 88, 65, 90, 31, 90, 4, 93, 0, 6, 76, 11, @@ -176,9 +176,7 @@ const EMBEDDED_DEFAULTS = { 90, 64, 69, 83, 78, 18, 65, 90, 15, 89, 90, 21, ], // GitHub Copilot CLI — github oauth app id (public, device flow) - github_copilot_id: [ - 38, 27, 95, 71, 16, 90, 69, 67, 4, 29, 72, 22, 90, 91, 12, 0, 75, 19, 8, 87, - ], + github_copilot_id: [38, 27, 95, 71, 16, 90, 69, 67, 4, 29, 72, 22, 90, 91, 12, 0, 75, 19, 8, 87], // Grok Build CLI (xAI) — public oauth client id (import-token flow) grok_id: [ 13, 92, 15, 89, 66, 91, 76, 70, 72, 29, 71, 70, 3, 65, 93, 84, 72, 23, 28, 87, 92, 88, 15, 95, @@ -203,7 +201,7 @@ export function resolvePublicCred(key: EmbeddedDefaultKey, envName?: string): st /** * Resolve with multiple env-var aliases (first non-empty wins). Useful for - * providers that support both legacy and new env names (e.g. Gemini CLI). + * providers that support both legacy and new env names. */ export function resolvePublicCredMulti( key: EmbeddedDefaultKey, diff --git a/open-sse/utils/stream.ts b/open-sse/utils/stream.ts index 10b08bca65..046875666a 100644 --- a/open-sse/utils/stream.ts +++ b/open-sse/utils/stream.ts @@ -1934,9 +1934,7 @@ export function createSSEStream(options: StreamOptions = {}) { // Cloud Code API wraps in { response: { candidates: [...] } }, so unwrap. // Only applies to Gemini-family formats — skip for OpenAI, Claude, etc. const isGeminiFormat = - targetFormat === FORMATS.GEMINI || - targetFormat === FORMATS.GEMINI_CLI || - targetFormat === FORMATS.ANTIGRAVITY; + targetFormat === FORMATS.GEMINI || targetFormat === FORMATS.ANTIGRAVITY; const geminiChunk = isGeminiFormat ? unwrapGeminiChunk(parsed) : parsed; if (geminiChunk.candidates?.[0]?.content?.parts) { for (const part of geminiChunk.candidates[0].content.parts) { diff --git a/open-sse/utils/streamPayloadCollector.ts b/open-sse/utils/streamPayloadCollector.ts index 0dc409aeb1..630c964e89 100644 --- a/open-sse/utils/streamPayloadCollector.ts +++ b/open-sse/utils/streamPayloadCollector.ts @@ -596,7 +596,6 @@ export function buildStreamSummaryFromEvents( case FORMATS.CLAUDE: return buildClaudeSummary(events, fallbackModel); case FORMATS.GEMINI: - case FORMATS.GEMINI_CLI: case FORMATS.ANTIGRAVITY: return buildGeminiSummary(events, fallbackModel); default: diff --git a/open-sse/utils/usageTracking.ts b/open-sse/utils/usageTracking.ts index 7528960ace..8fc24f511a 100644 --- a/open-sse/utils/usageTracking.ts +++ b/open-sse/utils/usageTracking.ts @@ -237,7 +237,7 @@ export function filterUsageForFormat(usage, targetFormat) { let fields = formatFields[targetFormat]; // Use same fields for similar formats - if (targetFormat === FORMATS.GEMINI_CLI || targetFormat === FORMATS.ANTIGRAVITY) { + if (targetFormat === FORMATS.ANTIGRAVITY) { fields = formatFields[FORMATS.GEMINI]; } else if (targetFormat === FORMATS.OPENAI_RESPONSE) { fields = formatFields[FORMATS.OPENAI_RESPONSES]; diff --git a/package-lock.json b/package-lock.json index 3a038348fb..aeae5186c1 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "omniroute", - "version": "3.8.39", + "version": "3.8.40", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "omniroute", - "version": "3.8.39", + "version": "3.8.40", "hasInstallScript": true, "license": "MIT", "workspaces": [ @@ -30,6 +30,7 @@ "clsx": "^2.1.1", "commander": "^15.0.0", "csv-stringify": "^6.7.0", + "dompurify": "^3.4.11", "express": "^5.2.1", "fetch-socks": "^1.3.3", "fflate": "^0.8.3", @@ -98,10 +99,8 @@ "@tailwindcss/postcss": "^4.3.0", "@testing-library/jest-dom": "^6.9.1", "@testing-library/react": "^16.3.2", - "@types/bcryptjs": "^3.0.0", "@types/better-sqlite3": "^7.6.13", "@types/bun": "latest", - "@types/keytar": "^4.4.2", "@types/node": "^26.0.0", "@types/react": "^19.2.15", "@types/react-dom": "^19.2.3", @@ -9028,17 +9027,6 @@ "tslib": "^2.4.0" } }, - "node_modules/@types/bcryptjs": { - "version": "3.0.0", - "resolved": "https://registry.npmjs.org/@types/bcryptjs/-/bcryptjs-3.0.0.tgz", - "integrity": "sha512-WRZOuCuaz8UcZZE4R5HXTco2goQSI2XxjGY3hbM/xDvwmqFWd4ivooImsMx65OKM6CtNKbnZ5YL+YwAwK7c1dg==", - "deprecated": "This is a stub types definition. bcryptjs provides its own type definitions, so you do not need this installed.", - "dev": true, - "license": "MIT", - "dependencies": { - "bcryptjs": "*" - } - }, "node_modules/@types/better-sqlite3": { "version": "7.6.13", "resolved": "https://registry.npmjs.org/@types/better-sqlite3/-/better-sqlite3-7.6.13.tgz", @@ -9390,17 +9378,6 @@ "dev": true, "license": "MIT" }, - "node_modules/@types/keytar": { - "version": "4.4.2", - "resolved": "https://registry.npmjs.org/@types/keytar/-/keytar-4.4.2.tgz", - "integrity": "sha512-xtQcDj9ruGnMwvSu1E2BH4SFa5Dv2PvSPd0CKEBLN5hEj/v5YpXJY+B6hAfuKIbvEomD7vJTc/P1s1xPNh2kRw==", - "deprecated": "This is a stub types definition. keytar provides its own type definitions, so you do not need this installed.", - "dev": true, - "license": "MIT", - "dependencies": { - "keytar": "*" - } - }, "node_modules/@types/long": { "version": "4.0.2", "resolved": "https://registry.npmjs.org/@types/long/-/long-4.0.2.tgz", @@ -10350,25 +10327,24 @@ "node": ">=18.12.0" } }, - "node_modules/@yarnpkg/parsers/node_modules/argparse": { - "version": "1.0.10", - "resolved": "https://registry.npmjs.org/argparse/-/argparse-1.0.10.tgz", - "integrity": "sha512-o5Roy6tNG4SL/FOkCAN6RzjiakZS25RLYFrcMttJqbdd8BWrnA+fGz57iN5Pb06pvBGvl5gQ0B48dJlslXvoTg==", - "dev": true, - "license": "MIT", - "dependencies": { - "sprintf-js": "~1.0.2" - } - }, "node_modules/@yarnpkg/parsers/node_modules/js-yaml": { - "version": "3.14.2", - "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-3.14.2.tgz", - "integrity": "sha512-PMSmkqxr106Xa156c2M265Z+FTrPl+oxd/rgOQy2tijQeK5TxQ43psO1ZCwhVOSdnn+RzkzlRz/eY4BgJBYVpg==", + "version": "4.3.0", + "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.0.tgz", + "integrity": "sha512-1td788aAnnZ5qs7V2QIRl1owjtYpbKt749Y3xauqQgwIIGF/xXWz1wMTEBx5O3LK3lXLVuqXPdPxj2BoFHaW9Q==", "dev": true, + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/puzrin" + }, + { + "type": "github", + "url": "https://github.com/sponsors/nodeca" + } + ], "license": "MIT", "dependencies": { - "argparse": "^1.0.7", - "esprima": "^4.0.0" + "argparse": "^2.0.1" }, "bin": { "js-yaml": "bin/js-yaml.js" @@ -11042,7 +11018,6 @@ "version": "1.5.1", "resolved": "https://registry.npmjs.org/base64-js/-/base64-js-1.5.1.tgz", "integrity": "sha512-AKpaYlHn8t4SVbOHCy+b5+KKgvR4vrsD8vbvrbiQJps7fKDTkjkDry6ji0rUJjC0kzbNePLwzxq8iypo41qeWA==", - "devOptional": true, "funding": [ { "type": "github", @@ -11057,7 +11032,8 @@ "url": "https://feross.org/support" } ], - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/baseline-browser-mapping": { "version": "2.10.13", @@ -11146,8 +11122,8 @@ "version": "4.1.0", "resolved": "https://registry.npmjs.org/bl/-/bl-4.1.0.tgz", "integrity": "sha512-1W07cM9gS6DcLperZfFSj+bWLtaPGSOHWhPiGzXmvVJbRLdG82sH/Kn8EtW1VqWVA54AKf2h5k5BbnIbwF3h6w==", - "devOptional": true, "license": "MIT", + "optional": true, "dependencies": { "buffer": "^5.5.0", "inherits": "^2.0.4", @@ -11353,7 +11329,6 @@ "version": "5.7.1", "resolved": "https://registry.npmjs.org/buffer/-/buffer-5.7.1.tgz", "integrity": "sha512-EHcyIPBQ4BSGlvjB16k5KgAJ27CIsHY/2JBmCRReo48y9rQ3MaUzWX3KVlBa4U7MyX02HdVj0K7C3WaB3ju7FQ==", - "devOptional": true, "funding": [ { "type": "github", @@ -11369,6 +11344,7 @@ } ], "license": "MIT", + "optional": true, "dependencies": { "base64-js": "^1.3.1", "ieee754": "^1.1.13" @@ -11737,8 +11713,8 @@ "version": "1.1.4", "resolved": "https://registry.npmjs.org/chownr/-/chownr-1.1.4.tgz", "integrity": "sha512-jJ0bqzaylmJtVnNgzTeSOs8DPavpbYgEr/b0YL8/2GO3xJEhInFmhKMUnEJQjZumK7KXGFhUy89PrsJWlakBVg==", - "devOptional": true, - "license": "ISC" + "license": "ISC", + "optional": true }, "node_modules/class-variance-authority": { "version": "0.7.1", @@ -13253,8 +13229,8 @@ "version": "6.0.0", "resolved": "https://registry.npmjs.org/decompress-response/-/decompress-response-6.0.0.tgz", "integrity": "sha512-aW35yZM6Bb/4oJlZncMH2LCoZtJXTRxES17vE3hoRiowU2kWHaJKFkSBDnDR+cm9J+9QhXmREyIfv0pji9ejCQ==", - "devOptional": true, "license": "MIT", + "optional": true, "dependencies": { "mimic-response": "^3.1.0" }, @@ -14661,20 +14637,6 @@ "url": "https://opencollective.com/eslint" } }, - "node_modules/esprima": { - "version": "4.0.1", - "resolved": "https://registry.npmjs.org/esprima/-/esprima-4.0.1.tgz", - "integrity": "sha512-eGuFFw7Upda+g4p+QHvnW0RyTX/SVeJBDM/gCtMARO0cLuT2HcEKnTPvhjV6aGeqrCB/sbNop0Kszm0jsaWU4A==", - "dev": true, - "license": "BSD-2-Clause", - "bin": { - "esparse": "bin/esparse.js", - "esvalidate": "bin/esvalidate.js" - }, - "engines": { - "node": ">=4" - } - }, "node_modules/esquery": { "version": "1.7.0", "resolved": "https://registry.npmjs.org/esquery/-/esquery-1.7.0.tgz", @@ -14921,8 +14883,8 @@ "version": "2.0.3", "resolved": "https://registry.npmjs.org/expand-template/-/expand-template-2.0.3.tgz", "integrity": "sha512-XYfuKMvj4O35f/pOXLObndIRvyQ+/+6AhODh+OKWj9S9498pHHn/IMszH+gt0fBCRWMNfk1ZSp5x3AifmnI2vg==", - "devOptional": true, "license": "(MIT OR WTFPL)", + "optional": true, "engines": { "node": ">=6" } @@ -15504,8 +15466,8 @@ "version": "1.0.0", "resolved": "https://registry.npmjs.org/fs-constants/-/fs-constants-1.0.0.tgz", "integrity": "sha512-y6OAwoSIf7FyjMIv94u+b5rdheZEjzR63GTyZJm5qh4Bi+2YgwLCcI/fPFZkL5PSixOt6ZNKm+w+Hfp/Bciwow==", - "devOptional": true, - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/fs-extra": { "version": "11.3.5", @@ -15996,8 +15958,8 @@ "version": "0.0.0", "resolved": "https://registry.npmjs.org/github-from-package/-/github-from-package-0.0.0.tgz", "integrity": "sha512-SyHy3T1v2NUXn29OsWdxmK6RwHD+vkj3v8en8AOBZ1wBQ/hCAQ5bAQTD02kW4W9tUp/3Qh6J8r9EvntiyCmOOw==", - "devOptional": true, - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/github-slugger": { "version": "2.0.0", @@ -16757,7 +16719,6 @@ "version": "1.2.1", "resolved": "https://registry.npmjs.org/ieee754/-/ieee754-1.2.1.tgz", "integrity": "sha512-dcyqhDvX1C46lXZcVqCpK+FtMRQVdIMN6/Df5js2zouUsqG7I6sFxitIC+7KYK29KdXOLHdu9zL4sFnoVQnqaA==", - "devOptional": true, "funding": [ { "type": "github", @@ -16772,7 +16733,8 @@ "url": "https://feross.org/support" } ], - "license": "BSD-3-Clause" + "license": "BSD-3-Clause", + "optional": true }, "node_modules/ignore": { "version": "5.3.2", @@ -18436,9 +18398,9 @@ "version": "7.9.0", "resolved": "https://registry.npmjs.org/keytar/-/keytar-7.9.0.tgz", "integrity": "sha512-VPD8mtVtm5JNtA2AErl6Chp06JBfy7diFQ7TQQhdpWOl6MrCRB+eRbvAZUsbGQS9kiMq0coJsy0W0vHpDCkWsQ==", - "devOptional": true, "hasInstallScript": true, "license": "MIT", + "optional": true, "dependencies": { "node-addon-api": "^4.3.0", "prebuild-install": "^7.0.1" @@ -21023,8 +20985,8 @@ "version": "3.1.0", "resolved": "https://registry.npmjs.org/mimic-response/-/mimic-response-3.1.0.tgz", "integrity": "sha512-z0yWI+4FDrrweS8Zmt4Ej5HdJmky15+L2e6Wgn3+iK5fWzb6T3fhNFq2+MeTRb064c6Wr4N/wv0DzQTjNzHNGQ==", - "devOptional": true, "license": "MIT", + "optional": true, "engines": { "node": ">=10" }, @@ -21221,8 +21183,8 @@ "version": "0.5.3", "resolved": "https://registry.npmjs.org/mkdirp-classic/-/mkdirp-classic-0.5.3.tgz", "integrity": "sha512-gKLcREMhtuZRwRAfqP3RFW+TK4JqApVBtOIftVgjuABpAtpxhPGaDcfvbhNvD0B8iD1oUr/txX35NjcaY6Ns/A==", - "devOptional": true, - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/mlly": { "version": "1.8.2", @@ -21411,8 +21373,8 @@ "version": "2.0.0", "resolved": "https://registry.npmjs.org/napi-build-utils/-/napi-build-utils-2.0.0.tgz", "integrity": "sha512-GEbrYkbfF7MoNaoh2iGG84Mnf/WZfB0GdGEsM8wz7Expx/LlWf5U8t9nvJKXSp3qr5IsEbK04cBGhol/KwOsWA==", - "devOptional": true, - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/napi-postinstall": { "version": "0.3.4", @@ -21591,8 +21553,8 @@ "version": "3.89.0", "resolved": "https://registry.npmjs.org/node-abi/-/node-abi-3.89.0.tgz", "integrity": "sha512-6u9UwL0HlAl21+agMN3YAMXcKByMqwGx+pq+P76vii5f7hTPtKDp08/H9py6DY+cfDw7kQNTGEj/rly3IgbNQA==", - "devOptional": true, "license": "MIT", + "optional": true, "dependencies": { "semver": "^7.3.5" }, @@ -21604,8 +21566,8 @@ "version": "7.7.4", "resolved": "https://registry.npmjs.org/semver/-/semver-7.7.4.tgz", "integrity": "sha512-vFKC2IEtQnVhpT78h1Yp8wzwrf8CM+MzKMHGJZfBtzhZNycRFnXsHk6E5TxIkkMsgNS7mdX3AGB7x2QM2di4lA==", - "devOptional": true, "license": "ISC", + "optional": true, "bin": { "semver": "bin/semver.js" }, @@ -21617,8 +21579,8 @@ "version": "4.3.0", "resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-4.3.0.tgz", "integrity": "sha512-73sE9+3UaLYYFmDsFZnqCInzPyh3MqIwZO9cw58yIqAZhONrrabrYyYe3TuIqtIiOuTXVhsGau8hcrhhwSsDIQ==", - "devOptional": true, - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/node-emoji": { "version": "2.2.0", @@ -23129,8 +23091,8 @@ "resolved": "https://registry.npmjs.org/prebuild-install/-/prebuild-install-7.1.3.tgz", "integrity": "sha512-8Mf2cbV7x1cXPUILADGI3wuhfqWvtiLA1iclTDbFRZkgRQS0NqsPZphna9V+HyTEadheuPmjaJMsbzKQFOzLug==", "deprecated": "No longer maintained. Please contact the author of the relevant native addon; alternatives are available.", - "devOptional": true, "license": "MIT", + "optional": true, "dependencies": { "detect-libc": "^2.0.0", "expand-template": "^2.0.3", @@ -23884,8 +23846,8 @@ "version": "3.6.2", "resolved": "https://registry.npmjs.org/readable-stream/-/readable-stream-3.6.2.tgz", "integrity": "sha512-9u/sniCrY3D5WdsERHzHE4G2YCXqoG5FTHUiCC4SIbr6XcLZBY05ya9EKjYek9O5xOAwjGq+1JdGBAS7Q9ScoA==", - "devOptional": true, "license": "MIT", + "optional": true, "dependencies": { "inherits": "^2.0.3", "string_decoder": "^1.1.1", @@ -24642,7 +24604,6 @@ "version": "5.2.1", "resolved": "https://registry.npmjs.org/safe-buffer/-/safe-buffer-5.2.1.tgz", "integrity": "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ==", - "devOptional": true, "funding": [ { "type": "github", @@ -24657,7 +24618,8 @@ "url": "https://feross.org/support" } ], - "license": "MIT" + "license": "MIT", + "optional": true }, "node_modules/safe-push-apply": { "version": "1.0.0", @@ -25178,28 +25140,6 @@ "version": "1.0.1", "resolved": "https://registry.npmjs.org/simple-concat/-/simple-concat-1.0.1.tgz", "integrity": "sha512-cSFtAPtRhljv69IK0hTVZQ+OfE9nePi/rtJmw5UjHeVyVroEqJXP1sFztKUy1qU+xvz3u/sfYJLa947b7nAN2Q==", - "devOptional": true, - "funding": [ - { - "type": "github", - "url": "https://github.com/sponsors/feross" - }, - { - "type": "patreon", - "url": "https://www.patreon.com/feross" - }, - { - "type": "consulting", - "url": "https://feross.org/support" - } - ], - "license": "MIT" - }, - "node_modules/simple-get": { - "version": "4.0.1", - "resolved": "https://registry.npmjs.org/simple-get/-/simple-get-4.0.1.tgz", - "integrity": "sha512-brv7p5WgH0jmQJr1ZDDfKDOSeWWg+OVypG99A/5vYGPqJ6pxiaHLy8nxtFjBA7oMa01ebA9gfh1uMCFqOuXxvA==", - "devOptional": true, "funding": [ { "type": "github", @@ -25215,6 +25155,28 @@ } ], "license": "MIT", + "optional": true + }, + "node_modules/simple-get": { + "version": "4.0.1", + "resolved": "https://registry.npmjs.org/simple-get/-/simple-get-4.0.1.tgz", + "integrity": "sha512-brv7p5WgH0jmQJr1ZDDfKDOSeWWg+OVypG99A/5vYGPqJ6pxiaHLy8nxtFjBA7oMa01ebA9gfh1uMCFqOuXxvA==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/feross" + }, + { + "type": "patreon", + "url": "https://www.patreon.com/feross" + }, + { + "type": "consulting", + "url": "https://feross.org/support" + } + ], + "license": "MIT", + "optional": true, "dependencies": { "decompress-response": "^6.0.0", "once": "^1.3.1", @@ -25524,8 +25486,8 @@ "version": "1.0.3", "resolved": "https://registry.npmjs.org/sprintf-js/-/sprintf-js-1.0.3.tgz", "integrity": "sha512-D9cPgkvLlV3t3IzL0D0YLvGA9Ahk4PcvVwUbN0dSGr1aP0Nrt4AEnTUbuGvquEC0mA64Gqt1fzirlRs5ibXx8g==", - "devOptional": true, - "license": "BSD-3-Clause" + "license": "BSD-3-Clause", + "optional": true }, "node_modules/sql.js": { "version": "1.14.1", @@ -25729,8 +25691,8 @@ "version": "1.3.0", "resolved": "https://registry.npmjs.org/string_decoder/-/string_decoder-1.3.0.tgz", "integrity": "sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA==", - "devOptional": true, "license": "MIT", + "optional": true, "dependencies": { "safe-buffer": "~5.2.0" } @@ -26258,8 +26220,8 @@ "version": "2.1.4", "resolved": "https://registry.npmjs.org/tar-fs/-/tar-fs-2.1.4.tgz", "integrity": "sha512-mDAjwmZdh7LTT6pNleZ05Yt65HC3E+NiQzl672vQG38jIrehtJk/J3mNwIg+vShQPcLF/LV7CMnDW6vjj6sfYQ==", - "devOptional": true, "license": "MIT", + "optional": true, "dependencies": { "chownr": "^1.1.1", "mkdirp-classic": "^0.5.2", @@ -26271,8 +26233,8 @@ "version": "2.2.0", "resolved": "https://registry.npmjs.org/tar-stream/-/tar-stream-2.2.0.tgz", "integrity": "sha512-ujeqbceABgwMZxEJnk2HDY2DlnUZ+9oEcb1KzTVfYHio0UE6dG71n60d8D2I4qNvleWrrXpmjpt7vZeF1LnMZQ==", - "devOptional": true, "license": "MIT", + "optional": true, "dependencies": { "bl": "^4.0.3", "end-of-stream": "^1.4.1", @@ -26750,8 +26712,8 @@ "version": "0.6.0", "resolved": "https://registry.npmjs.org/tunnel-agent/-/tunnel-agent-0.6.0.tgz", "integrity": "sha512-McnNiV1l8RYeY8tBgEpuodCC1mLUdbSN+CYBL7kJsJNInOP8UjDDEwdk6Mw60vdLLrr5NHKZhMAOSrR2NZuQ+w==", - "devOptional": true, "license": "Apache-2.0", + "optional": true, "dependencies": { "safe-buffer": "^5.0.1" }, @@ -28583,7 +28545,7 @@ }, "open-sse": { "name": "@omniroute/open-sse", - "version": "3.8.39", + "version": "3.8.40", "dependencies": { "@toon-format/toon": "^2.3.0", "safe-regex": "^2.1.1" diff --git a/package.json b/package.json index 2c88af470b..3bcc660390 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "omniroute", - "version": "3.8.39", + "version": "3.8.40", "description": "Unified AI router with 160+ providers, RTK+Caveman compression, auto fallback, MCP/A2A, desktop, PWA, and OpenAI-compatible APIs.", "type": "module", "bin": { @@ -25,6 +25,7 @@ "bin/cli/runtime/", "scripts/postinstall.mjs", "scripts/build/postinstallSupport.mjs", + "scripts/build/runtime-env.mjs", "scripts/build/colocateOptionals.mjs", "scripts/build/sync-env.mjs", "scripts/dev/responses-ws-proxy.mjs", @@ -32,6 +33,7 @@ "scripts/dev/sync-env.mjs", "scripts/build/native-binary-compat.mjs", "scripts/build/build-next-isolated.mjs", + "scripts/build/runtime-env.mjs", "README.md", "LICENSE", "!**/__tests__/**", @@ -221,6 +223,7 @@ "clsx": "^2.1.1", "commander": "^15.0.0", "csv-stringify": "^6.7.0", + "dompurify": "^3.4.11", "express": "^5.2.1", "fetch-socks": "^1.3.3", "fflate": "^0.8.3", @@ -295,10 +298,8 @@ "@tailwindcss/postcss": "^4.3.0", "@testing-library/jest-dom": "^6.9.1", "@testing-library/react": "^16.3.2", - "@types/bcryptjs": "^3.0.0", "@types/better-sqlite3": "^7.6.13", "@types/bun": "latest", - "@types/keytar": "^4.4.2", "@types/node": "^26.0.0", "@types/react": "^19.2.15", "@types/react-dom": "^19.2.3", @@ -362,6 +363,9 @@ "protobufjs": "^7.6.3", "@babel/core": "^7.29.6", "hono": "^4.12.25", + "@yarnpkg/parsers": { + "js-yaml": "^4.2.0" + }, "jsdom": { "undici": "^7.28.0" }, diff --git a/public/providers/gemini-cli.svg b/public/providers/gemini-cli.svg deleted file mode 100644 index 20665d7654..0000000000 --- a/public/providers/gemini-cli.svg +++ /dev/null @@ -1 +0,0 @@ - \ No newline at end of file diff --git a/scripts/ad-hoc/diag-trae-auth.mjs b/scripts/ad-hoc/diag-trae-auth.mjs index fcacb0b3fd..fc9c5d37fa 100644 --- a/scripts/ad-hoc/diag-trae-auth.mjs +++ b/scripts/ad-hoc/diag-trae-auth.mjs @@ -26,7 +26,7 @@ const commonHeaders = { Referer: "https://solo.trae.ai/", "User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 " + - "(KHTML, like Gecko) Chrome/148.0.0.0 Safari/537.36", + "(KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36", }; // Each variant = a set of auth-bearing headers to try. diff --git a/scripts/ad-hoc/resolve_all_conflicts.js b/scripts/ad-hoc/resolve_all_conflicts.js index a79942dd11..bf597c8c71 100644 --- a/scripts/ad-hoc/resolve_all_conflicts.js +++ b/scripts/ad-hoc/resolve_all_conflicts.js @@ -76,30 +76,6 @@ async function main() { runCmd("git add open-sse/executors/index.ts"); } - // 5. Resolve tests/unit/t20-t22-provider-headers.test.ts (combine imports) - const testFile1 = path.join(projectRoot, "tests/unit/t20-t22-provider-headers.test.ts"); - if (fs.existsSync(testFile1)) { - let content = fs.readFileSync(testFile1, "utf-8"); - content = content.replace( - /<<<<<<< HEAD\r?\nconst \{ getCodexClientVersion \} = await import\("\.\.\/\.\.\/open-sse\/config\/codexClient\.ts"\);\r?\nconst \{ geminiCliUserAgent, GEMINI_CLI_VERSION \} =\r?\n=======\r?\nconst \{ geminiCliUserAgent, GEMINI_CLI_VERSION, GEMINI_CLI_GOOGLE_API_NODE_CLIENT_VERSION \} =\r?\n>>>>>>> release\/v3\.8\.4/g, - 'const { getCodexClientVersion } = await import("../../open-sse/config/codexClient.ts");\nconst { geminiCliUserAgent, GEMINI_CLI_VERSION, GEMINI_CLI_GOOGLE_API_NODE_CLIENT_VERSION } =' - ); - fs.writeFileSync(testFile1, content); - runCmd("git add tests/unit/t20-t22-provider-headers.test.ts"); - } - - // 6. Resolve tests/integration/chat-pipeline.test.ts (combine imports) - const testFile2 = path.join(projectRoot, "tests/integration/chat-pipeline.test.ts"); - if (fs.existsSync(testFile2)) { - let content = fs.readFileSync(testFile2, "utf-8"); - content = content.replace( - /<<<<<<< HEAD\r?\nconst \{ getCodexClientVersion \} = await import\("\.\.\/\.\.\/open-sse\/config\/codexClient\.ts"\);\r?\nconst \{ GEMINI_CLI_VERSION \} = await import\("\.\.\/\.\.\/open-sse\/services\/geminiCliHeaders\.ts"\);\r?\n=======\r?\nconst \{ GEMINI_CLI_VERSION, GEMINI_CLI_GOOGLE_API_NODE_CLIENT_VERSION \} =\r?\n\s+await import\("\.\.\/\.\.\/open-sse\/services\/geminiCliHeaders\.ts"\);\r?\n>>>>>>> release\/v3\.8\.4/g, - 'const { getCodexClientVersion } = await import("../../open-sse/config/codexClient.ts");\nconst { GEMINI_CLI_VERSION, GEMINI_CLI_GOOGLE_API_NODE_CLIENT_VERSION } =\n await import("../../open-sse/services/geminiCliHeaders.ts");' - ); - fs.writeFileSync(testFile2, content); - runCmd("git add tests/integration/chat-pipeline.test.ts"); - } - // 7. Resolve src/app/api/providers/[id]/models/route.ts (combine imports) const modelsRoute = path.join(projectRoot, "src/app/api/providers/[id]/models/route.ts"); if (fs.existsSync(modelsRoute)) { diff --git a/scripts/build/bootstrap-env.mjs b/scripts/build/bootstrap-env.mjs index 54d4f8d3c6..90ee46ddd6 100644 --- a/scripts/build/bootstrap-env.mjs +++ b/scripts/build/bootstrap-env.mjs @@ -30,10 +30,6 @@ const require = createRequire(import.meta.url); const OPTIONAL_OAUTH_SECRETS = [ { keys: ["ANTIGRAVITY_OAUTH_CLIENT_SECRET"], label: "Antigravity OAuth" }, { keys: ["QODER_OAUTH_CLIENT_SECRET"], label: "Qoder OAuth" }, - { - keys: ["GEMINI_CLI_OAUTH_CLIENT_SECRET", "GEMINI_OAUTH_CLIENT_SECRET"], - label: "Gemini OAuth", - }, ]; // ── Resolve DATA_DIR (mirrors dataPaths.ts logic) ─────────────────────────── diff --git a/scripts/build/pack-artifact-policy.ts b/scripts/build/pack-artifact-policy.ts index 12ed81d7a4..1cae4e6b03 100644 --- a/scripts/build/pack-artifact-policy.ts +++ b/scripts/build/pack-artifact-policy.ts @@ -79,10 +79,10 @@ export const PACK_ARTIFACT_ROOT_ALLOWED_EXACT_PATHS: string[] = [ "bin/nodeRuntimeSupport.mjs", "bin/omniroute.mjs", "bin/reset-password.mjs", - // Operator / incident-runbook shell tooling (rollback, snapshot, restore, - // cold-start bench) shipped in bin/ for self-hosters — referenced by - // docs/INCIDENT_RESPONSE.md, not imported by the runtime. Included via the - // package.json "files": ["bin/"] entry, so they must be allowed here. + // Operator incident-recovery / cold-start shell tooling (rollback, snapshot, + // restore, cold-start bench) shipped in bin/ for self-hosters — not imported by + // the runtime. Included via the package.json "files": ["bin/"] entry, so they + // must be allowed here. Each script is self-documenting via --help. "bin/_ops-common.sh", "bin/cold-start-bench.sh", "bin/restore-data.sh", @@ -106,6 +106,8 @@ export const PACK_ARTIFACT_ROOT_ALLOWED_EXACT_PATHS: string[] = [ "scripts/build/postinstall.mjs", "scripts/build/postinstallSupport.mjs", "scripts/build/colocateOptionals.mjs", + // #5227: imported at runtime by bin/cli/commands/serve.mjs (heap auto-calibration). + "scripts/build/runtime-env.mjs", "scripts/build/sync-env.mjs", "scripts/dev/responses-ws-proxy.mjs", "scripts/dev/sync-env.mjs", @@ -148,6 +150,7 @@ export const PACK_ARTIFACT_REQUIRED_PATHS: string[] = [ "scripts/build/postinstall.mjs", "scripts/build/postinstallSupport.mjs", "scripts/build/colocateOptionals.mjs", + "scripts/build/runtime-env.mjs", "src/shared/utils/nodeRuntimeSupport.ts", ]; diff --git a/scripts/check/check-docs-symbols.mjs b/scripts/check/check-docs-symbols.mjs index f05299fe4e..bde02caaf9 100644 --- a/scripts/check/check-docs-symbols.mjs +++ b/scripts/check/check-docs-symbols.mjs @@ -58,13 +58,9 @@ export const KNOWN_STALE_DOC_REFS = new Set([ // existing guardrailRegistry); the fictional enable/disable/logs rows and the entire // shadow table were removed from the doc (shadow A-B comparison is combo-config + // /api/combos/metrics). No allowlist entries needed for these anymore. - // docs/research/DISCOVERY_TOOL_DESIGN.md — design doc de feature NÃO implementada - // (Phase 2). Refs INTENCIONAIS: o doc agora traz um banner "⚠️ Not yet implemented - // — Phase 2" acima da tabela de endpoints. Mantidos aqui até a feature existir. — #3498 - "/api/discovery/results", - "/api/discovery/results/:id", - "/api/discovery/scan", - "/api/discovery/verify/:id", + // (DISCOVERY_TOOL_DESIGN.md saiu de docs/research/ para o repo isolado _tasks/research/ + // — gitignored, fora do escopo deste gate. As 4 entradas /api/discovery/* viraram + // obsoletas e foram removidas para satisfazer o stale-enforcement da allowlist.) // docs/reference/ENVIRONMENT.md — endpoint UPSTREAM do provedor Blackbox Web, // citado na descrição de env var (não é rota do OmniRoute): "/api/chat", @@ -85,9 +81,7 @@ function walk(dir, filter, acc = []) { export function collectRouteFiles() { return new Set( - walk(API, (n) => /^route\.tsx?$/.test(n)).map((p) => - path.relative(ROOT, p).replace(/\\/g, "/") - ) + walk(API, (n) => /^route\.tsx?$/.test(n)).map((p) => path.relative(ROOT, p).replace(/\\/g, "/")) ); } diff --git a/scripts/check/check-fabricated-docs.mjs b/scripts/check/check-fabricated-docs.mjs index 3dea3bf5a8..785ca9395b 100644 --- a/scripts/check/check-fabricated-docs.mjs +++ b/scripts/check/check-fabricated-docs.mjs @@ -107,8 +107,6 @@ const ENV_VAR_ALLOWLIST = new Set([ "NINEROUTER_API_KEY", // injected into the 9router subprocess at spawn (EMBEDDED-SERVICES.md) "CLAUDE_CODE_MAX_OUTPUT_TOKENS", // Claude Code CLI's own env var (CODEX-CLI-CONFIGURATION.md) "CODEX_HOME", // Codex CLI's own config-home env var (CODEX-CLI-CONFIGURATION.md) - "GEMINI_API_KEY", // Gemini CLI's own API-key env var, set by `omniroute setup-gemini` (REMOTE-MODE.md) - "GOOGLE_GEMINI_BASE_URL", // Gemini CLI's own base-URL env var, set by `omniroute setup-gemini` (REMOTE-MODE.md) "OPENAI_API_BASE", // legacy OpenAI base-URL env var some downstream tools (e.g. Aider) read (CLI-INTEGRATIONS.md) "PROMPTFOO_PROVIDER_KEY", // promptfoo's own provider-key env var, used by the red-team suite (GUARDRAILS.md) "REDIS_PORT", // docker-compose host-port override (DOCKER_GUIDE.md) @@ -361,9 +359,6 @@ const SKIP_DOC_FILES = new Set([ "docs/reference/PROVIDER_REFERENCE.md", // auto-generated from providers.ts "docs/openapi.yaml", "docs/i18n", // translations — separate workflow - // Point-in-time documentation audit (v3.8.24): intentionally references drift, - // counts, and not-yet-existing files as part of documenting them — not living docs. - "docs/ops/DOCUMENTATION_AUDIT_REPORT.md", // Design / research / plan docs: by definition describe not-yet-built files and // proposed (not-yet-shipped) endpoints (each carries a `Status: Design`/`Active // research`/`Plano` header). Same rationale as the audit report above — these are diff --git a/scripts/check/check-known-symbols.ts b/scripts/check/check-known-symbols.ts index b8f49e02c5..d88320ed85 100644 --- a/scripts/check/check-known-symbols.ts +++ b/scripts/check/check-known-symbols.ts @@ -181,8 +181,6 @@ export const KNOWN_TRANSLATOR_PAIRS: readonly string[] = [ "claude:gemini", "claude:openai", "cursor:openai", - "gemini-cli:claude", - "gemini-cli:openai", "gemini:claude", "gemini:openai", "kiro:openai", @@ -191,7 +189,6 @@ export const KNOWN_TRANSLATOR_PAIRS: readonly string[] = [ "openai:claude", "openai:cursor", "openai:gemini", - "openai:gemini-cli", "openai:kiro", "openai:openai-responses", ]; @@ -200,10 +197,7 @@ export const KNOWN_TRANSLATOR_PAIRS: readonly string[] = [ * Pares frozen que sumiram do registry vivo (regressão). frozen = snapshot; * live = pares observados em runtime. Retorna os que estão no frozen mas não no live. */ -export function findMissingTranslatorPairs( - frozen: readonly string[], - live: Set -): string[] { +export function findMissingTranslatorPairs(frozen: readonly string[], live: Set): string[] { return frozen.filter((pair) => !live.has(pair)); } @@ -375,10 +369,7 @@ export type A2ASkillDiff = { * - inHandlersNotCard: skill is routable but agents can't discover it * - inCardNotHandlers: skill is advertised but calling it fails silently */ -export function diffA2ASkills( - handlers: Set, - agentCard: Set -): A2ASkillDiff { +export function diffA2ASkills(handlers: Set, agentCard: Set): A2ASkillDiff { const inHandlersNotCard = [...handlers].filter((s) => !agentCard.has(s)).sort(); const inCardNotHandlers = [...agentCard].filter((s) => !handlers.has(s)).sort(); return { inHandlersNotCard, inCardNotHandlers }; @@ -456,13 +447,12 @@ async function main(): Promise { const executorsMod = await import("@omniroute/open-sse/executors/index.ts"); const getExecutor = executorsMod.getExecutor as (alias: string) => ExecutorLike; const BaseExecutor = executorsMod.BaseExecutor as new (...args: never[]) => unknown; - const indexSource = readFileSync( - resolvePath(REPO_ROOT, "open-sse/executors/index.ts"), - "utf8" - ); + const indexSource = readFileSync(resolvePath(REPO_ROOT, "open-sse/executors/index.ts"), "utf8"); const aliases = extractExecutorAliases(indexSource); if (aliases.length === 0) { - failures.push("[executor] parse do mapa `executors` não encontrou nenhum alias (regex quebrada?)"); + failures.push( + "[executor] parse do mapa `executors` não encontrou nenhum alias (regex quebrada?)" + ); } const isExecutorInstance = (value: unknown) => value instanceof BaseExecutor; const badExecutors = findNonConformingExecutors(aliases, getExecutor, isExecutorInstance); @@ -493,7 +483,11 @@ async function main(): Promise { // EMPTY implicit-defaults map). An entry whose key IS already in `handled` suppresses // nothing → it is stale and the gate must fail asking for its removal. const liveImplicitNeeded = diffComboStrategies(canonical, handled, {}).canonicalNotHandled; - assertNoStale(Object.keys(IMPLICIT_DEFAULT_STRATEGIES), liveImplicitNeeded, "known-symbols:combo"); + assertNoStale( + Object.keys(IMPLICIT_DEFAULT_STRATEGIES), + liveImplicitNeeded, + "known-symbols:combo" + ); const { canonicalNotHandled, handledNotCanonical } = diffComboStrategies( canonical, @@ -554,9 +548,8 @@ async function main(): Promise { const { MCP_TOOLS } = await import("@omniroute/open-sse/mcp-server/schemas/tools.ts"); const { memoryTools } = await import("@omniroute/open-sse/mcp-server/tools/memoryTools.ts"); const { skillTools } = await import("@omniroute/open-sse/mcp-server/tools/skillTools.ts"); - const { gamificationTools } = await import( - "@omniroute/open-sse/mcp-server/tools/gamificationTools.ts" - ); + const { gamificationTools } = + await import("@omniroute/open-sse/mcp-server/tools/gamificationTools.ts"); const { pluginTools } = await import("@omniroute/open-sse/mcp-server/tools/pluginTools.ts"); const { notionTools } = await import("@omniroute/open-sse/mcp-server/tools/notionTools.ts"); const { obsidianTools } = await import("@omniroute/open-sse/mcp-server/tools/obsidianTools.ts"); @@ -669,7 +662,9 @@ async function main(): Promise { // ── Resultado ───────────────────────────────────────────────────────────── if (failures.length) { - console.error(`[known-symbols] ${failures.length} sub-checagem(ns) falharam:\n\n${failures.join("\n\n")}`); + console.error( + `[known-symbols] ${failures.length} sub-checagem(ns) falharam:\n\n${failures.join("\n\n")}` + ); process.exit(1); } // assertNoStale (combo) seta process.exitCode=1 sem lançar — não imprima o OK @@ -695,7 +690,9 @@ async function main(): Promise { if (import.meta.url === pathToFileURL(process.argv[1] || "").href) { main().catch((err) => { - console.error(`[known-symbols] erro fatal: ${err instanceof Error ? err.message : String(err)}`); + console.error( + `[known-symbols] erro fatal: ${err instanceof Error ? err.message : String(err)}` + ); process.exit(1); }); } diff --git a/scripts/check/check-test-discovery.mjs b/scripts/check/check-test-discovery.mjs index c29da68017..5a1bfe6b3f 100644 --- a/scripts/check/check-test-discovery.mjs +++ b/scripts/check/check-test-discovery.mjs @@ -67,6 +67,7 @@ export const COLLECTORS = [ // vitest.mcp.config.ts — test:vitest { glob: "open-sse/mcp-server/__tests__/**/*.test.ts", sources: ["vitest.mcp.config.ts"] }, { glob: "open-sse/services/autoCombo/__tests__/**/*.test.ts", sources: ["vitest.mcp.config.ts"] }, + { glob: "open-sse/services/combo/__tests__/**/*.test.ts", sources: ["vitest.mcp.config.ts"] }, // Single-file include: the rest of open-sse/services/__tests__/ are frozen orphans // (empty/dormant stubs); only this one is wired to run under test:vitest. { diff --git a/scripts/ci/should-promote-latest.sh b/scripts/ci/should-promote-latest.sh new file mode 100755 index 0000000000..12704b7962 --- /dev/null +++ b/scripts/ci/should-promote-latest.sh @@ -0,0 +1,44 @@ +#!/usr/bin/env bash +# Decide whether the just-built release VERSION should also move the Docker +# `:latest` / `:latest-web` tags. +# +# Promote ONLY when VERSION is the highest STABLE semver among the union of +# {existing git tags} ∪ {VERSION}. Folding VERSION into the candidate set makes +# the decision independent of git-tag sync timing: on a `release: released` +# event the freshly-created tag is often not yet visible to `git fetch --tags` +# when this job runs, so a candidate set built purely from `git tag -l` would +# resolve HIGHEST to the *previous* version and skip the `:latest` promotion — +# leaving `latest` one release behind (#5301). +# +# Usage: +# git tag -l 'v[0-9]*' | sed 's/^v//' | scripts/ci/should-promote-latest.sh "$VERSION" +# +# - $1 : the version being published (no leading `v`, e.g. 3.8.40). +# - stdin : newline-separated candidate tags (a leading `v` is stripped; +# pre-release tags — anything containing `-`, e.g. 3.9.0-rc.1 — +# are ignored). May be empty (first release). +# Prints "true" if VERSION should move :latest, "false" otherwise. +set -euo pipefail + +VERSION="${1:?version required}" + +# A pre-release VERSION must never grab :latest (callers already short-circuit +# this, but stay safe as a standalone unit). +case "$VERSION" in + *-*) echo "false"; exit 0 ;; +esac + +# Build the stable candidate set: incoming tags (v-stripped, pre-releases +# dropped) plus VERSION itself, then pick the numerically highest. +HIGHEST="$( + { + sed 's/^v//' | grep -vE -- '-' || true + printf '%s\n' "$VERSION" + } | sort -V | tail -1 +)" + +if [ "$VERSION" = "$HIGHEST" ]; then + echo "true" +else + echo "false" +fi diff --git a/scripts/dev/responses-ws-proxy.mjs b/scripts/dev/responses-ws-proxy.mjs index abbb435e8f..5fa562d752 100644 --- a/scripts/dev/responses-ws-proxy.mjs +++ b/scripts/dev/responses-ws-proxy.mjs @@ -616,7 +616,7 @@ class ResponsesWsSession { }; const upstream = await this.wsFactory(prepared.json.upstreamUrl, { - browser: prepared.json.browser || "chrome_142", + browser: prepared.json.browser || "chrome_149", os: prepared.json.os || "windows", headers: prepared.json.headers || {}, }); diff --git a/scripts/dev/system-info.mjs b/scripts/dev/system-info.mjs index 6d112a11e9..d0b51e3e23 100644 --- a/scripts/dev/system-info.mjs +++ b/scripts/dev/system-info.mjs @@ -90,7 +90,6 @@ lines.push(section("Agent CLI Tools")); const cliTools = [ { name: "qoder-cli", cmd: "qoder", args: "--version" }, - { name: "gemini-cli", cmd: "gemini", args: "--version" }, { name: "claude-code", cmd: "claude", args: "--version" }, { name: "openai-codex", cmd: "codex", args: "--version" }, { name: "antigravity", cmd: "antigravity", args: "--version" }, diff --git a/scripts/docs/sync-wiki.mjs b/scripts/docs/sync-wiki.mjs index ff8f6529dd..ec70f685cf 100644 --- a/scripts/docs/sync-wiki.mjs +++ b/scripts/docs/sync-wiki.mjs @@ -56,13 +56,37 @@ export const NEW_PAGE_EXCLUDE = new Set([ // Acronyms kept upper-case when minting a NEW page name (existing pages keep their // curated name via fuzzy match, so this only affects brand-new pages). const ACRONYMS = new Set([ - "api", "mcp", "a2a", "acp", "cli", "sse", "i18n", "pii", "oauth", "vm", "ai", - "llm", "sdk", "ide", "ui", "ux", "tls", "mitm", "ws", "cors", "jwt", "db", "vps", + "api", + "mcp", + "a2a", + "acp", + "cli", + "sse", + "i18n", + "pii", + "oauth", + "vm", + "ai", + "llm", + "sdk", + "ide", + "ui", + "ux", + "tls", + "mitm", + "ws", + "cors", + "jwt", + "db", + "vps", ]); /** Normalized matching key: lowercase, drop extension + every non-alphanumeric char. */ export function normKey(s) { - return s.toLowerCase().replace(/\.md$/, "").replace(/[^a-z0-9]/g, ""); + return s + .toLowerCase() + .replace(/\.md$/, "") + .replace(/[^a-z0-9]/g, ""); } /** Deterministic wiki page name for a brand-new page (acronym-aware Title-Case-dashed). */ @@ -71,7 +95,11 @@ export function toWikiName(basename) { .replace(/\.md$/, "") .split(/[_\-\s]+/) .filter(Boolean) - .map((t) => (ACRONYMS.has(t.toLowerCase()) ? t.toUpperCase() : t[0].toUpperCase() + t.slice(1).toLowerCase())) + .map((t) => + ACRONYMS.has(t.toLowerCase()) + ? t.toUpperCase() + : t[0].toUpperCase() + t.slice(1).toLowerCase() + ) .join("-"); } @@ -124,13 +152,19 @@ export function syncHomeCounts(home, counts) { let out = home; if (counts.providers) { out = out - .replace(/Connect every AI tool to \d+ providers/g, `Connect every AI tool to ${counts.providers} providers`) + .replace( + /Connect every AI tool to \d+ providers/g, + `Connect every AI tool to ${counts.providers} providers` + ) .replace(/\*\*\d+ AI Providers\*\*/g, `**${counts.providers} AI Providers**`) .replace(/All \d+ supported providers/g, `All ${counts.providers} supported providers`) .replace(/\b\d+ providers\b/g, `${counts.providers} providers`); } if (counts.strategies) { - out = out.replace(/\*\*\d+ Routing Strategies\*\*/g, `**${counts.strategies} Routing Strategies**`); + out = out.replace( + /\*\*\d+ Routing Strategies\*\*/g, + `**${counts.strategies} Routing Strategies**` + ); } if (counts.mcpTools) { out = out.replace(/(\|\s*\*\*MCP Server\*\*\s*\|\s*)\d+( tools)/g, `$1${counts.mcpTools}$2`); @@ -251,7 +285,7 @@ function main() { // ARCHITECTURE.md still says "177 providers / 37 MCP tools" while the wiki cover was // hand-patched to 226/87). Overwriting from a staler source would REGRESS the wiki, so // by default we only ADD missing pages and sync Home counts. Pass --update-existing - // once the docs sources are regenerated (see docs/ops/DOCUMENTATION_AUDIT_REPORT.md). + // once the docs sources are regenerated. const updates = updateExisting ? plan.update : []; const total = updates.length + plan.add.length + (plan.countsChanged ? 1 : 0); console.log(`[wiki-sync] counts: ${JSON.stringify(counts)}`); @@ -263,7 +297,10 @@ function main() { if (plan.add.length) console.log(` add → ${plan.add.map((a) => a.page).join(", ")}`); if (plan.update.length) console.log( - ` ${updateExisting ? "update" : "would-update (skipped)"} → ${plan.update.map((u) => u.page).slice(0, 60).join(", ")}${plan.update.length > 60 ? " …" : ""}` + ` ${updateExisting ? "update" : "would-update (skipped)"} → ${plan.update + .map((u) => u.page) + .slice(0, 60) + .join(", ")}${plan.update.length > 60 ? " …" : ""}` ); if (check) { if (total > 0) { @@ -277,7 +314,10 @@ function main() { // ---- write ---- for (const { page, srcFile } of [...updates, ...plan.add]) { - fs.writeFileSync(path.join(wikiDir, `${page}.md`), toWikiContent(fs.readFileSync(srcFile, "utf8"))); + fs.writeFileSync( + path.join(wikiDir, `${page}.md`), + toWikiContent(fs.readFileSync(srcFile, "utf8")) + ); } if (plan.countsChanged && homeAfter != null) fs.writeFileSync(homePath, homeAfter); console.log( diff --git a/scripts/i18n/i18n_autotranslate.py b/scripts/i18n/i18n_autotranslate.py index 1ff8c1c617..417e66b592 100755 --- a/scripts/i18n/i18n_autotranslate.py +++ b/scripts/i18n/i18n_autotranslate.py @@ -134,7 +134,7 @@ def main(): parser = argparse.ArgumentParser(description="OmniRoute Auto-Translator for i18n Markdown") parser.add_argument("--api-url", default="http://localhost:20128/v1", help="Base URL of OmniRoute or target provider") parser.add_argument("--api-key", default="sk-test", help="API Key for the provider") - parser.add_argument("--model", default="gc/gemini-3-flash", help="Model name to use") + parser.add_argument("--model", default="gemini/gemini-3-flash", help="Model name to use") parser.add_argument("--lang", default=None, help="Process only a specific language code (e.g. pt-BR)") args = parser.parse_args() diff --git a/scripts/quality/mutation-radiography.mjs b/scripts/quality/mutation-radiography.mjs index 792f88c19e..0770ea84ad 100644 --- a/scripts/quality/mutation-radiography.mjs +++ b/scripts/quality/mutation-radiography.mjs @@ -4,8 +4,7 @@ * * Classifies every COVERING test file by its mutation-kill contribution, using the * `killedBy` attribution that the Stryker tap-runner emits per mutant - * (`coverageAnalysis: perTest`, validated by the Task 12 spike — see - * docs/ops/MUTATION_GATE_SPIKE_VERDICT.md): + * (`coverageAnalysis: perTest`, validated by the Task 12 spike): * * 🔴 empty — the test file never appears in any `killedBy` (kills no mutant * of the mutated modules). Prime R1-prune candidate (Task 2). @@ -279,7 +278,9 @@ function main(argv) { const reports = paths.map(loadMutationReport); const universe = useConfUniverse ? tapTestFilesUniverse() : null; if (wantCandidates) { - process.stdout.write(renderCandidates(redundancyCandidates(reports, universe || undefined)) + "\n"); + process.stdout.write( + renderCandidates(redundancyCandidates(reports, universe || undefined)) + "\n" + ); return; } const classification = aggregateRadiography(reports, universe || undefined); diff --git a/skills/cli-serve/SKILL.md b/skills/cli-serve/SKILL.md index 1235beec41..2818910229 100644 --- a/skills/cli-serve/SKILL.md +++ b/skills/cli-serve/SKILL.md @@ -2,6 +2,7 @@ name: cli-serve description: Start, stop, and restart the OmniRoute server from the CLI. Manage daemon mode, port configuration, auto-recovery, system tray integration, and the dashboard open shortcut. --- + ## Overview @@ -84,7 +85,7 @@ npm install -g omniroute # npm registry # or: use the binary bundled with the desktop app ``` -Requires Node.js ≥20.20.2, ≥22.22.2, or ≥24. +Requires Node.js ≥22.22.2 or ≥24. Verify: diff --git a/src/app/(dashboard)/dashboard/acp-agents/page.tsx b/src/app/(dashboard)/dashboard/acp-agents/page.tsx index 79d4d90f34..47773ac768 100644 --- a/src/app/(dashboard)/dashboard/acp-agents/page.tsx +++ b/src/app/(dashboard)/dashboard/acp-agents/page.tsx @@ -30,7 +30,6 @@ const AGENT_ICON_MAP: Record = { claude: "anthropic", "claude-code": "anthropic", codex: "openai", - "gemini-cli": "google", gemini: "google", opencode: "opencode", openclaw: "openclaw", diff --git a/src/app/(dashboard)/dashboard/analytics/CompressionAnalyticsTab.tsx b/src/app/(dashboard)/dashboard/analytics/CompressionAnalyticsTab.tsx index fcc47b3862..1ad0ea9926 100644 --- a/src/app/(dashboard)/dashboard/analytics/CompressionAnalyticsTab.tsx +++ b/src/app/(dashboard)/dashboard/analytics/CompressionAnalyticsTab.tsx @@ -16,9 +16,14 @@ interface CompressionAnalyticsSummary { totalTokensSaved: number; avgSavingsPct: number; avgDurationMs: number; - byMode: Record; + byMode: Record< + string, + { count: number; tokensSaved: number; avgSavingsPct: number; skipped?: number } + >; byProvider: Record; last24h: Array<{ hour: string; count: number; tokensSaved: number }>; + totalSkipped?: number; + bySkipReason?: Record; validationFallbacks: number; realUsage: { requestsWithReceipts: number; @@ -60,11 +65,13 @@ function ModeBar({ count, total, tokensSaved, + skipped = 0, }: { mode: string; count: number; total: number; tokensSaved: number; + skipped?: number; }) { const pct = total > 0 ? Math.round((count / total) * 100) : 0; return ( @@ -73,6 +80,11 @@ function ModeBar({ {mode} {count} requests · {tokensSaved.toLocaleString()} tokens saved + {skipped > 0 && ( + // #4268: attempted-but-no-op runs (e.g. Stacked saved nothing) are + // recorded now, so this mode is visible even when count is 0. + · {skipped.toLocaleString()} skipped (no-op) + )}
@@ -288,6 +300,7 @@ export default function CompressionAnalyticsTab() { count={data.count} total={stats.totalRequests} tokensSaved={data.tokensSaved} + skipped={data.skipped ?? 0} /> ))}
diff --git a/src/app/(dashboard)/dashboard/cli-code/components/CodexToolCard.tsx b/src/app/(dashboard)/dashboard/cli-code/components/CodexToolCard.tsx index 92af14aafd..8abb015c9a 100644 --- a/src/app/(dashboard)/dashboard/cli-code/components/CodexToolCard.tsx +++ b/src/app/(dashboard)/dashboard/cli-code/components/CodexToolCard.tsx @@ -33,9 +33,7 @@ export default function CodexToolCard({ "gpt-5.5", "gpt-5.3-codex", "gpt-5.4", - "gpt-5.2-codex", "gpt-5.1-codex-max", - "gpt-5.2", "gpt-5.1-codex-mini", ]; const [modelMappings, setModelMappings] = useState>({}); diff --git a/src/app/(dashboard)/dashboard/combos/page.tsx b/src/app/(dashboard)/dashboard/combos/page.tsx index 350ec88e79..ca3a4e105f 100644 --- a/src/app/(dashboard)/dashboard/combos/page.tsx +++ b/src/app/(dashboard)/dashboard/combos/page.tsx @@ -406,7 +406,7 @@ const COMBO_TEMPLATE_FALLBACK = { balancedDesc: "Least-used routing to spread demand over time.", freeStackTitle: "Free Stack ($0)", freeStackDesc: - "Round-robin across all free providers: Kiro, Qoder, Qwen, Gemini CLI. Zero cost, never stops.", + "Round-robin across free providers: Kiro, Qoder, Qwen, Antigravity CLI. Zero cost, never stops.", paidPremiumTitle: "Paid Premium", paidPremiumDesc: "Round-robin across paid subscriptions: Cursor, Antigravity. Top-tier models, distributed load.", @@ -2641,7 +2641,7 @@ function ComboFormModal({ isOpen, combo, onClose, onSave, activeProviders, combo }; const FREE_STACK_PRESET_MODELS = [ - { model: "gemini-cli/gemini-3-flash-preview", weight: 0 }, + { model: "agy/gemini-3.5-flash-medium", weight: 0 }, { model: "kr/claude-sonnet-4.5", weight: 0 }, { model: "if/kimi-k2-thinking", weight: 0 }, { model: "if/qwen3-coder-plus", weight: 0 }, diff --git a/src/app/(dashboard)/dashboard/compression/studio/CompressionAnnotation.tsx b/src/app/(dashboard)/dashboard/compression/studio/CompressionAnnotation.tsx new file mode 100644 index 0000000000..fdf6d1f75c --- /dev/null +++ b/src/app/(dashboard)/dashboard/compression/studio/CompressionAnnotation.tsx @@ -0,0 +1,36 @@ +import type { CompressionStats } from "@omniroute/open-sse/services/compression/types"; + +export interface CompressionAnnotationProps { + stats: CompressionStats; +} + +/** + * Renders a token-savings badge (`847→312`) plus per-rule count pills when + * rulesApplied is non-empty. Returns null when there are no rules to display. + */ +export function CompressionAnnotation({ stats }: CompressionAnnotationProps) { + const rules = stats.rulesApplied; + if (!rules || rules.length === 0) return null; + + const counts = new Map(); + for (const rule of rules) { + counts.set(rule, (counts.get(rule) ?? 0) + 1); + } + const sorted = [...counts.entries()].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])); + + return ( +
+ + {stats.originalTokens}→{stats.compressedTokens} + + {sorted.map(([name, n]) => ( + + {name}×{n} + + ))} +
+ ); +} diff --git a/src/app/(dashboard)/dashboard/compression/studio/CompressionCockpit.tsx b/src/app/(dashboard)/dashboard/compression/studio/CompressionCockpit.tsx index fbfaec3376..bdd2d633e6 100644 --- a/src/app/(dashboard)/dashboard/compression/studio/CompressionCockpit.tsx +++ b/src/app/(dashboard)/dashboard/compression/studio/CompressionCockpit.tsx @@ -8,6 +8,7 @@ import { IoNode } from "./nodes/IoNode"; import { compressionRunToFlow, type CompressionRunModel } from "./compressionFlowModel"; import { useCompressionReplay, type ReplaySpeed } from "./useCompressionReplay"; import { WaterfallInspector } from "./WaterfallInspector"; +import { CompressionAnnotation } from "./CompressionAnnotation"; // ── View modes ──────────────────────────────────────────────────────────── @@ -179,6 +180,17 @@ export function CompressionCockpit({ run: runProp }: CompressionCockpitProps) { )} {run.requestId} + s.rulesApplied ?? []), + }} + />
{/* View toggle: ReactFlow canvas (A2) ↔ waterfall list (A1) */}
diff --git a/src/app/(dashboard)/dashboard/compression/studio/PlayView.tsx b/src/app/(dashboard)/dashboard/compression/studio/PlayView.tsx index 47dfd23fc5..0537d861b3 100644 --- a/src/app/(dashboard)/dashboard/compression/studio/PlayView.tsx +++ b/src/app/(dashboard)/dashboard/compression/studio/PlayView.tsx @@ -5,6 +5,9 @@ import { WaterfallInspector } from "./WaterfallInspector"; import { DiffPane } from "./DiffPane"; import { EncoderComparisonTable } from "./EncoderComparisonTable"; import { PlaygroundInput, LANE_ENGINES } from "./PlaygroundInput"; +import { RiskGateBadge } from "./RiskGateBadge"; +import { QuantumLockBadge } from "./QuantumLockBadge"; +import { SaliencyHeatmap } from "./SaliencyHeatmap"; export interface PlayViewProps { text: string; onText: (t: string) => void; @@ -45,10 +48,19 @@ export function PlayView({ text, onText, laneEngines = LANE_ENGINES }: PlayViewP const [fuzzyDedup, setFuzzyDedup] = useState(false); const [selectedLane, setSelectedLane] = useState(null); const [fidelityGate, setFidelityGate] = useState(false); + const [riskGate, setRiskGate] = useState(false); + const [quantumLock, setQuantumLock] = useState(false); + const [heatmapMode, setHeatmapMode] = useState<"ultra" | "universal" | false>(false); const { batch, loading, run } = usePreviewCompression(); const messages = [{ role: "user", content: text }]; const toggle = (e: string) => setActive((a) => (a.includes(e) ? a.filter((x) => x !== e) : [...a, e])); + const toggleHeatmap = () => + setHeatmapMode((m) => { + if (!m) return "ultra"; + if (m === "ultra") return "universal"; + return false; + }); const onRun = () => run({ messages, @@ -56,6 +68,9 @@ export function PlayView({ text, onText, laneEngines = LANE_ENGINES }: PlayViewP activeEngines: orderByStack(active, laneEngines), fidelityGate, fuzzyDedup, + riskGate, + quantumLock, + ...(heatmapMode ? { heatmap: heatmapMode } : {}), }); const activeDiff = resolveActiveDiff(batch, selectedLane); return ( @@ -72,15 +87,23 @@ export function PlayView({ text, onText, laneEngines = LANE_ENGINES }: PlayViewP onToggleFidelity={() => setFidelityGate((v) => !v)} fuzzyDedup={fuzzyDedup} onToggleFuzzy={() => setFuzzyDedup((v) => !v)} + riskGate={riskGate} + onToggleRisk={() => setRiskGate((v) => !v)} + quantumLock={quantumLock} + onToggleQuantum={() => setQuantumLock((v) => !v)} + heatmap={heatmapMode} + onToggleHeatmap={toggleHeatmap} />
{batch?.combined && (
- Fluxo combinado — {active.join(" → ")} + Fluxo combinado — {active.join(" → ")}{" "} +
+
)}
@@ -100,6 +123,14 @@ export function PlayView({ text, onText, laneEngines = LANE_ENGINES }: PlayViewP
)} + {batch?.heatmap && ( +
+
+ Saliency heatmap — {batch.heatmap.mode} +
+ +
+ )}
); diff --git a/src/app/(dashboard)/dashboard/compression/studio/PlaygroundInput.tsx b/src/app/(dashboard)/dashboard/compression/studio/PlaygroundInput.tsx index e1e894824f..9c3977a7b1 100644 --- a/src/app/(dashboard)/dashboard/compression/studio/PlaygroundInput.tsx +++ b/src/app/(dashboard)/dashboard/compression/studio/PlaygroundInput.tsx @@ -1,7 +1,7 @@ "use client"; export const LANE_ENGINES = ["session-dedup", "ccr", "lite", "rtk", "ionizer", "headroom", "caveman", "aggressive", "ultra"] as const; -export interface PlaygroundInputProps { text: string; onText: (t: string) => void; active: string[]; onToggleActive: (engine: string) => void; onRun: () => void; loading: boolean; fidelityGate: boolean; onToggleFidelity: () => void; fuzzyDedup: boolean; onToggleFuzzy: () => void; } -export function PlaygroundInput({ text, onText, active, onToggleActive, onRun, loading, fidelityGate, onToggleFidelity, fuzzyDedup, onToggleFuzzy }: PlaygroundInputProps) { +export interface PlaygroundInputProps { text: string; onText: (t: string) => void; active: string[]; onToggleActive: (engine: string) => void; onRun: () => void; loading: boolean; fidelityGate: boolean; onToggleFidelity: () => void; fuzzyDedup: boolean; onToggleFuzzy: () => void; riskGate: boolean; onToggleRisk: () => void; quantumLock: boolean; onToggleQuantum: () => void; heatmap: "ultra" | "universal" | false; onToggleHeatmap: () => void; } +export function PlaygroundInput({ text, onText, active, onToggleActive, onRun, loading, fidelityGate, onToggleFidelity, fuzzyDedup, onToggleFuzzy, riskGate, onToggleRisk, quantumLock, onToggleQuantum, heatmap, onToggleHeatmap }: PlaygroundInputProps) { return (
🚫 Never hit limits
Auto-fallback across 231 providers in milliseconds. Quota out? Next provider takes over — zero downtime.
🚫 Never hit limits
Auto-fallback across 237 providers in milliseconds. Quota out? Next provider takes over — zero downtime.
💸 Save up to 95% tokens
RTK + Caveman stacked compression cuts 15–95% of eligible tokens (~89% avg on tool-heavy sessions).
🆓 $0 to start
50+ providers with a free tier, 11 free forever (Kiro, Qoder, Pollinations, LongCat…). No card needed.
Claude Code
Claude Code
Codex CLI
Codex CLI
Gemini CLI
Gemini CLI
Cursor
Cursor
Copilot
Copilot
Continue
Continue
Cloudflare AI
50+ models
10K neurons/day
Gemini CLI
gemini-3-flash
180K/mo free
NVIDIA NIM
129 models
~40 RPM free
Cerebras
Qwen3 235B
1M tokens/day