Files
OmniRoute/open-sse/executors/default.ts
Diego Rodrigues de Sa e Souza a7ae9550bd Release v3.8.36 (#4854)
* chore(release): open v3.8.36 development cycle

* refactor(chatCore): extrai resolveCompressionSettings (#3501) (#4826)

Integrated into release/v3.8.36 (#3501 chatCore extraction stack 1/13)

* refactor(chatCore): extrai predicados puros de combo de compressão (#3501) (#4824)

Integrated into release/v3.8.36 (#3501 chatCore extraction stack 2/13)

* refactor(chatCore): extrai emitOutputStyleTelemetry (#3501) (#4811)

Integrated into release/v3.8.36 (#3501 chatCore extraction stack 3/13)

* refactor(chatCore): extrai writeCompressionAnalytics (bloco analytics completo, #3501) (#4817)

Integrated into release/v3.8.36 (#3501 chatCore extraction stack 4/13)

* refactor(chatCore): extrai runPluginOnRequestHook (#3501) (#4827)

Integrated into release/v3.8.36 (#3501 chatCore extraction stack 5/13)

* refactor(chatCore): extrai applyClientUsageBuffer (buffer/estimate de usage non-streaming, #3501) (#4832)

Integrated into release/v3.8.36 (#3501 chatCore extraction stack 6/13)

* refactor(chatCore): extrai buildPostCallGuardrailContext (contexto guardrail post-call, #3501) (#4831)

Integrated into release/v3.8.36 (#3501 chatCore extraction stack 7/13)

* refactor(chatCore): extrai storeSemanticCacheResponse (cache-store non-streaming, #3501) (#4828)

Integrated into release/v3.8.36 (#3501 chatCore extraction stack 8/13)

* refactor(chatCore): extrai buildNonStreamingResponseHeaders (headers de resposta non-streaming, #3501) (#4835)

Integrated into release/v3.8.36 (#3501 chatCore extraction stack 9/13)

* refactor(chatCore): extrai maybeConvertJsonBodyToSse (#3089 JSON→SSE streaming, #3501) (#4833)

Integrated into release/v3.8.36 (#3501 chatCore extraction stack 10/13)

* refactor(chatCore): extrai assembleStreamingResponseHeaders (headers de resposta streaming, #3501) (#4836)

Integrated into release/v3.8.36 (#3501 chatCore extraction stack 11/13)

* refactor(chatCore): extrai storeStreamingSemanticCacheResponse (cache-store streaming, #3501) (#4829)

Integrated into release/v3.8.36 (#3501 chatCore extraction stack 12/13)

* refactor(chatCore): extrai assembleStreamingPipeline (chain de transforms streaming, #3501) (#4837)

Integrated into release/v3.8.36 (#3501 chatCore extraction stack 13/13)

* ci(quality): shift heavy validations to the PR→release fast-path (release-acceleration) (#4857)

* feat(quality): add check:test-runner-api gate (vitest-only dirs must use vitest API)

* feat(release): reusable CHANGELOG i18n-mirror sync script

* chore(ops): add prune-stale-worktrees.sh (dry-run by default)

* ci(quality): run test-runner-api + docs-all + vitest + full unit suite on PR->release fast-path

---------

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* fix(quota): cota exclusiva lista qtSd/ no /v1/models (#4806) + limite EPSILON não bloqueia (#4830)

Integrated into release/v3.8.36 — quota-exclusive qtSd/ listing (#4806) + EPSILON placeholder no longer blocks; rebuilt from stale base (3 defining commits cherry-picked clean over release tip)

* feat(sse): add Google Flow video-generation provider (#4569) (#4769)

Integrated into release/v3.8.36 — Google Flow video-generation provider (#4569), release-green validated (typecheck + 21 tests + file-size)

* fix(api): auth on compression run-telemetry + document OMNIROUTE_EVAL_CREDENTIALS (#4694, #4720) (#4796)

Integrated into release/v3.8.36 — auth on compression run-telemetry + OMNIROUTE_EVAL_CREDENTIALS doc, release-green validated (typecheck + 3 tests + env-doc-sync)

* fix(translator): strip top-level client_metadata on the OpenAI passthrough (port from 9router#1157) (#4624)

Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated)

* fix(translator): normalize `developer` role to `system` for OpenAI-format providers (#4625)

Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated)

* fix(translator): emit </think> close marker for Anthropic thinking blocks (#4633)

Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated)

* fix(translator): normalize tools to Anthropic-native shape for non-Anthropic providers (#4650)

Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated)

* fix(gemini): preserve `pattern` in antigravity tool schema sanitizer (#4651)

Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated)

* fix(perplexity): validate API keys via /v1/models endpoint (#4654)

Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated)

* fix(image): prevent compatible nodes from shadowing provider aliases (#4656)

Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated)

* fix(cli-tools): tolerate JSONC (comments, trailing commas) in tool settings (#4659)

Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated)

* fix(security): validate kiro region to prevent SSRF (GHSA-6mwv-4mrm-5p3m) (#4629)

Integrated into release/v3.8.36 — kiro region SSRF guard (GHSA-6mwv-4mrm-5p3m), port rebuilt clean over release tip

* fix(cli): harden the systray2 tray runtime (port of 9router#1080) (#4628)

Integrated into release/v3.8.36 — port rebuilt clean over release tip, release-green validated

* fix(test): validate anthropic-compatible connections via POST /v1/messages (#4657)

Integrated into release/v3.8.36 — anthropic-compat validation via POST /v1/messages (port 584cf66a), rebuilt clean + baseline; release-green

* fix(executors): strip params unsupported by the target provider/model (#4658)

Integrated into release/v3.8.36 — port rebuilt clean over release tip, release-green validated

* fix(claude-oauth): respect 429 backoff on usage endpoint to reduce spam (#4655)

Integrated into release/v3.8.36 — port rebuilt clean over release tip, release-green validated

* feat(api/v1): include alias-backed models in /v1/models listing (#4630)

Integrated into release/v3.8.36 — port rebuilt clean over release tip, release-green validated

* chore(quality): rebaseline catalog.ts 1574->1577 (#4630 aliases sobre quota-exclusive da release) (#4879)

rebaseline

* feat(compression): Kiro/CodeWhisperer tool-result compression engine (#4635)

Integrated into release/v3.8.36 — port rebuilt clean, release-green

* fix(security): don't trust loopback socket as local when behind reverse proxy (#4632)

Integrated into release/v3.8.36 — port rebuilt clean, release-green

* fix(opencode): preserve DeepSeek reasoning content in streamed responses (#4631)

Integrated into release/v3.8.36 — DeepSeek reasoning_content injection (port #1099); release-green

* fix(copilot,antigravity): cap maxOutputTokens at 16384 to stop "Invalid Argument" 400 (#4636)

Integrated into release/v3.8.36 — cap maxOutputTokens 16384 antigravity (port #779); release-green

* fix(dashboard): show custom vision models in LLM selector (#4653)

Integrated into release/v3.8.36 — custom vision models in LLM selector (port 5e5e78d3); release-green

* fix(claude): omit adaptive thinking + output_config.effort for haiku (#4661)

Integrated into release/v3.8.36 — haiku adaptive-thinking omit (port); release-green

* feat(provider): CodeBuddy CN (copilot.tencent.com) — full stack (#4664)

Integrated into release/v3.8.36 — CodeBuddy CN provider (port efd20be8); usage.ts import + public-creds allowlist line reconciled; release-green

* feat(combo): Fusion strategy — parallel panel + judge synthesis (16th strategy) (#4652)

Integrated into release/v3.8.36 — Fusion combo strategy (16th, port 87e5c1c6); combo.ts baseline reconciled; release-green

* feat(proxy-pool): Deno Deploy relays + group action buttons (#4643)

Integrated into release/v3.8.36 — Deno Deploy relays (port #1437); proxies.ts baseline reconciled + env docs restored; release-green

* fix(security): pin image fetch DNS resolution to prevent SSRF rebinding (GHSA-cmhj-wh2f-9cgx) (#4634)

Integrated into release/v3.8.36 — pin DNS for image fetch SSRF rebinding guard (GHSA-cmhj-wh2f-9cgx, port c7d07448); caller DNS stubs + test-file baseline reconciled; release-green

* fix(github): route Copilot Codex models to /responses (port from 9router#102) (#4626)

Integrated into release/v3.8.36 — route Copilot Codex models to /responses (port #102); release-green

* fix(copilot): never route Gemini/Claude variants to /responses (chat-completions only) (#4627)

Integrated into release/v3.8.36 — never route Gemini/Claude to /responses (port #1536); fused with #4626 codex routing via supportsResponsesEndpoint gate; release-green

* docs(ops): add canonical incident response runbook (#4868)

Integrated into release/v3.8.36

* docs(perf): add per-endpoint p50/p95/p99 latency + cost budgets (#4867)

Integrated into release/v3.8.36

* fix(proxy): fan out direct dispatcher streams (#4803)

Integrated into release/v3.8.36

* fix(antigravity): exclude standard Gemini rate limit message from quota exhaustion keywords (#4810)

Integrated into release/v3.8.36

* fix(sse): skip third-party tool-name cloak for Anthropic server tools (#4808)

Integrated into release/v3.8.36

* fix(install): make transformers optional for CUDA-host installs (#4807)

Integrated into release/v3.8.36

* fix(combo): propagate selected connection ID to fallback error responses for correct model lockout (#4809)

Integrated into release/v3.8.36

* fix db storage tuning settings (#4834)

Integrated into release/v3.8.36

* fix(sse): drop ccp pin when pinned provider is durably unhealthy (failover + anti-flap) (#4864)

Integrated into release/v3.8.36

* fix(claude): skip mcp__ tool-name cloak + guard missing connectionId (#4861)

Integrated into release/v3.8.36

* chore(quality): reconcile env-doc + file-size base-reds in release/v3.8.36 (#4886)

- env-doc-sync: document PIN_DROP_BACKOFF_LEVEL / PIN_DROP_GRACE_MS (added by the
  ccp-pin health gate #4864) in .env.example + ENVIRONMENT.md.
- file-size: rebaseline image-generation-handler.test.ts 1996 -> 2019 to its actual
  size (pre-existing drift).

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* fix(codex): drop non-standard codex.* events that break responses.stream (env-gated, #4602) (#4715)

Integrated into release/v3.8.36

* feat(routing): honor X-Route-Model header to override body.model (#4863)

Integrated into release/v3.8.36

* feat(live-ws): allow non-loopback clients via LIVE_WS_ALLOWED_HOSTS (closes #4873) (#4877)

Integrated into release/v3.8.36 (live-ws + combo-api commits; Tailscale CGNAT commit held pending opt-in/opt-out decision)

* chore(claude,codex): bump pinned CLI identity — Claude 2.1.158→2.1.187, Codex 0.132.0→0.142.0 (#4883)

Integrated into release/v3.8.36

* fix(security): SSRF allowlist bypass via x-relay-path nos relays Deno/Vercel (#4899)

Integrated into release/v3.8.36

* feat(quota): recuperação proativa de conexões em cooldown (cron heal) [Fase 3 #8] (#4900)

Integrated into release/v3.8.36

* fix(quota): policy inválida não vaza allow + guard connectionIds vazio [Fase 3 #10] (#4901)

Integrated into release/v3.8.36

* feat(quota): saturação real do Claude no fair-share via /api/oauth/usage (#4885)

Integrated into release/v3.8.36

* chore(dashboard): rename Qoder display label from "Qoder AI" to "Qoder" (#4733)

Integrated into release/v3.8.36

* fix(ci): include coverage/lcov.info in coverage-report artifact for SonarQube (#4670)

Integrated into release/v3.8.36

* fix(cli): bump better-sqlite3 runtime pin to 12.10.1 for Node 26 (#4685)

Integrated into release/v3.8.36

* docs: clarify Kiro is ~50 credits/month per account, not unlimited (#4690)

Integrated into release/v3.8.36

* docs(agentbridge): document Electron NODE_EXTRA_CA_CERTS, real model IDs, identity caveat (#4718)

Integrated into release/v3.8.36

* docs(ops): document the release-green family (green-prs, check:release-green, babysit, nightly) (#4679)

Integrated into release/v3.8.36

* fix(translator): replay reasoning_content on plain Xiaomi MiMo turns (port from 9router#1321) (#4639)

Integrated into release/v3.8.36

* feat(opencode-go): advertise glm-5.2 and kimi-k2.7-code (align with official Go endpoints) (#4711)

Integrated into release/v3.8.36

* feat(db): track API endpoint dimension on usage_history (#4676)

Integrated into release/v3.8.36 (migration renumbered 103→105; endpoint plumbed through extracted usage-stats helpers)

* fix(cli): SIGKILL systray child PID before IPC close to avoid macOS NSStatusItem orphan (#4732)

Integrated into release/v3.8.36

* feat(proxy-pool): Cloudflare Workers proxy deployer + pool integration (#4640)

Integrated into release/v3.8.36 (relay type added to RELAY_TYPES set; dropdown UX preserved + Cloudflare item added; proxies.ts file-size rebaselined 1057→1060)

* chore(quality): conserta base-red de release/v3.8.36 (gates + 7 testes + build MDX) (#4915)

A base tinha base-red sistêmica herdada de PRs de outras sessões, bloqueando
TODOS os PRs do ciclo (o TIA roda a suíte full em fail-safe p/ diffs hub).

4 Fast Quality Gates:
- test-discovery (#4877): live-server-allowlist.test.ts em tests/unit/server/
  (não-coletado) + vitest → nunca rodava. Convertido p/ node:test em tests/unit/security/.
- any-budget:t11 (#4664): 3 explicit-any em tokenRefresh.ts tipados (sem crescer file-size).
- docs-symbols (#4868): rotas inexistentes → /api/system/version e
  PUT /api/providers/{id} {isActive:false}.
- docs-all fabricated-claim (#4868 + #4718): 5 bin/*.sh reais criados (rollback,
  snapshot-data, restore-data, restore-policies, cold-start-bench) + _ops-common.sh
  (snapshot VACUUM INTO, guards de confirmação/TTY, testes de contrato); NODE_EXTRA_CA_CERTS
  (env de runtime Node) na allowlist do checker.

7 testes unit base-red (de features alheias à quota):
- oauth-providers-config (#4664): teste alinhado ao provider codebuddy-cn do registry.
- antigravity-model-aliases (#4636): maxOutputTokens esperado 32769→16384 (cap intencional).
- provider-request-capture #4091 (#4861): exemplo do teste trocado de mcp__ (que #4861
  isenta de cloak por causa dos 400s de assimetria de histórico) para um tool de terceiro
  cloakável — preserva o invariante de #4091 SEM reverter #4861.
- combo-error-response: convertido de vitest p/ node:test (era coletado pelo glob node:test
  e crashava); api/** e server/** removidos do vitest.config (config morta).

Build MDX (dast-smoke, #4679):
- docs/ops/RELEASE_GREEN.md não tinha frontmatter `title` → fumadocs-mdx rejeitava no
  webpack compile ("invalid frontmatter: title expected string"), quebrando o next build
  (e o deploy). Frontmatter title adicionado (único doc do collection sem ele).

17/17 Fast Quality Gates + suíte unit completa (17737 testes, 0 fail) + vitest verdes localmente.

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* feat(quota): saturação proativa por headers de tokens (universal) [Fase 3 #2] (#4907)

storeRateLimitHeaders só capturava os headers de REQUESTS (RPM/min), que não
refletem a pressão de TOKENS. Agora também parseia os headers de tokens (em toda
resposta, sucesso também) para throttle proativo antes do 429:
- Anthropic: anthropic-ratelimit-tokens-{limit,remaining,reset} (+ input/output), RFC3339.
- OpenAI: x-ratelimit-{limit,remaining,reset}-tokens, reset em duração (6m0s).

saturation = 1 − remaining/limit; resetAt normalizado a epoch (parse de duração
ReDoS-safe). getTokenHeaderSaturation por (provider, connectionId). fetchGeneric-
Saturation passa a usar esse sinal (complementa o oauth/usage do #1, que segue
primário p/ Claude). Fail-open, cache mantido, request-path inalterado.

16 testes novos + regressão (oauth/usage #1 8/8, signals 6/6) = 30/30;
typecheck:core + eslint limpos.

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* feat(quota): estratégia de combo "headroom" — seleção por folga de cota [Fase 3 #4] (#4908)

Nova estratégia de roteamento que escolhe a conexão com MAIS folga de plano:
headroom = 1 − max(util_5h, util_7d) (técnica do dario), via getSaturation
(melhorado p/ Claude no #1). Proativo em vez de só fill-first reativo.

- Helper PURO headroomRanking.ts (computeHeadroom + rankByHeadroom; saturação
  injetada, não-mutante, tie-break estável, fail-open).
- Orderer async em combo/quotaStrategies.ts (reusa a maquinaria reset-aware de
  expansão de conexões + concorrência limitada; seam injetável).
- Registrada como "headroom" em routingStrategies (combo-only); fill-first segue
  default — nenhuma estratégia existente tocada.
- baseline file-size combo.ts 3168->3180 (só +12L de dispatch; lógica fora do god-file).

16 testes novos + combo-strategies 15/15 = 31/31; typecheck:core + eslint +
file-size limpos.

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* feat(quota): cap per-(key,model) — quota_allocation_model_caps [Fase 3 #7] (#4927)

* feat(quota): cap per-(key,model) com tabela quota_allocation_model_caps [Fase 3 #7]

Fecha o buraco onde uma API key pode drenar o pool inteiro consumindo um único modelo.

Tabela nova: quota_allocation_model_caps(pool_id, api_key_id, model, cap_value, cap_unit)
PK composta (pool_id, api_key_id, model). cap_unit alinhado ao QuotaUnit existente.

Comportamento: keyA acima do cap para modelo M → bloqueada somente em M; ainda
permitida em qualquer outro modelo no mesmo pool. Cap <= EPSILON → ignorado (seed).

Consumo por-(key,model) usa bucket segregado no quota_consumption existente
(poolId mangled ':model:<model>') com window fixa 'hourly'; nenhuma nova tabela
ou método de store necessário.

Módulo novo: src/lib/db/quotaModelCaps.ts (getModelCap/setModelCap/deleteModelCap/listModelCaps)
enforce.ts ganha o pre-check em enforceQuotaShare + recording em recordConsumption.
EnforceInput e RecordConsumptionInput ganham model?: string (backward-compatible).
localDb.ts re-exporta os 4 helpers (Hard Rule #2).

TDD: tests/unit/quota-per-key-model.test.ts — 4 cenários (bloqueia em M, permite em M2,
sem cap → sem bloqueio, EPSILON → ignorado). Todos os gates de qualidade passam.

* feat(quota): plumba model resolvido no hot path para ativar o per-(key,model) cap [Fase 3 #7]

A tabela/enforce do commit anterior estavam INERTES: o hot path não passava `model`
ao enforce nem ao record, então nenhum model-cap disparava em produção.

Plumbagem (model resolvido = mesma var usada no log/roteamento, pós background-redirect/alias):
- chatCore.ts: enforceQuotaShare ganha `model`; scheduleQuotaShareConsumption recebe `model`.
- chatCore/quotaShareConsumption.ts: threade `model` no RecordConsumptionInput (non-streaming).
- spendRecorder.ts: recordStreamingConsumption já recebia `model` — agora o coloca no
  RecordConsumptionInput (streaming accrue por-modelo).
- embeddings.ts: enforce + record ganham `model`.

Namespace do cap = id do modelo RESOLVIDO (o mesmo de modelForScope/pendingScope/getUnsupportedParams),
não o requestedModel cru nem o finalModelToUpstream (sem prefixo de provider). Operador configura
o cap contra esse id. `model || undefined` em todos os pontos: vazio/null → check pulado (fail-safe,
zero latência — só um campo no objeto).

Teste de integração novo (tests/unit/quota-per-key-model-hotpath.test.ts): prova end-to-end que
N consumos via scheduleQuotaShareConsumption({model}) → enforceQuotaShare({model}) bloqueia, e que
outro modelo no mesmo pool ainda passa; + guard de que enforce SEM model nunca dispara model-cap.

---------

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* feat(quota): session stickiness p/ integridade de prompt-cache [Fase 3 #5] (#4929)

* feat(quota): session stickiness p/ integridade de prompt-cache [Fase 3 #5]

Adiciona stickiness de sessão ao roteamento de combo: uma conversa multi-turno
é roteada para a MESMA conexão enquanto ela permanecer saudável, evitando a
perda do prompt-cache do provider (custo 5-10× sem stickiness, efeito conhecido
no dario/clewdr).

Implementação:
- `open-sse/services/combo/sessionStickiness.ts` (novo, <800 linhas):
  mapa em memória (messageHash → connectionId) com TTL 15 min + cap 500 entradas;
  `applySessionStickiness` promove a conexão sticky ao índice 0 dos targets
  ordenados pelo strategy, guardado por `computeHeadroom > 0.15` (threshold);
  quando saturada (headroom ≤ 0.15), o binding é limpo e a seleção normal reage.
  Hash da sessão = SHA-256 dos primeiros chars da 1ª mensagem user → 16 hex chars.
  Seam de teste: `__setStickinessHeadroomFetcherForTests`.
- `open-sse/services/combo.ts`: import + 2 pontos de integração (pré-eval-scores
  e pós-success), dentro do orçamento congelado de 3180 linhas.
- `tests/unit/combo-session-stickiness.test.ts`: 19 testes node:test + assert/strict,
  todos via injeção de fetcher (zero rede/DB).

Threshold 0.15: conexão a >85% de utilização está a um burst de rate-limit;
o benefício de cache não compensa manter-se numa conexão degradada. Valor
alinhado com a zona de soft-penalty do restante do engine de quota-share.

* test(combo): isola combo-strategies da session stickiness (#5)

selectedConnectionFor reusa o mesmo body, então o sticky map (#5) fixava a
connection após a 1ª chamada e quebrava o round-robin tie-break do teste
reset-aware. Limpa o sticky map no início da helper — a stickiness tem suíte
própria (combo-session-stickiness). Sem enfraquecer asserts.

---------

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* feat(quota): buckets multi-janela por conexão (5h/7d/per-model) [Fase 3 #3] (#4928)

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* refactor(providers): decompõe catálogo providers.ts em módulos de dados (godfile sweep, #3501) (#4917)

Integrado em release/v3.8.36 (godfile sweep providers.ts, #3501)

* refactor(pricing): decompõe pricing.ts em shared-tiers + DEFAULT_PRICING particionado (godfile sweep, #3501) (#4918)

Integrado em release/v3.8.36 (godfile sweep pricing.ts, #3501)

* refactor(api): extrai camada-folha pura de validation.ts (URL/headers/transport) (#4921)

Integrado em release/v3.8.36 (validation.ts split fatia 1 — leaf layer)

* refactor(api): extrai validators web-cookie + Meta AI de validation.ts (#4922)

Integrado em release/v3.8.36 (validation.ts split fatia 2 — web-cookie + Meta AI)

* refactor(api): extrai validators enterprise-cloud + probe compartilhado de validation.ts (#4923)

Integrado em release/v3.8.36 (validation.ts split fatia 3 — enterprise-cloud + probe)

* refactor(api): extrai validators áudio/speech + misc apikey de validation.ts (#4930)

Integrado em release/v3.8.36 (validation.ts split fatia 4 — áudio/speech + misc apikey)

* feat(quota): estratégia dedicada de quota-share (DRR + P2C in-flight + gating per-model) [Fase 3 #9] (#4939)

* feat(quota): estratégia dedicada de quota-share (DRR + P2C in-flight + gating per-model) [Fase 3 #9]

Estratégia interna "quota-share" isolada num módulo dedicado — NÃO toca a seleção/
fair-share genérica (decisão do dono: não mexer no que já funciona). Os combos qtSd/
(quotaCombos.ts) passam de fill-first para essa strategy; combo.ts ganha só 1 branch
de dispatch que delega 100% ao módulo (nenhum case existente alterado).

- quotaShareStrategy.ts: gating per-model (isBucketSaturated do #3) + DRR (quantum
  proporcional ao weight) + P2C sobre carga in-flight.
- quotaShareInflight.ts: contador in-flight com TTL/lease de 120s — fallback do
  decrement-on-abort sem precisar instrumentar o combo genérico.
- "quota-share" registrada como strategy INTERNA (não exposta na UI).
- testes de síntese (quota-combo-balancing, quota-multiprovider) alinhados: a strategy
  esperada dos combos qtSd/ passa de "fill-first" para "quota-share" (alinhamento ao novo
  comportamento intencional, não mascaramento — os 73 testes de qtSd/ seguem verdes).

* test(quota-share): alinha 2 scope-guards ao godfile sweep (base-reds que bloqueavam o CI)

Dois testes de "arquivo contém X" quebraram por decomposições de godfile que outras
sessões mergearam no release DURANTE a validação de #9 — NÃO são regressão de #9
(que não toca validation/oauth). Alinhados ao novo layout, asserts preservados:
- proxy-bypass-scope-guard #3226: bypassProxyPatch foi extraído de validation.ts para
  validation/headers.ts (split #4921–#4930) → o teste lê a camada de validação.
- sse-error-passthrough #3324: a windsurf authHint foi extraída de providers.ts para
  providers/oauth.ts → o teste lê o novo local.

---------

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* refactor(api): extrai validators search + embedding/rerank de validation.ts (#4932)

Integrated into release/v3.8.36

* refactor(api): extrai format-validators (OpenAI/Anthropic) de validation.ts (#4933)

Integrated into release/v3.8.36

* refactor(db): extrai model-permission matching de db/apiKeys.ts (#4936)

Integrated into release/v3.8.36

* refactor(db): extrai row-parsers + tipos compartilhados de db/apiKeys.ts (#4943)

Integrated into release/v3.8.36

* refactor(db): extrai column-mapping (snake↔camel) de db/core.ts (#4947)

Integrated into release/v3.8.36

* refactor(db): extrai schema-column reconciliation de db/core.ts (#4948)

Integrated into release/v3.8.36

* refactor(sse): extrai scalar/format helpers de services/usage.ts (#4949)

Integrated into release/v3.8.36

* refactor(sse): extrai quota-core (UsageQuota + builders) de services/usage.ts (#4950)

Integrated into release/v3.8.36

* fix(translator): regroup parallel tool results adjacent to their assistant (#4714) (#4882)

Integrated into release/v3.8.36 (fixes #4714)

* fix(qoder): exchange PAT for jt-* job token before Cosy chat (#4683) (#4884)

Integrated into release/v3.8.36 (fixes #4683)

* refactor(sse): dedup fallback tool_call id helper (#4736)

Integrated into release/v3.8.36

* refactor(open-sse): extract safeParseJSON util, dedup tryParseJSON (#4735)

Integrated into release/v3.8.36

* fix(compression): eliminate ReDoS in math_inline preservation pattern (#4795) (#4838)

Integrated into release/v3.8.36 (fixes #4795)

* fix(combo): fetch models dynamically from custom provider endpoints (#4860)

Integrated into release/v3.8.36

* feat(providers): update volcengine-ark model list with DeepSeek V4 (#4905)

Integrated into release/v3.8.36

* fix(translator): provider thinking compatibility (DeepSeek/Gemini) (#4946)

Integrated into release/v3.8.36

* feat(combo): task-aware routing strategy (#4945)

Integrated into release/v3.8.36

* refactor(sse): extrai a família MiniMax de services/usage.ts (#4952)

Integrated into release/v3.8.36

* refactor(sse): extrai a família GLM de services/usage.ts (#4953)

Integrated into release/v3.8.36

* refactor(sse): extrai a família Antigravity de services/usage.ts (#4956)

Integrated into release/v3.8.36

* fix(dashboard): show custom provider given-name instead of internal id across dashboard pages (#4603) (#4960)

Integrated into release/v3.8.36 (fixes #4603)

* fix(api): evict stale in-memory rate-limit windows to stop slow heap leak (#4041) (#4957)

Integrated into release/v3.8.36 (fixes #4041)

* fix(api): parse /v1/responses body once instead of 3-4x on the hot path (#4041) (#4958)

Integrated into release/v3.8.36 (fixes #4041)

* fix(translator): preserve legitimate empty-string tool arguments in openai-to-claude streaming (#4951) (#4959)

Integrated into release/v3.8.36 (fixes #4951)

* chore(quality): reconcile file-size baseline for #4960 provider-display-name (#4961)

Integrated into release/v3.8.36

* fix(dashboard): restore home provider-topology card hidden by #4596 default (#4963)

Integrated into release/v3.8.36 — restores home topology card (#4596 regression)

* fix(build): drop @omniroute/open-sse from optimizePackageImports (build OOM) (#4968)

Integrated into release/v3.8.36 — fixes build OOM (optimizePackageImports open-sse)

* fix(quota): migração 107 ativa estratégia quota-share nos combos qtSd/ existentes [Fase 3 #9] (#4962)

Integrated into release/v3.8.36

* feat(quota): respeita max_concurrent por conexão no roteamento (#4965)

Integrated into release/v3.8.36

* feat(quota): combo quota-share espera cooldown curto e re-despacha (Variante A) (#4967)

Integrated into release/v3.8.36

* fix(quality): resolve base-reds da release — db-rules allowlist + task-aware router precedence (#4973)

Dois base-reds pré-existentes que reprovavam o CI da release v3.8.36 (Fast
Quality Gates + Unit Tests fast-path), independentes de qualquer feature em voo:

1. check:db-rules / allowlist: os módulos db-internal caseMapping (#4947) e
   schemaColumns (#4948), extraídos de db/core.ts e importados só por ele, não
   estavam em INTENTIONALLY_INTERNAL. Registrados na allowlist (correção
   canônica — são internos legítimos, não re-exportados pelo localDb).

2. auto-strategy honra LKGP/cost (combo-routing-engine.test.ts, 2 testes): o
   task-aware reordering (#4945, reorderByTaskWeight) roda para strategy "auto" e
   era aplicado DEPOIS do router explícito (selectWithStrategy: lkgp/cost),
   sobrescrevendo o orderedTargets[0] que o operador escolheu. Instrumentação
   provou: post-filter [0]=claude (LKGP) → post-task [0]=gpt-oss. Correção: quando
   o auto usa router explícito, preserva o [0] dele e deixa o task-aware refinar
   só a cauda de fallback. gpt-oss-120b PERMANECE tool-capable (não é mudança de
   catálogo; o model-capabilities-registry test segue verde).

Validado: 121 testes (combo-routing-engine + combo-task-aware + registry) verdes,
red-check confirmado, db-rules/file-size/typecheck/lint/prettier OK.

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* feat(quota): serializa concorrência por conexão no caminho quota-share (FASE 2.1) (#4970)

O gating de quota-share em selectQuotaShareTarget é fail-open: uma conexão
at-cap só é despriorizada, nunca bloqueada. Com 1 conexão por conta de
assinatura (caso comum), chamadas concorrentes ainda floodam a conta (→ 429 +
cooldown) — provado live na .15: 3 chamadas concorrentes com max_concurrent=1
despacharam todas em 94ms.

Adiciona um semáforo POR CONEXÃO em torno do dispatch quota-share: chamadas
excedentes esperam na fila em vez de floodar (key qsconn:<connectionId>, cap =
max_concurrent da conexão). Fail-open em fila saturada/timeout para nunca
piorar disponibilidade. Gated por strategy===quota-share + kill-switch
resilienceSettings.quotaShareConcurrencyLimit (default on; UI no ResilienceTab).

Lógica extraível isolada no leaf puro combo/quotaShareConcurrency.ts
(unit-testado: estabilidade da key, no-op sem cap, serialização real,
fail-open). Settings + schema + UI espelham comboCooldownWait.

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* docs(resilience): document Quota-Share Concurrency Control (max_concurrent + serialization + cooldown-wait) (#4980)

Documents the v3.8.36 quota-share concurrency layers in RESILIENCE_GUIDE.md:
per-connection max_concurrent cap, the quota-share request serialization semaphore
(FASE 2.1, qsconn:<connectionId>, fail-open, kill-switch), and the combo
cooldown-aware retry — so operators know how to cap a subscription account's
concurrency and why the routing gate alone cannot contain a single-connection flood.

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* fix(dashboard): proxy-pool success gating, sync timestamp, opt-in Redis (#4878) (#4988)

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* fix(sse): fail over on 400 responses carrying rate-limit text (#4976) (#4986)

* fix(sse): fail over on 400 responses carrying rate-limit text (#4976)

* chore(quality): rebaseline accountFallback.ts file-size for #4976 fix

---------

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* fix(compression): stop RTK over-truncating file-read tool results (#4559) (#4987)

* fix(compression): stop RTK over-truncating file-read tool results (#4559)

* chore(quality): trim #4559 comment to keep rtk/index.ts within size cap

---------

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* fix(sse): honor per-account proxies and fingerprint rotation in opencode executor (#4954) (#4989)

* fix(sse): honor per-account proxies and fingerprint rotation in opencode executor (#4954)

* chore(quality): rebaseline auth.ts file-size for #4954 (+39: synthetic no-auth providerSpecificData hydration of fingerprints/accountProxies; irreducible credential-path wiring, covered by opencode-proxy-rotation-4954.test.ts + 159 auth/noauth regression)

---------

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* fix(sse): soft-penalize exhausted providers in auto-combo scoring (#4540) (#4990)

* fix(sse): soft-penalize exhausted providers in auto-combo scoring (#4540)

* chore(quality): document STATUS_SOFT_DEPRIORITIZE_FACTOR + rebaseline combo.ts for #4540

---------

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* fix(dashboard): switch to visible filter after auto-hiding failed models in test-all (#4887) (#4991)

* fix(dashboard): switch to visible filter after auto-hiding failed models in OAuth provider test-all (#4887)

* test(dashboard): move #4887 test into tests/unit/ui so a CI runner collects it

---------

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* fix(pollinations): only enable jsonMode when JSON output is requested (#3981) (#5009)

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* fix(antigravity): default safetySettings to all-OFF for parity with native Gemini paths (#5003) (#5008)

* fix(antigravity): default safetySettings to all-OFF for parity with native Gemini paths (#5003)

* docs(changelog): restore #3981 pollinations entry eaten by merge

---------

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* fix(chatgpt-web): map advertised gpt-5.5/5.4-pro/5.2-pro slugs to prevent silent model substitution (#4665) (#5010)

* fix(chatgpt-web): map advertised gpt-5.5/5.4-pro/5.2-pro slugs to prevent silent model substitution (#4665)

MODEL_MAP was missing the advertised catalog ids gpt-5.5, gpt-5.5-pro,
gpt-5.4-pro and gpt-5.2-pro, so MODEL_MAP[model] ?? model sent the dot-form
id verbatim to the ChatGPT backend-api, which silently rejected it and served
the default Plus model. Map each to its dash-form slug. gpt-4-5 is already
dash-form and falls through correctly, so it is intentionally left unmapped.

Extends the executor MODEL_MAP test with the four ids and adds a drift guard
asserting every advertised dot-form catalog id reaches the backend in dash-form
(never verbatim), guarding future catalog<->map drift.

file-size: tests/unit/chatgpt-web.test.ts frozen baseline 2809->2855 (+46) for
the added test cases and drift-guard test; executor source unchanged in baseline.

* docs(changelog): restore #3981/#5003 entries eaten by merge

---------

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* feat(combos): add editable per-combo description field persisted via /api/combos (#5005) (#5011)

* feat(combos): add editable per-combo description field persisted via /api/combos (#5005)

* docs(changelog): restore #3981/#5003/#4665 entries eaten by merge

---------

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>

* Fix Ollama Cloud max reasoning effort (#4993)

Integrated into release/v3.8.36

* fix(copilot): replace execSync with execFile to prevent command injection (#5024)

Integrated into release/v3.8.36

* fix(plugin): auth.json dual-key fallback for auto-prefix migration (#5027)

Integrated into release/v3.8.36

* feat(endpoint): per-endpoint custom system prompt injection (#5022)

Integrated into release/v3.8.36

* fix(headroom): translate openai-responses input through OpenAI for compression (#5023)

Integrated into release/v3.8.36

* docs(changelog): add entries for #4993, #5024, #5027 (release notes credit)

* fix(api): stop /api/system/env/repair 500 on packaged install (#5006) (#5028)

* fix(api): stop /api/system/env/repair 500 on packaged install — lazy createRequire in sync-env.mjs (#5006)

scripts/dev/sync-env.mjs ran createRequire(import.meta.url) at module
top-level. When webpack bundles it into the standalone env-repair route,
import.meta.url is frozen to the build-machine path (file:///home/runner/...)
and createRequire throws during module evaluation, so the whole route
module fails to load and every GET returns HTTP 500 — breaking the
onboarding wizard on packaged/global installs.

- Move createRequire into the guarded better-sqlite3 block (only place
  that needs it); a bad import.meta.url now returns the safe default.
- resolveRootDir() falls back to process.cwd() when fileURLToPath throws.
- route.ts passes an explicit rootDir (process.cwd()) so the helper never
  derives the root from the frozen import.meta.url, matching the .env
  target used by createEnvBackup().
- Regression guard: assert sync-env.mjs has no top-level createRequire +
  getEnvSyncPlan(oauth) works with explicit rootDir without throwing.

* docs(changelog): restore #4993/#5023/#5024/#5027 + custom-system-prompt/headroom entries eaten by release merge

* chore(quality): rebaseline 3 inherited base-reds from release merge

Files NOT touched by this PR — grew on release/v3.8.36 via --admin merges and
inherited here through 'git merge origin/release':
- open-sse/executors/base.ts 1414->1416 (#4993 Ollama Cloud max-effort)
- src/lib/db/settings.ts 1149->1151 (#5023 custom system prompt)
- src/app/(dashboard)/.../endpoint/EndpointPageClient.tsx 2570->2612 (custom system prompt UI)

* chore(release): finalize v3.8.36 CHANGELOG + docs (2026-06-25)

---------

Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com>
Co-authored-by: KooshaPari <42529354+KooshaPari@users.noreply.github.com>
Co-authored-by: Makcim Ivanov <makcimbx@gmail.com>
Co-authored-by: Chewji <126886556+Chewji9875@users.noreply.github.com>
Co-authored-by: Anton <39598727+NomenAK@users.noreply.github.com>
Co-authored-by: Demiurge The Single <megamen932@gmail.com>
Co-authored-by: Randi <55005611+rdself@users.noreply.github.com>
Co-authored-by: Éder Costa <eder.almeida.costa@gmail.com>
Co-authored-by: Jefferson Felizardo <jeffer1312@gmail.com>
Co-authored-by: Arthur Bodera <abodera@gmail.com>
Co-authored-by: Hamsa_M <116961508+hamsa0x7@users.noreply.github.com>
Co-authored-by: Hernan Javier Ardila Sanchez <hjasgr@gmail.com>
2026-06-25 13:17:40 -03:00

766 lines
30 KiB
TypeScript

import { BaseExecutor, setUserAgentHeader, type ExecuteInput } from "./base.ts";
import { PROVIDERS, OAUTH_ENDPOINTS } from "../config/constants.ts";
import { getAccessToken } from "../services/tokenRefresh.ts";
import {
getRotatingApiKey,
getValidApiKey,
resolveKeyForRequest,
} from "../services/apiKeyRotator.ts";
import type { KeyHealth } from "../services/apiKeyRotator.ts";
import {
buildClaudeCodeCompatibleHeaders,
CLAUDE_CODE_COMPATIBLE_DEFAULT_CHAT_PATH,
joinClaudeCodeCompatibleUrl,
} from "../services/claudeCodeCompatible.ts";
import { getGigachatAccessToken } from "../services/gigachatAuth.ts";
import { getRegistryEntry } from "../config/providerRegistry.ts";
import { mergeClientAnthropicBeta } from "../config/anthropicHeaders.ts";
import { applyProviderRequestDefaults } from "../services/providerRequestDefaults.ts";
import { stripUnsupportedParams } from "../translator/paramSupport.ts";
import {
detectFormat,
getOpenAICompatibleType,
getTargetFormat,
isClaudeCodeCompatible,
} from "../services/provider.ts";
import { sanitizeQwenThinkingToolChoice } from "../services/qwenThinking.ts";
import { buildDataRobotChatUrl } from "../config/datarobot.ts";
import { buildAzureAiChatUrl } from "../config/azureAi.ts";
import { buildWatsonxChatUrl } from "../config/watsonx.ts";
import { buildOciChatUrl } from "../config/oci.ts";
import { buildSapChatUrl, getSapResourceGroup } from "../config/sap.ts";
import { buildMaritalkChatUrl } from "../config/maritalk.ts";
import { LOCAL_PROVIDERS } from "@/shared/constants/providers";
import { isForbiddenCustomHeaderName } from "@/shared/constants/upstreamHeaders";
import { getClaudeCodeCompatibleRequestDefaults } from "@/lib/providers/requestDefaults";
import type { PoolConfig } from "../services/sessionPool/types.ts";
/**
* Apply operator-configured per-provider custom headers onto an outgoing header
* map. Defense-in-depth on top of the Zod `customHeadersSchema`:
* - skip hop-by-hop/framing AND auth header names (canonical denylist, so a row
* written before the schema tightening still can't override credential auth);
* - skip control-char (CR/LF/NUL) names/values before they reach undici;
* - assign case-insensitively, replacing any existing same-named header (e.g.
* the executor's own Content-Type/Accept) instead of emitting a duplicate.
* Used for every *-compatible node, INCLUDING anthropic-compatible-cc-* (whose
* header builder returns early, so custom headers must be merged in explicitly).
*/
function applyCustomHeaders(headers: Record<string, string>, rawCustomHeaders: unknown): void {
let customHeaders: Record<string, unknown> | null = null;
if (
rawCustomHeaders &&
typeof rawCustomHeaders === "object" &&
!Array.isArray(rawCustomHeaders)
) {
customHeaders = rawCustomHeaders as Record<string, unknown>;
} else if (typeof rawCustomHeaders === "string") {
try {
const parsed = JSON.parse(rawCustomHeaders);
if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
customHeaders = parsed as Record<string, unknown>;
}
} catch {
/* ignore invalid JSON */
}
}
if (!customHeaders) return;
for (const [k, v] of Object.entries(customHeaders)) {
if (typeof k !== "string" || typeof v !== "string") continue;
if (isForbiddenCustomHeaderName(k)) continue;
if (/[\r\n\0]/.test(k) || /[\r\n]/.test(v)) continue;
const lower = k.toLowerCase();
for (const existing of Object.keys(headers)) {
if (existing.toLowerCase() === lower) delete headers[existing];
}
headers[k] = v;
}
}
function normalizeBaseUrl(baseUrl) {
return (baseUrl || "").trim().replace(/\/$/, "");
}
function normalizeBailianMessagesUrl(baseUrl) {
const normalized = normalizeBaseUrl(baseUrl).replace(/\?beta=true$/, "");
const messagesUrl = normalized.endsWith("/messages") ? normalized : `${normalized}/messages`;
return messagesUrl;
}
function normalizeHerokuChatUrl(baseUrl) {
const normalized = normalizeBaseUrl(baseUrl);
if (normalized.endsWith("/v1/chat/completions")) return normalized;
return `${normalized}/v1/chat/completions`;
}
function normalizeDatabricksChatUrl(baseUrl) {
const normalized = normalizeBaseUrl(baseUrl);
if (normalized.endsWith("/chat/completions")) return normalized;
return `${normalized}/chat/completions`;
}
function normalizeDataRobotChatUrl(baseUrl) {
return buildDataRobotChatUrl(baseUrl);
}
function normalizeAzureAiChatUrl(baseUrl: string, apiType: "chat" | "responses" = "chat") {
return buildAzureAiChatUrl(baseUrl, apiType);
}
function normalizeWatsonxChatUrl(baseUrl: string) {
return buildWatsonxChatUrl(baseUrl);
}
function normalizeOciChatUrl(baseUrl: string, apiType: "chat" | "responses" = "chat") {
return buildOciChatUrl(baseUrl, apiType);
}
function normalizeSapChatUrl(baseUrl) {
return buildSapChatUrl(baseUrl);
}
function normalizeXiaomiMimoChatUrl(baseUrl) {
const normalized = normalizeBaseUrl(baseUrl).replace(/\/chat\/completions$/, "");
return `${normalized}/chat/completions`;
}
function normalizeSnowflakeChatUrl(baseUrl) {
const normalized = normalizeBaseUrl(baseUrl)
.replace(/\/cortex\/inference:complete$/, "")
.replace(/\/api\/v2$/, "");
return `${normalized}/api/v2/cortex/inference:complete`;
}
function normalizeGigachatChatUrl(baseUrl) {
const normalized = normalizeBaseUrl(baseUrl).replace(/\/chat\/completions$/, "");
return `${normalized}/chat/completions`;
}
function normalizeOpenAIChatUrl(baseUrl) {
const normalized = normalizeBaseUrl(baseUrl);
if (
normalized.endsWith("/chat/completions") ||
normalized.endsWith("/responses") ||
normalized.endsWith("/chat")
) {
return normalized;
}
if (normalized.endsWith("/v1")) {
return `${normalized}/chat/completions`;
}
// Assume OpenAI-compatible /v1/chat/completions path structure
// when the base URL is a bare hostname or custom path (e.g. llama.cpp, vLLM, LM Studio).
return `${normalized}/v1/chat/completions`;
}
function getOpenRouterConnectionPreset(
providerSpecificData?: Record<string, unknown> | null
): string | null {
const preset =
typeof providerSpecificData?.preset === "string" ? providerSpecificData.preset.trim() : "";
return preset || null;
}
export class DefaultExecutor extends BaseExecutor {
constructor(provider) {
super(provider, PROVIDERS[provider] || PROVIDERS.openai);
const registryEntry = getRegistryEntry(provider);
if (registryEntry?.poolConfig) {
this.poolConfig = registryEntry.poolConfig as PoolConfig;
}
}
buildUrl(model, stream, urlIndex = 0, credentials = null) {
void model;
void stream;
void urlIndex;
if (this.provider?.startsWith?.("openai-compatible-")) {
const psd = credentials?.providerSpecificData;
const baseUrl = psd?.baseUrl || "https://api.openai.com/v1";
const normalized = baseUrl.replace(/\/$/, "");
const customPath = typeof psd?.chatPath === "string" && psd.chatPath ? psd.chatPath : null;
if (customPath) return `${normalized}${customPath}`;
const path =
getOpenAICompatibleType(this.provider, psd) === "responses"
? "/responses"
: "/chat/completions";
return `${normalized}${path}`;
}
if (this.provider?.startsWith?.("anthropic-compatible-")) {
const psd = credentials?.providerSpecificData;
const baseUrl = psd?.baseUrl || "https://api.anthropic.com/v1";
const customPath = typeof psd?.chatPath === "string" && psd.chatPath ? psd.chatPath : null;
if (isClaudeCodeCompatible(this.provider)) {
return joinClaudeCodeCompatibleUrl(
baseUrl,
customPath || CLAUDE_CODE_COMPATIBLE_DEFAULT_CHAT_PATH
);
}
const normalized = baseUrl.replace(/\/$/, "");
return `${normalized}${customPath || "/messages"}`;
}
switch (this.provider) {
case "bailian-coding-plan": {
const baseUrl = credentials?.providerSpecificData?.baseUrl || this.config.baseUrl;
return normalizeBailianMessagesUrl(baseUrl);
}
case "heroku": {
const baseUrl = credentials?.providerSpecificData?.baseUrl || this.config.baseUrl;
return normalizeHerokuChatUrl(baseUrl);
}
case "databricks": {
const baseUrl = credentials?.providerSpecificData?.baseUrl || this.config.baseUrl;
return normalizeDatabricksChatUrl(baseUrl);
}
case "datarobot": {
const baseUrl = credentials?.providerSpecificData?.baseUrl || this.config.baseUrl;
return normalizeDataRobotChatUrl(baseUrl);
}
case "azure-ai": {
const forceResponses =
credentials?.providerSpecificData?._omnirouteForceResponsesUpstream === true;
const apiType =
forceResponses || credentials?.providerSpecificData?.apiType === "responses"
? "responses"
: "chat";
const baseUrl = credentials?.providerSpecificData?.baseUrl || this.config.baseUrl;
return normalizeAzureAiChatUrl(baseUrl, apiType);
}
case "watsonx": {
const baseUrl = credentials?.providerSpecificData?.baseUrl || this.config.baseUrl;
return normalizeWatsonxChatUrl(baseUrl);
}
case "oci": {
const forceResponses =
credentials?.providerSpecificData?._omnirouteForceResponsesUpstream === true;
const apiType =
forceResponses || credentials?.providerSpecificData?.apiType === "responses"
? "responses"
: "chat";
const baseUrl = credentials?.providerSpecificData?.baseUrl || this.config.baseUrl;
return normalizeOciChatUrl(baseUrl, apiType);
}
case "sap": {
const baseUrl = credentials?.providerSpecificData?.baseUrl || this.config.baseUrl;
return normalizeSapChatUrl(baseUrl);
}
case "xiaomi-mimo": {
const baseUrl = credentials?.providerSpecificData?.baseUrl || this.config.baseUrl;
return normalizeXiaomiMimoChatUrl(baseUrl);
}
case "snowflake": {
const baseUrl = credentials?.providerSpecificData?.baseUrl || this.config.baseUrl;
return normalizeSnowflakeChatUrl(baseUrl);
}
case "gigachat": {
const baseUrl = credentials?.providerSpecificData?.baseUrl || this.config.baseUrl;
return normalizeGigachatChatUrl(baseUrl);
}
case "maritalk": {
const baseUrl = credentials?.providerSpecificData?.baseUrl || this.config.baseUrl;
return buildMaritalkChatUrl(baseUrl);
}
case "siliconflow": {
const baseUrl = credentials?.providerSpecificData?.baseUrl || this.config.baseUrl;
return normalizeOpenAIChatUrl(baseUrl);
}
case "llama-cpp":
case "lm-studio":
case "modal":
case "reka":
case "vllm":
case "lemonade":
case "llamafile":
case "triton":
case "docker-model-runner":
case "xinference":
case "oobabooga": {
// #3197 (residual of #3136): for self-hosted/local providers, prefer the
// catalog's localDefault when no explicit baseUrl is set. `this.config`
// falls back to PROVIDERS.openai for providers not in the open-sse
// registry (llama-cpp, etc.), so without this guard an empty baseUrl
// silently hits OpenAI's API. Fall back to localDefault BEFORE config.
const localDefault = LOCAL_PROVIDERS[this.provider]?.localDefault;
const baseUrl =
credentials?.providerSpecificData?.baseUrl || localDefault || this.config.baseUrl;
return normalizeOpenAIChatUrl(baseUrl);
}
case "zai":
case "glm-coding-apikey": {
const zaiBaseUrl = credentials?.providerSpecificData?.baseUrl || this.config.baseUrl;
return `${zaiBaseUrl}?beta=true`;
}
case "claude":
case "glm":
case "glmt":
case "kimi-coding":
case "minimax":
case "minimax-cn":
return `${this.config.baseUrl}?beta=true`;
case "gemini":
return `${this.config.baseUrl}/${model}:${stream ? "streamGenerateContent?alt=sse" : "generateContent"}`;
case "qwen": {
const resourceUrl = credentials?.providerSpecificData?.resourceUrl;
return `https://${resourceUrl || "portal.qwen.ai"}/v1/chat/completions`;
}
default: {
// Honor a user-supplied custom base URL (providerSpecificData.baseUrl) for
// OpenAI-format providers (e.g. the built-in "openai" provider pointed at a
// proxy/gateway). Without this, a configured custom base URL was silently
// ignored and requests always hit the hardcoded this.config.baseUrl
// (https://api.openai.com/v1/...). Scoped to openai-format providers so
// non-OpenAI default-branch providers keep their existing behavior.
const customBaseUrl =
typeof credentials?.providerSpecificData?.baseUrl === "string" &&
credentials.providerSpecificData.baseUrl.trim()
? (credentials.providerSpecificData.baseUrl as string)
: null;
const isOpenAIFormat = !this.config.format || this.config.format === "openai";
if (customBaseUrl && isOpenAIFormat) {
return normalizeOpenAIChatUrl(customBaseUrl);
}
const url = this.config.baseUrl;
const entry = getRegistryEntry(this.provider);
return entry?.urlSuffix ? `${url}${entry.urlSuffix}` : url;
}
}
}
buildHeaders(credentials, stream = true, clientHeaders?: Record<string, string> | null) {
const headers = { "Content-Type": "application/json", ...this.config.headers };
// Allow per-provider User-Agent override via environment variable.
const providerId = this.config?.id || this.provider;
if (providerId) {
const envKey = `${providerId.toUpperCase().replace(/[^A-Z0-9]/g, "_")}_USER_AGENT`;
const envUA = process.env[envKey]?.trim();
if (envUA) {
headers["User-Agent"] = envUA;
if ("user-agent" in headers) {
headers["user-agent"] = envUA;
}
}
}
// T07: resolve extra keys round-robin locally since DefaultExecutor overrides BaseExecutor buildHeaders
const extraKeys =
(credentials.providerSpecificData?.extraApiKeys as string[] | undefined) ?? [];
const selectedKeyId = (credentials.providerSpecificData as Record<string, unknown> | undefined)
?.selectedKeyId as string | undefined;
let effectiveKey = credentials.apiKey;
if (extraKeys.length > 0 && credentials.connectionId && credentials.apiKey) {
const resolved = resolveKeyForRequest(
credentials.connectionId,
credentials.apiKey,
extraKeys,
selectedKeyId ?? null
);
effectiveKey = resolved?.key ?? credentials.apiKey;
if (resolved && credentials.providerSpecificData) {
(credentials.providerSpecificData as Record<string, unknown>).selectedKeyId =
resolved.keyId;
}
}
switch (this.provider) {
case "gemini":
effectiveKey
? (headers["x-goog-api-key"] = effectiveKey)
: (headers["Authorization"] = `Bearer ${credentials.accessToken}`);
break;
case "snowflake": {
const rawToken = effectiveKey || credentials.accessToken || "";
const usesProgrammaticAccessToken = rawToken.startsWith("pat/");
headers["Authorization"] =
`Bearer ${usesProgrammaticAccessToken ? rawToken.slice(4) : rawToken}`;
headers["X-Snowflake-Authorization-Token-Type"] = usesProgrammaticAccessToken
? "PROGRAMMATIC_ACCESS_TOKEN"
: "KEYPAIR_JWT";
break;
}
case "gigachat":
headers["Authorization"] = `Bearer ${credentials.accessToken || effectiveKey}`;
break;
case "clarifai": {
const clarifaiToken = effectiveKey || credentials.accessToken;
if (clarifaiToken) {
headers["Authorization"] = `Key ${clarifaiToken}`;
}
break;
}
case "azure-ai":
if (effectiveKey || credentials.accessToken) {
headers["api-key"] = effectiveKey || credentials.accessToken;
}
delete headers["Authorization"];
break;
case "oci": {
const bearerToken = effectiveKey || credentials.accessToken;
if (bearerToken) {
headers["Authorization"] = `Bearer ${bearerToken}`;
}
const projectId =
credentials.projectId ||
credentials?.providerSpecificData?.projectId ||
credentials?.providerSpecificData?.project;
if (projectId) {
headers["OpenAI-Project"] = projectId;
}
break;
}
case "sap": {
const bearerToken = effectiveKey || credentials.accessToken;
if (bearerToken) {
headers["Authorization"] = `Bearer ${bearerToken}`;
}
headers["AI-Resource-Group"] = getSapResourceGroup(credentials?.providerSpecificData);
break;
}
case "reka": {
const bearerToken = effectiveKey || credentials.accessToken;
if (bearerToken) {
headers["Authorization"] = `Bearer ${bearerToken}`;
headers["X-Api-Key"] = bearerToken;
}
break;
}
case "maritalk": {
const token = effectiveKey || credentials.accessToken;
if (token) {
headers["Authorization"] = `Key ${token}`;
}
break;
}
case "claude":
case "anthropic":
effectiveKey
? (headers["x-api-key"] = effectiveKey)
: (headers["Authorization"] = `Bearer ${credentials.accessToken}`);
break;
case "glm":
case "glmt":
case "kimi-coding":
case "bailian-coding-plan":
case "kimi-coding-apikey":
case "zai":
case "glm-coding-apikey":
headers["x-api-key"] = effectiveKey || credentials.accessToken;
break;
default:
if (isClaudeCodeCompatible(this.provider)) {
const ccRequestDefaults = getClaudeCodeCompatibleRequestDefaults(
credentials?.providerSpecificData
);
const ccHeaders = buildClaudeCodeCompatibleHeaders(
effectiveKey || credentials.accessToken || "",
stream,
credentials?.providerSpecificData?.ccSessionId,
{ redactThinking: ccRequestDefaults.redactThinking === true }
);
// CC nodes are also anthropic-compatible-*, so honor operator custom
// headers here (the early return skips the shared block below).
applyCustomHeaders(ccHeaders, credentials.providerSpecificData?.customHeaders);
return ccHeaders;
}
if (this.provider?.startsWith?.("anthropic-compatible-")) {
if (effectiveKey) {
headers["x-api-key"] = effectiveKey;
} else if (credentials.accessToken) {
headers["Authorization"] = `Bearer ${credentials.accessToken}`;
}
if (!headers["anthropic-version"]) {
headers["anthropic-version"] = "2023-06-01";
}
} else {
// Use registry authHeader if available, otherwise default to bearer
const entry = getRegistryEntry(this.provider);
const authHeader = entry?.authHeader || "bearer";
const token = effectiveKey || credentials.accessToken;
if (token) {
if (authHeader === "x-api-key") {
headers["x-api-key"] = token;
} else if (authHeader === "x-goog-api-key") {
headers["x-goog-api-key"] = token;
} else {
headers["Authorization"] = `Bearer ${token}`;
}
}
}
}
headers["Accept"] = stream ? "text/event-stream" : "application/json";
// Qwen header cleanup: Remove X-Dashscope-* headers if using an API key (DashScope compatible mode).
// If using OAuth (Qwen Code), we MUST keep them for portal.qwen.ai to accept the request.
if (this.provider === "qwen" && effectiveKey) {
for (const key of Object.keys(headers)) {
if (key.toLowerCase().startsWith("x-dashscope-")) {
delete headers[key];
}
}
}
const isCompatibleProvider =
this.provider?.startsWith?.("openai-compatible-") ||
this.provider?.startsWith?.("anthropic-compatible-");
if (isCompatibleProvider) {
applyCustomHeaders(headers, credentials.providerSpecificData?.customHeaders);
}
// Forward client request metadata headers (from OpenCode or similar clients)
// Allowlist-based: only specific x-opencode-* headers and User-Agent are forwarded
if (clientHeaders) {
const clientUA = clientHeaders["User-Agent"] || clientHeaders["user-agent"];
if (clientUA) {
setUserAgentHeader(headers, clientUA);
}
const opencodeHeaderKeys = [
"x-opencode-session",
"x-opencode-request",
"x-opencode-project",
"x-opencode-client",
];
for (const headerName of opencodeHeaderKeys) {
const value = Object.entries(clientHeaders).find(
([key]) => key.toLowerCase() === headerName.toLowerCase()
)?.[1];
if (value) {
headers[headerName] = value;
}
}
// #3974: merge the client's negotiated anthropic-beta (allowlisted) into the
// outbound set. The registry's static ANTHROPIC_BETA_CLAUDE_OAUTH lacks
// tool-search-tool-2025-10-19, so deferred-tool requests were rejected with
// 400 "Tool reference not found". Allowlist-merge preserves it without
// forwarding betas the backend rejects.
const clientBeta = clientHeaders["anthropic-beta"] ?? clientHeaders["Anthropic-Beta"] ?? null;
const betaKey = Object.keys(headers).find((key) => key.toLowerCase() === "anthropic-beta");
if (betaKey && clientBeta) {
headers[betaKey] = mergeClientAnthropicBeta(headers[betaKey], clientBeta);
}
}
return headers;
}
/**
* For compatible providers, the model name is already clean by the time
* it reaches the executor (chatCore sets body.model = modelInfo.model,
* which is the parsed model ID without internal routing prefixes).
*
* Models may legitimately contain "/" as part of their ID (e.g. "zai-org/GLM-5-FP8",
* "org/model-name") — we must NOT strip path segments. (Fix #493)
*/
transformRequest(model, body, stream, credentials) {
const cleanedBody = super.transformRequest(model, body, stream, credentials);
let withDefaults = applyProviderRequestDefaults(cleanedBody, this.config.requestDefaults);
const targetFormat = getTargetFormat(this.provider, credentials?.providerSpecificData);
const requestFormat =
withDefaults && typeof withDefaults === "object" && !Array.isArray(withDefaults)
? detectFormat(withDefaults as Record<string, unknown>)
: "openai";
if (typeof withDefaults === "object" && withDefaults !== null && !Array.isArray(withDefaults)) {
if (this.provider?.startsWith?.("anthropic-compatible-")) {
if (Object.prototype.hasOwnProperty.call(withDefaults, "stream_options")) {
const withoutStreamOptions = { ...withDefaults } as Record<string, unknown>;
delete withoutStreamOptions.stream_options;
withDefaults = withoutStreamOptions;
}
} else if (stream && targetFormat === "openai" && requestFormat !== "openai-responses") {
// Port of decolua/9router#663 (closes upstream #557): Qwen rejects with
// 400 "'stream_options' only set this when you set stream: true" when the
// outgoing body carries `stream: false` (Claude Code / Claude-Code-
// compatible callers force the executor-level stream flag on via
// `upstreamStream = stream || isClaudeCodeCompatible`, but the body keeps
// the caller's original `stream: false`). Same upstream also rejects the
// injection when `thinking` / `enable_thinking` is set. Skip injection in
// those cases instead of unconditionally adding `stream_options`.
const defaultsRecord = withDefaults as Record<string, unknown>;
const qwenBlocksStreamOptions =
this.provider === "qwen" &&
(defaultsRecord.stream === false ||
Boolean(defaultsRecord.thinking) ||
Boolean(defaultsRecord.enable_thinking));
if (qwenBlocksStreamOptions) {
if (Object.prototype.hasOwnProperty.call(defaultsRecord, "stream_options")) {
const withoutStreamOptions = { ...defaultsRecord };
delete withoutStreamOptions.stream_options;
withDefaults = withoutStreamOptions;
}
} else if (!credentials?.providerSpecificData?.disableStreamOptions) {
withDefaults = {
...withDefaults,
stream_options: {
...((defaultsRecord.stream_options as object) || {}),
include_usage: true,
},
};
} else if (Object.prototype.hasOwnProperty.call(withDefaults, "stream_options")) {
const withoutStreamOptions = { ...withDefaults } as Record<string, unknown>;
delete withoutStreamOptions.stream_options;
withDefaults = withoutStreamOptions;
}
} else if (!stream && Object.prototype.hasOwnProperty.call(withDefaults, "stream_options")) {
// #3884: stream_options is only valid on streaming requests. NVIDIA NIM
// (and the OpenAI spec) reject "Stream options can only be defined when
// stream=True" on non-streaming calls. Strip any client-sent
// stream_options when the outbound request is not streaming.
const withoutStreamOptions = { ...withDefaults } as Record<string, unknown>;
delete withoutStreamOptions.stream_options;
withDefaults = withoutStreamOptions;
} else if (
(targetFormat === "openai-responses" || requestFormat === "openai-responses") &&
Object.prototype.hasOwnProperty.call(withDefaults, "stream_options")
) {
const withoutStreamOptions = { ...withDefaults } as Record<string, unknown>;
delete withoutStreamOptions.stream_options;
withDefaults = withoutStreamOptions;
}
// #1961: Map max_tokens -> max_completion_tokens for recent OpenAI models
if (targetFormat === "openai") {
const isRecentOpenAI = /^(o1|o3|o4|gpt-5)/i.test(model);
if (isRecentOpenAI && withDefaults && typeof withDefaults === "object") {
const defaultsRecord = withDefaults as Record<string, unknown>;
if ("max_tokens" in defaultsRecord) {
defaultsRecord.max_completion_tokens = defaultsRecord.max_tokens;
delete defaultsRecord.max_tokens;
}
}
}
if (this.provider === "openrouter") {
const connectionPreset = getOpenRouterConnectionPreset(credentials?.providerSpecificData);
if (connectionPreset && (withDefaults as Record<string, unknown>).preset === undefined) {
withDefaults = {
...(withDefaults as Record<string, unknown>),
preset: connectionPreset,
};
}
}
}
if (this.provider === "qwen" && typeof withDefaults === "object" && withDefaults !== null) {
return sanitizeQwenThinkingToolChoice(
withDefaults as Record<string, unknown>,
"QwenExecutor"
);
}
// Config-driven strip of params unsupported by the target provider/model
// (e.g. claude-opus-4 deprecated `temperature` → Anthropic 400). Port from
// 9router#7ae9fff6 (fixes upstream #1748). Rules live in
// ../translator/paramSupport.ts so adding one means editing one table.
if (typeof withDefaults === "object" && withDefaults !== null) {
const bodyRecord = withDefaults as Record<string, unknown>;
const outboundModel =
typeof bodyRecord.model === "string" ? bodyRecord.model : model;
stripUnsupportedParams(this.provider, outboundModel, bodyRecord);
}
// Apply modelIdPrefix from RegistryEntry (e.g. "accounts/fireworks/models/")
// so registry can store short model IDs while the upstream API receives the full path.
if (typeof withDefaults === "object" && withDefaults !== null) {
const entry = getRegistryEntry(this.provider);
if (entry?.modelIdPrefix) {
const body = withDefaults as Record<string, unknown>;
if (typeof body.model === "string") {
// Skip prepending when the model already carries the canonical prefix OR any
// other accepted fully-qualified prefix (e.g. Fireworks router IDs). #3133.
const acceptedPrefixes = [entry.modelIdPrefix, ...(entry.acceptedModelIdPrefixes ?? [])];
const alreadyQualified = acceptedPrefixes.some((prefix) =>
(body.model as string).startsWith(prefix)
);
if (!alreadyQualified) {
body.model = `${entry.modelIdPrefix}${body.model}`;
}
}
}
}
return withDefaults;
}
/**
* Refresh credentials via the centralized tokenRefresh service.
* Delegates to getAccessToken() which handles all providers with
* race-condition protection (deduplication via refreshPromiseCache).
*/
async refreshCredentials(credentials, log) {
if (this.provider === "gigachat") {
if (!credentials.apiKey) return null;
try {
return await getGigachatAccessToken({
credentials: credentials.apiKey,
});
} catch (error) {
log?.error?.("TOKEN", `gigachat refresh error: ${error.message}`);
return null;
}
}
if (!credentials.refreshToken) return null;
try {
return await getAccessToken(this.provider, credentials, log);
} catch (error) {
log?.error?.("TOKEN", `${this.provider} refresh error: ${error.message}`);
return null;
}
}
needsRefresh(credentials) {
if (this.provider === "gigachat") {
if (credentials.apiKey && !credentials.accessToken) return true;
if (!credentials.expiresAt) return false;
}
return super.needsRefresh(credentials);
}
async execute(input: ExecuteInput) {
const pool = this.getPool();
if (!pool) return super.execute(input);
const session = pool.acquire();
if (session) {
input.upstreamExtraHeaders = {
...session.buildHeaders(),
...input.upstreamExtraHeaders,
};
}
let result;
try {
result = await super.execute(input);
} catch (err) {
if (session) {
pool.reportCooldown(session);
session.release();
}
throw err;
}
if (session) {
try {
const status = result?.response?.status;
if (status === 429) {
pool.reportCooldown(session);
} else if (status >= 500) {
pool.reportDead(session);
} else {
pool.reportSuccess(session);
}
} finally {
session.release();
}
}
return result;
}
}
export default DefaultExecutor;