mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-07-26 09:52:11 +03:00
* chore(release): open v3.8.36 development cycle * refactor(chatCore): extrai resolveCompressionSettings (#3501) (#4826) Integrated into release/v3.8.36 (#3501 chatCore extraction stack 1/13) * refactor(chatCore): extrai predicados puros de combo de compressão (#3501) (#4824) Integrated into release/v3.8.36 (#3501 chatCore extraction stack 2/13) * refactor(chatCore): extrai emitOutputStyleTelemetry (#3501) (#4811) Integrated into release/v3.8.36 (#3501 chatCore extraction stack 3/13) * refactor(chatCore): extrai writeCompressionAnalytics (bloco analytics completo, #3501) (#4817) Integrated into release/v3.8.36 (#3501 chatCore extraction stack 4/13) * refactor(chatCore): extrai runPluginOnRequestHook (#3501) (#4827) Integrated into release/v3.8.36 (#3501 chatCore extraction stack 5/13) * refactor(chatCore): extrai applyClientUsageBuffer (buffer/estimate de usage non-streaming, #3501) (#4832) Integrated into release/v3.8.36 (#3501 chatCore extraction stack 6/13) * refactor(chatCore): extrai buildPostCallGuardrailContext (contexto guardrail post-call, #3501) (#4831) Integrated into release/v3.8.36 (#3501 chatCore extraction stack 7/13) * refactor(chatCore): extrai storeSemanticCacheResponse (cache-store non-streaming, #3501) (#4828) Integrated into release/v3.8.36 (#3501 chatCore extraction stack 8/13) * refactor(chatCore): extrai buildNonStreamingResponseHeaders (headers de resposta non-streaming, #3501) (#4835) Integrated into release/v3.8.36 (#3501 chatCore extraction stack 9/13) * refactor(chatCore): extrai maybeConvertJsonBodyToSse (#3089 JSON→SSE streaming, #3501) (#4833) Integrated into release/v3.8.36 (#3501 chatCore extraction stack 10/13) * refactor(chatCore): extrai assembleStreamingResponseHeaders (headers de resposta streaming, #3501) (#4836) Integrated into release/v3.8.36 (#3501 chatCore extraction stack 11/13) * refactor(chatCore): extrai storeStreamingSemanticCacheResponse (cache-store streaming, #3501) (#4829) Integrated into release/v3.8.36 (#3501 chatCore extraction stack 12/13) * refactor(chatCore): extrai assembleStreamingPipeline (chain de transforms streaming, #3501) (#4837) Integrated into release/v3.8.36 (#3501 chatCore extraction stack 13/13) * ci(quality): shift heavy validations to the PR→release fast-path (release-acceleration) (#4857) * feat(quality): add check:test-runner-api gate (vitest-only dirs must use vitest API) * feat(release): reusable CHANGELOG i18n-mirror sync script * chore(ops): add prune-stale-worktrees.sh (dry-run by default) * ci(quality): run test-runner-api + docs-all + vitest + full unit suite on PR->release fast-path --------- Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * fix(quota): cota exclusiva lista qtSd/ no /v1/models (#4806) + limite EPSILON não bloqueia (#4830) Integrated into release/v3.8.36 — quota-exclusive qtSd/ listing (#4806) + EPSILON placeholder no longer blocks; rebuilt from stale base (3 defining commits cherry-picked clean over release tip) * feat(sse): add Google Flow video-generation provider (#4569) (#4769) Integrated into release/v3.8.36 — Google Flow video-generation provider (#4569), release-green validated (typecheck + 21 tests + file-size) * fix(api): auth on compression run-telemetry + document OMNIROUTE_EVAL_CREDENTIALS (#4694, #4720) (#4796) Integrated into release/v3.8.36 — auth on compression run-telemetry + OMNIROUTE_EVAL_CREDENTIALS doc, release-green validated (typecheck + 3 tests + env-doc-sync) * fix(translator): strip top-level client_metadata on the OpenAI passthrough (port from 9router#1157) (#4624) Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated) * fix(translator): normalize `developer` role to `system` for OpenAI-format providers (#4625) Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated) * fix(translator): emit </think> close marker for Anthropic thinking blocks (#4633) Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated) * fix(translator): normalize tools to Anthropic-native shape for non-Anthropic providers (#4650) Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated) * fix(gemini): preserve `pattern` in antigravity tool schema sanitizer (#4651) Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated) * fix(perplexity): validate API keys via /v1/models endpoint (#4654) Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated) * fix(image): prevent compatible nodes from shadowing provider aliases (#4656) Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated) * fix(cli-tools): tolerate JSONC (comments, trailing commas) in tool settings (#4659) Integrated into release/v3.8.36 — port (rebuilt from stale base; defining commit cherry-picked clean over release tip, release-green validated) * fix(security): validate kiro region to prevent SSRF (GHSA-6mwv-4mrm-5p3m) (#4629) Integrated into release/v3.8.36 — kiro region SSRF guard (GHSA-6mwv-4mrm-5p3m), port rebuilt clean over release tip * fix(cli): harden the systray2 tray runtime (port of 9router#1080) (#4628) Integrated into release/v3.8.36 — port rebuilt clean over release tip, release-green validated * fix(test): validate anthropic-compatible connections via POST /v1/messages (#4657) Integrated into release/v3.8.36 — anthropic-compat validation via POST /v1/messages (port 584cf66a), rebuilt clean + baseline; release-green * fix(executors): strip params unsupported by the target provider/model (#4658) Integrated into release/v3.8.36 — port rebuilt clean over release tip, release-green validated * fix(claude-oauth): respect 429 backoff on usage endpoint to reduce spam (#4655) Integrated into release/v3.8.36 — port rebuilt clean over release tip, release-green validated * feat(api/v1): include alias-backed models in /v1/models listing (#4630) Integrated into release/v3.8.36 — port rebuilt clean over release tip, release-green validated * chore(quality): rebaseline catalog.ts 1574->1577 (#4630 aliases sobre quota-exclusive da release) (#4879) rebaseline * feat(compression): Kiro/CodeWhisperer tool-result compression engine (#4635) Integrated into release/v3.8.36 — port rebuilt clean, release-green * fix(security): don't trust loopback socket as local when behind reverse proxy (#4632) Integrated into release/v3.8.36 — port rebuilt clean, release-green * fix(opencode): preserve DeepSeek reasoning content in streamed responses (#4631) Integrated into release/v3.8.36 — DeepSeek reasoning_content injection (port #1099); release-green * fix(copilot,antigravity): cap maxOutputTokens at 16384 to stop "Invalid Argument" 400 (#4636) Integrated into release/v3.8.36 — cap maxOutputTokens 16384 antigravity (port #779); release-green * fix(dashboard): show custom vision models in LLM selector (#4653) Integrated into release/v3.8.36 — custom vision models in LLM selector (port 5e5e78d3); release-green * fix(claude): omit adaptive thinking + output_config.effort for haiku (#4661) Integrated into release/v3.8.36 — haiku adaptive-thinking omit (port); release-green * feat(provider): CodeBuddy CN (copilot.tencent.com) — full stack (#4664) Integrated into release/v3.8.36 — CodeBuddy CN provider (port efd20be8); usage.ts import + public-creds allowlist line reconciled; release-green * feat(combo): Fusion strategy — parallel panel + judge synthesis (16th strategy) (#4652) Integrated into release/v3.8.36 — Fusion combo strategy (16th, port 87e5c1c6); combo.ts baseline reconciled; release-green * feat(proxy-pool): Deno Deploy relays + group action buttons (#4643) Integrated into release/v3.8.36 — Deno Deploy relays (port #1437); proxies.ts baseline reconciled + env docs restored; release-green * fix(security): pin image fetch DNS resolution to prevent SSRF rebinding (GHSA-cmhj-wh2f-9cgx) (#4634) Integrated into release/v3.8.36 — pin DNS for image fetch SSRF rebinding guard (GHSA-cmhj-wh2f-9cgx, port c7d07448); caller DNS stubs + test-file baseline reconciled; release-green * fix(github): route Copilot Codex models to /responses (port from 9router#102) (#4626) Integrated into release/v3.8.36 — route Copilot Codex models to /responses (port #102); release-green * fix(copilot): never route Gemini/Claude variants to /responses (chat-completions only) (#4627) Integrated into release/v3.8.36 — never route Gemini/Claude to /responses (port #1536); fused with #4626 codex routing via supportsResponsesEndpoint gate; release-green * docs(ops): add canonical incident response runbook (#4868) Integrated into release/v3.8.36 * docs(perf): add per-endpoint p50/p95/p99 latency + cost budgets (#4867) Integrated into release/v3.8.36 * fix(proxy): fan out direct dispatcher streams (#4803) Integrated into release/v3.8.36 * fix(antigravity): exclude standard Gemini rate limit message from quota exhaustion keywords (#4810) Integrated into release/v3.8.36 * fix(sse): skip third-party tool-name cloak for Anthropic server tools (#4808) Integrated into release/v3.8.36 * fix(install): make transformers optional for CUDA-host installs (#4807) Integrated into release/v3.8.36 * fix(combo): propagate selected connection ID to fallback error responses for correct model lockout (#4809) Integrated into release/v3.8.36 * fix db storage tuning settings (#4834) Integrated into release/v3.8.36 * fix(sse): drop ccp pin when pinned provider is durably unhealthy (failover + anti-flap) (#4864) Integrated into release/v3.8.36 * fix(claude): skip mcp__ tool-name cloak + guard missing connectionId (#4861) Integrated into release/v3.8.36 * chore(quality): reconcile env-doc + file-size base-reds in release/v3.8.36 (#4886) - env-doc-sync: document PIN_DROP_BACKOFF_LEVEL / PIN_DROP_GRACE_MS (added by the ccp-pin health gate #4864) in .env.example + ENVIRONMENT.md. - file-size: rebaseline image-generation-handler.test.ts 1996 -> 2019 to its actual size (pre-existing drift). Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * fix(codex): drop non-standard codex.* events that break responses.stream (env-gated, #4602) (#4715) Integrated into release/v3.8.36 * feat(routing): honor X-Route-Model header to override body.model (#4863) Integrated into release/v3.8.36 * feat(live-ws): allow non-loopback clients via LIVE_WS_ALLOWED_HOSTS (closes #4873) (#4877) Integrated into release/v3.8.36 (live-ws + combo-api commits; Tailscale CGNAT commit held pending opt-in/opt-out decision) * chore(claude,codex): bump pinned CLI identity — Claude 2.1.158→2.1.187, Codex 0.132.0→0.142.0 (#4883) Integrated into release/v3.8.36 * fix(security): SSRF allowlist bypass via x-relay-path nos relays Deno/Vercel (#4899) Integrated into release/v3.8.36 * feat(quota): recuperação proativa de conexões em cooldown (cron heal) [Fase 3 #8] (#4900) Integrated into release/v3.8.36 * fix(quota): policy inválida não vaza allow + guard connectionIds vazio [Fase 3 #10] (#4901) Integrated into release/v3.8.36 * feat(quota): saturação real do Claude no fair-share via /api/oauth/usage (#4885) Integrated into release/v3.8.36 * chore(dashboard): rename Qoder display label from "Qoder AI" to "Qoder" (#4733) Integrated into release/v3.8.36 * fix(ci): include coverage/lcov.info in coverage-report artifact for SonarQube (#4670) Integrated into release/v3.8.36 * fix(cli): bump better-sqlite3 runtime pin to 12.10.1 for Node 26 (#4685) Integrated into release/v3.8.36 * docs: clarify Kiro is ~50 credits/month per account, not unlimited (#4690) Integrated into release/v3.8.36 * docs(agentbridge): document Electron NODE_EXTRA_CA_CERTS, real model IDs, identity caveat (#4718) Integrated into release/v3.8.36 * docs(ops): document the release-green family (green-prs, check:release-green, babysit, nightly) (#4679) Integrated into release/v3.8.36 * fix(translator): replay reasoning_content on plain Xiaomi MiMo turns (port from 9router#1321) (#4639) Integrated into release/v3.8.36 * feat(opencode-go): advertise glm-5.2 and kimi-k2.7-code (align with official Go endpoints) (#4711) Integrated into release/v3.8.36 * feat(db): track API endpoint dimension on usage_history (#4676) Integrated into release/v3.8.36 (migration renumbered 103→105; endpoint plumbed through extracted usage-stats helpers) * fix(cli): SIGKILL systray child PID before IPC close to avoid macOS NSStatusItem orphan (#4732) Integrated into release/v3.8.36 * feat(proxy-pool): Cloudflare Workers proxy deployer + pool integration (#4640) Integrated into release/v3.8.36 (relay type added to RELAY_TYPES set; dropdown UX preserved + Cloudflare item added; proxies.ts file-size rebaselined 1057→1060) * chore(quality): conserta base-red de release/v3.8.36 (gates + 7 testes + build MDX) (#4915) A base tinha base-red sistêmica herdada de PRs de outras sessões, bloqueando TODOS os PRs do ciclo (o TIA roda a suíte full em fail-safe p/ diffs hub). 4 Fast Quality Gates: - test-discovery (#4877): live-server-allowlist.test.ts em tests/unit/server/ (não-coletado) + vitest → nunca rodava. Convertido p/ node:test em tests/unit/security/. - any-budget:t11 (#4664): 3 explicit-any em tokenRefresh.ts tipados (sem crescer file-size). - docs-symbols (#4868): rotas inexistentes → /api/system/version e PUT /api/providers/{id} {isActive:false}. - docs-all fabricated-claim (#4868 + #4718): 5 bin/*.sh reais criados (rollback, snapshot-data, restore-data, restore-policies, cold-start-bench) + _ops-common.sh (snapshot VACUUM INTO, guards de confirmação/TTY, testes de contrato); NODE_EXTRA_CA_CERTS (env de runtime Node) na allowlist do checker. 7 testes unit base-red (de features alheias à quota): - oauth-providers-config (#4664): teste alinhado ao provider codebuddy-cn do registry. - antigravity-model-aliases (#4636): maxOutputTokens esperado 32769→16384 (cap intencional). - provider-request-capture #4091 (#4861): exemplo do teste trocado de mcp__ (que #4861 isenta de cloak por causa dos 400s de assimetria de histórico) para um tool de terceiro cloakável — preserva o invariante de #4091 SEM reverter #4861. - combo-error-response: convertido de vitest p/ node:test (era coletado pelo glob node:test e crashava); api/** e server/** removidos do vitest.config (config morta). Build MDX (dast-smoke, #4679): - docs/ops/RELEASE_GREEN.md não tinha frontmatter `title` → fumadocs-mdx rejeitava no webpack compile ("invalid frontmatter: title expected string"), quebrando o next build (e o deploy). Frontmatter title adicionado (único doc do collection sem ele). 17/17 Fast Quality Gates + suíte unit completa (17737 testes, 0 fail) + vitest verdes localmente. Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * feat(quota): saturação proativa por headers de tokens (universal) [Fase 3 #2] (#4907) storeRateLimitHeaders só capturava os headers de REQUESTS (RPM/min), que não refletem a pressão de TOKENS. Agora também parseia os headers de tokens (em toda resposta, sucesso também) para throttle proativo antes do 429: - Anthropic: anthropic-ratelimit-tokens-{limit,remaining,reset} (+ input/output), RFC3339. - OpenAI: x-ratelimit-{limit,remaining,reset}-tokens, reset em duração (6m0s). saturation = 1 − remaining/limit; resetAt normalizado a epoch (parse de duração ReDoS-safe). getTokenHeaderSaturation por (provider, connectionId). fetchGeneric- Saturation passa a usar esse sinal (complementa o oauth/usage do #1, que segue primário p/ Claude). Fail-open, cache mantido, request-path inalterado. 16 testes novos + regressão (oauth/usage #1 8/8, signals 6/6) = 30/30; typecheck:core + eslint limpos. Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * feat(quota): estratégia de combo "headroom" — seleção por folga de cota [Fase 3 #4] (#4908) Nova estratégia de roteamento que escolhe a conexão com MAIS folga de plano: headroom = 1 − max(util_5h, util_7d) (técnica do dario), via getSaturation (melhorado p/ Claude no #1). Proativo em vez de só fill-first reativo. - Helper PURO headroomRanking.ts (computeHeadroom + rankByHeadroom; saturação injetada, não-mutante, tie-break estável, fail-open). - Orderer async em combo/quotaStrategies.ts (reusa a maquinaria reset-aware de expansão de conexões + concorrência limitada; seam injetável). - Registrada como "headroom" em routingStrategies (combo-only); fill-first segue default — nenhuma estratégia existente tocada. - baseline file-size combo.ts 3168->3180 (só +12L de dispatch; lógica fora do god-file). 16 testes novos + combo-strategies 15/15 = 31/31; typecheck:core + eslint + file-size limpos. Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * feat(quota): cap per-(key,model) — quota_allocation_model_caps [Fase 3 #7] (#4927) * feat(quota): cap per-(key,model) com tabela quota_allocation_model_caps [Fase 3 #7] Fecha o buraco onde uma API key pode drenar o pool inteiro consumindo um único modelo. Tabela nova: quota_allocation_model_caps(pool_id, api_key_id, model, cap_value, cap_unit) PK composta (pool_id, api_key_id, model). cap_unit alinhado ao QuotaUnit existente. Comportamento: keyA acima do cap para modelo M → bloqueada somente em M; ainda permitida em qualquer outro modelo no mesmo pool. Cap <= EPSILON → ignorado (seed). Consumo por-(key,model) usa bucket segregado no quota_consumption existente (poolId mangled ':model:<model>') com window fixa 'hourly'; nenhuma nova tabela ou método de store necessário. Módulo novo: src/lib/db/quotaModelCaps.ts (getModelCap/setModelCap/deleteModelCap/listModelCaps) enforce.ts ganha o pre-check em enforceQuotaShare + recording em recordConsumption. EnforceInput e RecordConsumptionInput ganham model?: string (backward-compatible). localDb.ts re-exporta os 4 helpers (Hard Rule #2). TDD: tests/unit/quota-per-key-model.test.ts — 4 cenários (bloqueia em M, permite em M2, sem cap → sem bloqueio, EPSILON → ignorado). Todos os gates de qualidade passam. * feat(quota): plumba model resolvido no hot path para ativar o per-(key,model) cap [Fase 3 #7] A tabela/enforce do commit anterior estavam INERTES: o hot path não passava `model` ao enforce nem ao record, então nenhum model-cap disparava em produção. Plumbagem (model resolvido = mesma var usada no log/roteamento, pós background-redirect/alias): - chatCore.ts: enforceQuotaShare ganha `model`; scheduleQuotaShareConsumption recebe `model`. - chatCore/quotaShareConsumption.ts: threade `model` no RecordConsumptionInput (non-streaming). - spendRecorder.ts: recordStreamingConsumption já recebia `model` — agora o coloca no RecordConsumptionInput (streaming accrue por-modelo). - embeddings.ts: enforce + record ganham `model`. Namespace do cap = id do modelo RESOLVIDO (o mesmo de modelForScope/pendingScope/getUnsupportedParams), não o requestedModel cru nem o finalModelToUpstream (sem prefixo de provider). Operador configura o cap contra esse id. `model || undefined` em todos os pontos: vazio/null → check pulado (fail-safe, zero latência — só um campo no objeto). Teste de integração novo (tests/unit/quota-per-key-model-hotpath.test.ts): prova end-to-end que N consumos via scheduleQuotaShareConsumption({model}) → enforceQuotaShare({model}) bloqueia, e que outro modelo no mesmo pool ainda passa; + guard de que enforce SEM model nunca dispara model-cap. --------- Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * feat(quota): session stickiness p/ integridade de prompt-cache [Fase 3 #5] (#4929) * feat(quota): session stickiness p/ integridade de prompt-cache [Fase 3 #5] Adiciona stickiness de sessão ao roteamento de combo: uma conversa multi-turno é roteada para a MESMA conexão enquanto ela permanecer saudável, evitando a perda do prompt-cache do provider (custo 5-10× sem stickiness, efeito conhecido no dario/clewdr). Implementação: - `open-sse/services/combo/sessionStickiness.ts` (novo, <800 linhas): mapa em memória (messageHash → connectionId) com TTL 15 min + cap 500 entradas; `applySessionStickiness` promove a conexão sticky ao índice 0 dos targets ordenados pelo strategy, guardado por `computeHeadroom > 0.15` (threshold); quando saturada (headroom ≤ 0.15), o binding é limpo e a seleção normal reage. Hash da sessão = SHA-256 dos primeiros chars da 1ª mensagem user → 16 hex chars. Seam de teste: `__setStickinessHeadroomFetcherForTests`. - `open-sse/services/combo.ts`: import + 2 pontos de integração (pré-eval-scores e pós-success), dentro do orçamento congelado de 3180 linhas. - `tests/unit/combo-session-stickiness.test.ts`: 19 testes node:test + assert/strict, todos via injeção de fetcher (zero rede/DB). Threshold 0.15: conexão a >85% de utilização está a um burst de rate-limit; o benefício de cache não compensa manter-se numa conexão degradada. Valor alinhado com a zona de soft-penalty do restante do engine de quota-share. * test(combo): isola combo-strategies da session stickiness (#5) selectedConnectionFor reusa o mesmo body, então o sticky map (#5) fixava a connection após a 1ª chamada e quebrava o round-robin tie-break do teste reset-aware. Limpa o sticky map no início da helper — a stickiness tem suíte própria (combo-session-stickiness). Sem enfraquecer asserts. --------- Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * feat(quota): buckets multi-janela por conexão (5h/7d/per-model) [Fase 3 #3] (#4928) Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * refactor(providers): decompõe catálogo providers.ts em módulos de dados (godfile sweep, #3501) (#4917) Integrado em release/v3.8.36 (godfile sweep providers.ts, #3501) * refactor(pricing): decompõe pricing.ts em shared-tiers + DEFAULT_PRICING particionado (godfile sweep, #3501) (#4918) Integrado em release/v3.8.36 (godfile sweep pricing.ts, #3501) * refactor(api): extrai camada-folha pura de validation.ts (URL/headers/transport) (#4921) Integrado em release/v3.8.36 (validation.ts split fatia 1 — leaf layer) * refactor(api): extrai validators web-cookie + Meta AI de validation.ts (#4922) Integrado em release/v3.8.36 (validation.ts split fatia 2 — web-cookie + Meta AI) * refactor(api): extrai validators enterprise-cloud + probe compartilhado de validation.ts (#4923) Integrado em release/v3.8.36 (validation.ts split fatia 3 — enterprise-cloud + probe) * refactor(api): extrai validators áudio/speech + misc apikey de validation.ts (#4930) Integrado em release/v3.8.36 (validation.ts split fatia 4 — áudio/speech + misc apikey) * feat(quota): estratégia dedicada de quota-share (DRR + P2C in-flight + gating per-model) [Fase 3 #9] (#4939) * feat(quota): estratégia dedicada de quota-share (DRR + P2C in-flight + gating per-model) [Fase 3 #9] Estratégia interna "quota-share" isolada num módulo dedicado — NÃO toca a seleção/ fair-share genérica (decisão do dono: não mexer no que já funciona). Os combos qtSd/ (quotaCombos.ts) passam de fill-first para essa strategy; combo.ts ganha só 1 branch de dispatch que delega 100% ao módulo (nenhum case existente alterado). - quotaShareStrategy.ts: gating per-model (isBucketSaturated do #3) + DRR (quantum proporcional ao weight) + P2C sobre carga in-flight. - quotaShareInflight.ts: contador in-flight com TTL/lease de 120s — fallback do decrement-on-abort sem precisar instrumentar o combo genérico. - "quota-share" registrada como strategy INTERNA (não exposta na UI). - testes de síntese (quota-combo-balancing, quota-multiprovider) alinhados: a strategy esperada dos combos qtSd/ passa de "fill-first" para "quota-share" (alinhamento ao novo comportamento intencional, não mascaramento — os 73 testes de qtSd/ seguem verdes). * test(quota-share): alinha 2 scope-guards ao godfile sweep (base-reds que bloqueavam o CI) Dois testes de "arquivo contém X" quebraram por decomposições de godfile que outras sessões mergearam no release DURANTE a validação de #9 — NÃO são regressão de #9 (que não toca validation/oauth). Alinhados ao novo layout, asserts preservados: - proxy-bypass-scope-guard #3226: bypassProxyPatch foi extraído de validation.ts para validation/headers.ts (split #4921–#4930) → o teste lê a camada de validação. - sse-error-passthrough #3324: a windsurf authHint foi extraída de providers.ts para providers/oauth.ts → o teste lê o novo local. --------- Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * refactor(api): extrai validators search + embedding/rerank de validation.ts (#4932) Integrated into release/v3.8.36 * refactor(api): extrai format-validators (OpenAI/Anthropic) de validation.ts (#4933) Integrated into release/v3.8.36 * refactor(db): extrai model-permission matching de db/apiKeys.ts (#4936) Integrated into release/v3.8.36 * refactor(db): extrai row-parsers + tipos compartilhados de db/apiKeys.ts (#4943) Integrated into release/v3.8.36 * refactor(db): extrai column-mapping (snake↔camel) de db/core.ts (#4947) Integrated into release/v3.8.36 * refactor(db): extrai schema-column reconciliation de db/core.ts (#4948) Integrated into release/v3.8.36 * refactor(sse): extrai scalar/format helpers de services/usage.ts (#4949) Integrated into release/v3.8.36 * refactor(sse): extrai quota-core (UsageQuota + builders) de services/usage.ts (#4950) Integrated into release/v3.8.36 * fix(translator): regroup parallel tool results adjacent to their assistant (#4714) (#4882) Integrated into release/v3.8.36 (fixes #4714) * fix(qoder): exchange PAT for jt-* job token before Cosy chat (#4683) (#4884) Integrated into release/v3.8.36 (fixes #4683) * refactor(sse): dedup fallback tool_call id helper (#4736) Integrated into release/v3.8.36 * refactor(open-sse): extract safeParseJSON util, dedup tryParseJSON (#4735) Integrated into release/v3.8.36 * fix(compression): eliminate ReDoS in math_inline preservation pattern (#4795) (#4838) Integrated into release/v3.8.36 (fixes #4795) * fix(combo): fetch models dynamically from custom provider endpoints (#4860) Integrated into release/v3.8.36 * feat(providers): update volcengine-ark model list with DeepSeek V4 (#4905) Integrated into release/v3.8.36 * fix(translator): provider thinking compatibility (DeepSeek/Gemini) (#4946) Integrated into release/v3.8.36 * feat(combo): task-aware routing strategy (#4945) Integrated into release/v3.8.36 * refactor(sse): extrai a família MiniMax de services/usage.ts (#4952) Integrated into release/v3.8.36 * refactor(sse): extrai a família GLM de services/usage.ts (#4953) Integrated into release/v3.8.36 * refactor(sse): extrai a família Antigravity de services/usage.ts (#4956) Integrated into release/v3.8.36 * fix(dashboard): show custom provider given-name instead of internal id across dashboard pages (#4603) (#4960) Integrated into release/v3.8.36 (fixes #4603) * fix(api): evict stale in-memory rate-limit windows to stop slow heap leak (#4041) (#4957) Integrated into release/v3.8.36 (fixes #4041) * fix(api): parse /v1/responses body once instead of 3-4x on the hot path (#4041) (#4958) Integrated into release/v3.8.36 (fixes #4041) * fix(translator): preserve legitimate empty-string tool arguments in openai-to-claude streaming (#4951) (#4959) Integrated into release/v3.8.36 (fixes #4951) * chore(quality): reconcile file-size baseline for #4960 provider-display-name (#4961) Integrated into release/v3.8.36 * fix(dashboard): restore home provider-topology card hidden by #4596 default (#4963) Integrated into release/v3.8.36 — restores home topology card (#4596 regression) * fix(build): drop @omniroute/open-sse from optimizePackageImports (build OOM) (#4968) Integrated into release/v3.8.36 — fixes build OOM (optimizePackageImports open-sse) * fix(quota): migração 107 ativa estratégia quota-share nos combos qtSd/ existentes [Fase 3 #9] (#4962) Integrated into release/v3.8.36 * feat(quota): respeita max_concurrent por conexão no roteamento (#4965) Integrated into release/v3.8.36 * feat(quota): combo quota-share espera cooldown curto e re-despacha (Variante A) (#4967) Integrated into release/v3.8.36 * fix(quality): resolve base-reds da release — db-rules allowlist + task-aware router precedence (#4973) Dois base-reds pré-existentes que reprovavam o CI da release v3.8.36 (Fast Quality Gates + Unit Tests fast-path), independentes de qualquer feature em voo: 1. check:db-rules / allowlist: os módulos db-internal caseMapping (#4947) e schemaColumns (#4948), extraídos de db/core.ts e importados só por ele, não estavam em INTENTIONALLY_INTERNAL. Registrados na allowlist (correção canônica — são internos legítimos, não re-exportados pelo localDb). 2. auto-strategy honra LKGP/cost (combo-routing-engine.test.ts, 2 testes): o task-aware reordering (#4945, reorderByTaskWeight) roda para strategy "auto" e era aplicado DEPOIS do router explícito (selectWithStrategy: lkgp/cost), sobrescrevendo o orderedTargets[0] que o operador escolheu. Instrumentação provou: post-filter [0]=claude (LKGP) → post-task [0]=gpt-oss. Correção: quando o auto usa router explícito, preserva o [0] dele e deixa o task-aware refinar só a cauda de fallback. gpt-oss-120b PERMANECE tool-capable (não é mudança de catálogo; o model-capabilities-registry test segue verde). Validado: 121 testes (combo-routing-engine + combo-task-aware + registry) verdes, red-check confirmado, db-rules/file-size/typecheck/lint/prettier OK. Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * feat(quota): serializa concorrência por conexão no caminho quota-share (FASE 2.1) (#4970) O gating de quota-share em selectQuotaShareTarget é fail-open: uma conexão at-cap só é despriorizada, nunca bloqueada. Com 1 conexão por conta de assinatura (caso comum), chamadas concorrentes ainda floodam a conta (→ 429 + cooldown) — provado live na .15: 3 chamadas concorrentes com max_concurrent=1 despacharam todas em 94ms. Adiciona um semáforo POR CONEXÃO em torno do dispatch quota-share: chamadas excedentes esperam na fila em vez de floodar (key qsconn:<connectionId>, cap = max_concurrent da conexão). Fail-open em fila saturada/timeout para nunca piorar disponibilidade. Gated por strategy===quota-share + kill-switch resilienceSettings.quotaShareConcurrencyLimit (default on; UI no ResilienceTab). Lógica extraível isolada no leaf puro combo/quotaShareConcurrency.ts (unit-testado: estabilidade da key, no-op sem cap, serialização real, fail-open). Settings + schema + UI espelham comboCooldownWait. Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * docs(resilience): document Quota-Share Concurrency Control (max_concurrent + serialization + cooldown-wait) (#4980) Documents the v3.8.36 quota-share concurrency layers in RESILIENCE_GUIDE.md: per-connection max_concurrent cap, the quota-share request serialization semaphore (FASE 2.1, qsconn:<connectionId>, fail-open, kill-switch), and the combo cooldown-aware retry — so operators know how to cap a subscription account's concurrency and why the routing gate alone cannot contain a single-connection flood. Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * fix(dashboard): proxy-pool success gating, sync timestamp, opt-in Redis (#4878) (#4988) Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * fix(sse): fail over on 400 responses carrying rate-limit text (#4976) (#4986) * fix(sse): fail over on 400 responses carrying rate-limit text (#4976) * chore(quality): rebaseline accountFallback.ts file-size for #4976 fix --------- Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * fix(compression): stop RTK over-truncating file-read tool results (#4559) (#4987) * fix(compression): stop RTK over-truncating file-read tool results (#4559) * chore(quality): trim #4559 comment to keep rtk/index.ts within size cap --------- Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * fix(sse): honor per-account proxies and fingerprint rotation in opencode executor (#4954) (#4989) * fix(sse): honor per-account proxies and fingerprint rotation in opencode executor (#4954) * chore(quality): rebaseline auth.ts file-size for #4954 (+39: synthetic no-auth providerSpecificData hydration of fingerprints/accountProxies; irreducible credential-path wiring, covered by opencode-proxy-rotation-4954.test.ts + 159 auth/noauth regression) --------- Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * fix(sse): soft-penalize exhausted providers in auto-combo scoring (#4540) (#4990) * fix(sse): soft-penalize exhausted providers in auto-combo scoring (#4540) * chore(quality): document STATUS_SOFT_DEPRIORITIZE_FACTOR + rebaseline combo.ts for #4540 --------- Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * fix(dashboard): switch to visible filter after auto-hiding failed models in test-all (#4887) (#4991) * fix(dashboard): switch to visible filter after auto-hiding failed models in OAuth provider test-all (#4887) * test(dashboard): move #4887 test into tests/unit/ui so a CI runner collects it --------- Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * fix(pollinations): only enable jsonMode when JSON output is requested (#3981) (#5009) Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * fix(antigravity): default safetySettings to all-OFF for parity with native Gemini paths (#5003) (#5008) * fix(antigravity): default safetySettings to all-OFF for parity with native Gemini paths (#5003) * docs(changelog): restore #3981 pollinations entry eaten by merge --------- Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * fix(chatgpt-web): map advertised gpt-5.5/5.4-pro/5.2-pro slugs to prevent silent model substitution (#4665) (#5010) * fix(chatgpt-web): map advertised gpt-5.5/5.4-pro/5.2-pro slugs to prevent silent model substitution (#4665) MODEL_MAP was missing the advertised catalog ids gpt-5.5, gpt-5.5-pro, gpt-5.4-pro and gpt-5.2-pro, so MODEL_MAP[model] ?? model sent the dot-form id verbatim to the ChatGPT backend-api, which silently rejected it and served the default Plus model. Map each to its dash-form slug. gpt-4-5 is already dash-form and falls through correctly, so it is intentionally left unmapped. Extends the executor MODEL_MAP test with the four ids and adds a drift guard asserting every advertised dot-form catalog id reaches the backend in dash-form (never verbatim), guarding future catalog<->map drift. file-size: tests/unit/chatgpt-web.test.ts frozen baseline 2809->2855 (+46) for the added test cases and drift-guard test; executor source unchanged in baseline. * docs(changelog): restore #3981/#5003 entries eaten by merge --------- Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * feat(combos): add editable per-combo description field persisted via /api/combos (#5005) (#5011) * feat(combos): add editable per-combo description field persisted via /api/combos (#5005) * docs(changelog): restore #3981/#5003/#4665 entries eaten by merge --------- Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> * Fix Ollama Cloud max reasoning effort (#4993) Integrated into release/v3.8.36 * fix(copilot): replace execSync with execFile to prevent command injection (#5024) Integrated into release/v3.8.36 * fix(plugin): auth.json dual-key fallback for auto-prefix migration (#5027) Integrated into release/v3.8.36 * feat(endpoint): per-endpoint custom system prompt injection (#5022) Integrated into release/v3.8.36 * fix(headroom): translate openai-responses input through OpenAI for compression (#5023) Integrated into release/v3.8.36 * docs(changelog): add entries for #4993, #5024, #5027 (release notes credit) * fix(api): stop /api/system/env/repair 500 on packaged install (#5006) (#5028) * fix(api): stop /api/system/env/repair 500 on packaged install — lazy createRequire in sync-env.mjs (#5006) scripts/dev/sync-env.mjs ran createRequire(import.meta.url) at module top-level. When webpack bundles it into the standalone env-repair route, import.meta.url is frozen to the build-machine path (file:///home/runner/...) and createRequire throws during module evaluation, so the whole route module fails to load and every GET returns HTTP 500 — breaking the onboarding wizard on packaged/global installs. - Move createRequire into the guarded better-sqlite3 block (only place that needs it); a bad import.meta.url now returns the safe default. - resolveRootDir() falls back to process.cwd() when fileURLToPath throws. - route.ts passes an explicit rootDir (process.cwd()) so the helper never derives the root from the frozen import.meta.url, matching the .env target used by createEnvBackup(). - Regression guard: assert sync-env.mjs has no top-level createRequire + getEnvSyncPlan(oauth) works with explicit rootDir without throwing. * docs(changelog): restore #4993/#5023/#5024/#5027 + custom-system-prompt/headroom entries eaten by release merge * chore(quality): rebaseline 3 inherited base-reds from release merge Files NOT touched by this PR — grew on release/v3.8.36 via --admin merges and inherited here through 'git merge origin/release': - open-sse/executors/base.ts 1414->1416 (#4993 Ollama Cloud max-effort) - src/lib/db/settings.ts 1149->1151 (#5023 custom system prompt) - src/app/(dashboard)/.../endpoint/EndpointPageClient.tsx 2570->2612 (custom system prompt UI) * chore(release): finalize v3.8.36 CHANGELOG + docs (2026-06-25) --------- Co-authored-by: Diego Rodrigues de Sa e Souza <souzamiriamrodrigues790@gmail.com> Co-authored-by: KooshaPari <42529354+KooshaPari@users.noreply.github.com> Co-authored-by: Makcim Ivanov <makcimbx@gmail.com> Co-authored-by: Chewji <126886556+Chewji9875@users.noreply.github.com> Co-authored-by: Anton <39598727+NomenAK@users.noreply.github.com> Co-authored-by: Demiurge The Single <megamen932@gmail.com> Co-authored-by: Randi <55005611+rdself@users.noreply.github.com> Co-authored-by: Éder Costa <eder.almeida.costa@gmail.com> Co-authored-by: Jefferson Felizardo <jeffer1312@gmail.com> Co-authored-by: Arthur Bodera <abodera@gmail.com> Co-authored-by: Hamsa_M <116961508+hamsa0x7@users.noreply.github.com> Co-authored-by: Hernan Javier Ardila Sanchez <hjasgr@gmail.com>
870 lines
30 KiB
TypeScript
870 lines
30 KiB
TypeScript
import test from "node:test";
|
|
import assert from "node:assert/strict";
|
|
|
|
const schemaCoercion = await import("../../open-sse/translator/helpers/schemaCoercion.ts");
|
|
const openaiHelper = await import("../../open-sse/translator/helpers/openaiHelper.ts");
|
|
const claudeHelper = await import("../../open-sse/translator/helpers/claudeHelper.ts");
|
|
const geminiHelper = await import("../../open-sse/translator/helpers/geminiHelper.ts");
|
|
const toolCallHelper = await import("../../open-sse/translator/helpers/toolCallHelper.ts");
|
|
const { FORMATS } = await import("../../open-sse/translator/formats.ts");
|
|
const { translateRequest } = await import("../../open-sse/translator/index.ts");
|
|
const { cacheReasoningByKey, clearReasoningCacheAll, getReasoningCacheServiceStats } =
|
|
await import("../../open-sse/services/reasoningCache.ts");
|
|
const { clearModelsDevCapabilities, saveModelsDevCapabilities } =
|
|
await import("../../src/lib/modelsDevSync.ts");
|
|
|
|
function buildCapability(overrides = {}) {
|
|
return {
|
|
tool_call: null,
|
|
reasoning: null,
|
|
attachment: null,
|
|
structured_output: null,
|
|
temperature: null,
|
|
modalities_input: "[]",
|
|
modalities_output: "[]",
|
|
knowledge_cutoff: null,
|
|
release_date: null,
|
|
last_updated: null,
|
|
status: null,
|
|
family: null,
|
|
open_weights: null,
|
|
limit_context: null,
|
|
limit_input: null,
|
|
limit_output: null,
|
|
interleaved_field: null,
|
|
...overrides,
|
|
};
|
|
}
|
|
|
|
const originalMathRandom = Math.random;
|
|
|
|
test.afterEach(() => {
|
|
Math.random = originalMathRandom;
|
|
});
|
|
|
|
test("schemaCoercion recursively coerces schema numeric fields across object variants", () => {
|
|
const result = schemaCoercion.coerceSchemaNumericFields({
|
|
minimum: "1",
|
|
maxItems: "5",
|
|
properties: {
|
|
nested: {
|
|
minLength: "2",
|
|
items: { maximum: "7" },
|
|
},
|
|
},
|
|
patternProperties: {
|
|
"^x-": { minProperties: "1" },
|
|
},
|
|
definitions: {
|
|
one: { exclusiveMaximum: "9" },
|
|
},
|
|
$defs: {
|
|
two: { minItems: "3" },
|
|
},
|
|
dependentSchemas: {
|
|
dep: { maxProperties: "4" },
|
|
},
|
|
additionalProperties: { maximum: "8" },
|
|
unevaluatedProperties: { minimum: "0" },
|
|
prefixItems: [{ minimum: "11" }],
|
|
anyOf: [{ maximum: "12" }],
|
|
oneOf: [{ minimum: "13" }],
|
|
allOf: [{ maxLength: "14" }],
|
|
not: { minimum: "15" },
|
|
if: { minimum: "16" },
|
|
then: { maximum: "17" },
|
|
else: { minItems: "18" },
|
|
});
|
|
|
|
assert.equal((result as any).minimum, 1);
|
|
(assert as any).equal((result as any).maxItems, 5);
|
|
(assert as any).equal((result as any).properties.nested.minLength, 2);
|
|
assert.equal((result as any).properties.nested.items.maximum, 7);
|
|
assert.equal((result as any).patternProperties["^x-"].minProperties, 1);
|
|
assert.equal((result as any).definitions.one.exclusiveMaximum, 9);
|
|
assert.equal((result as any).$defs.two.minItems, 3);
|
|
assert.equal((result as any).dependentSchemas.dep.maxProperties, 4);
|
|
assert.equal((result as any).additionalProperties.maximum, 8);
|
|
assert.equal((result as any).unevaluatedProperties.minimum, 0);
|
|
assert.equal((result as any).prefixItems[0].minimum, 11);
|
|
assert.equal((result as any).anyOf[0].maximum, 12);
|
|
assert.equal((result as any).oneOf[0].minimum, 13);
|
|
assert.equal((result as any).allOf[0].maxLength, 14);
|
|
assert.equal((result as any).not.minimum, 15);
|
|
assert.equal((result as any).if.minimum, 16);
|
|
assert.equal((result as any).then.maximum, 17);
|
|
assert.equal((result as any).else.minItems, 18);
|
|
|
|
assert.equal(schemaCoercion.coerceSchemaNumericFields("unchanged"), "unchanged");
|
|
assert.deepEqual(schemaCoercion.coerceSchemaNumericFields(["2", { minimum: "3" }]), [
|
|
"2",
|
|
{ minimum: 3 },
|
|
]);
|
|
});
|
|
|
|
test("schemaCoercion sanitizes descriptions, tool schemas, tool ids and deepseek reasoning placeholders", () => {
|
|
const sanitizedOpenAI = schemaCoercion.sanitizeToolDescription({
|
|
type: "function",
|
|
function: { name: "weather", description: 42 },
|
|
});
|
|
(assert as any).equal((sanitizedOpenAI as any).function.description, "42");
|
|
|
|
const sanitizedClaude = schemaCoercion.sanitizeToolDescription({
|
|
name: "weather",
|
|
description: null,
|
|
});
|
|
assert.equal((sanitizedClaude as any).description, "");
|
|
|
|
const sanitizedGemini = schemaCoercion.sanitizeToolDescription({
|
|
functionDeclarations: [{ name: "one", description: 12 }, { name: "two" }],
|
|
});
|
|
assert.equal((sanitizedGemini as any).functionDeclarations[0].description, "12");
|
|
assert.equal((sanitizedGemini as any).functionDeclarations[1].name, "two");
|
|
assert.equal(schemaCoercion.sanitizeToolDescription("plain"), "plain");
|
|
|
|
const coercedTools = schemaCoercion.coerceToolSchemas([
|
|
{
|
|
type: "function",
|
|
function: { parameters: { minimum: "4" } },
|
|
},
|
|
{
|
|
name: "claude-style",
|
|
input_schema: { minItems: "2" },
|
|
},
|
|
{
|
|
parameters: { maximum: "9" },
|
|
},
|
|
{
|
|
functionDeclarations: [{ parameters: { minLength: "1" } }],
|
|
},
|
|
"untouched",
|
|
]);
|
|
assert.equal(coercedTools[0].function.parameters.minimum, 4);
|
|
assert.equal(coercedTools[1].input_schema.minItems, 2);
|
|
assert.equal(coercedTools[2].parameters.maximum, 9);
|
|
assert.equal(coercedTools[3].functionDeclarations[0].parameters.minLength, 1);
|
|
assert.equal(coercedTools[4], "untouched");
|
|
assert.equal(schemaCoercion.coerceToolSchemas("not-array"), "not-array");
|
|
|
|
const descriptionList = schemaCoercion.sanitizeToolDescriptions([{ description: 7 }, "raw"]);
|
|
assert.equal(descriptionList[0].description, "7");
|
|
assert.equal(descriptionList[1], "raw");
|
|
assert.equal(schemaCoercion.sanitizeToolDescriptions("raw"), "raw");
|
|
|
|
assert.equal(schemaCoercion.sanitizeToolId("call.abc:123"), "call_abc_123");
|
|
assert.match(schemaCoercion.sanitizeToolId(""), /^tool_[a-z0-9_]+$/);
|
|
assert.match(schemaCoercion.sanitizeToolId(undefined), /^tool_[a-z0-9_]+$/);
|
|
|
|
const injected = schemaCoercion.injectEmptyReasoningContentForToolCalls(
|
|
[
|
|
{ role: "assistant", tool_calls: [{ id: "call_1" }] },
|
|
{ role: "assistant", tool_calls: [{ id: "call_2" }], reasoning_content: "keep" },
|
|
{ role: "user", tool_calls: [{ id: "call_3" }] },
|
|
],
|
|
"deepseek",
|
|
"deepseek-v4-flash"
|
|
);
|
|
assert.equal(injected[0].reasoning_content, "");
|
|
assert.equal(injected[1].reasoning_content, "keep");
|
|
assert.equal(injected[2].reasoning_content, undefined);
|
|
assert.equal(
|
|
schemaCoercion.injectEmptyReasoningContentForToolCalls(
|
|
[{ role: "assistant" }],
|
|
"openai",
|
|
"gpt-4o"
|
|
)[0].reasoning_content,
|
|
undefined
|
|
);
|
|
});
|
|
|
|
test("openaiHelper filters content, normalizes tools and removes OpenAI-incompatible fields", () => {
|
|
const body = {
|
|
messages: [
|
|
{ role: "tool", content: "" },
|
|
{ role: "assistant", tool_calls: [{ id: "call_1" }], content: "" },
|
|
{
|
|
role: "assistant",
|
|
content: [
|
|
{ type: "thinking", thinking: "plan first" },
|
|
{ type: "redacted_thinking", text: "skip" },
|
|
{ type: "text", text: "visible text" },
|
|
{ type: "image_url", image_url: { url: "https://example.com/a.png" }, signature: "x" },
|
|
{ type: "tool_use", id: "call_1" },
|
|
{ type: "tool_result", tool_use_id: "call_1", text: "done", cache_control: "drop" },
|
|
],
|
|
},
|
|
{ role: "user", content: [{ type: "text", text: " " }] },
|
|
{ role: "assistant", content: [{ type: "tool_result", tool_use_id: "call_2" }] },
|
|
],
|
|
tools: [
|
|
{
|
|
name: "claude-tool",
|
|
description: "Claude style",
|
|
input_schema: { type: "object" },
|
|
},
|
|
{
|
|
functionDeclarations: [
|
|
{ name: "gemini-tool", description: "Gemini style", parameters: { type: "object" } },
|
|
],
|
|
},
|
|
{
|
|
type: "function",
|
|
function: { name: "openai-tool", parameters: { type: "object" } },
|
|
},
|
|
],
|
|
tool_choice: { type: "tool", name: "forced_tool" },
|
|
metadata: { remove: true },
|
|
anthropic_version: "2023-06-01",
|
|
};
|
|
|
|
const result = openaiHelper.filterToOpenAIFormat(body);
|
|
|
|
assert.equal(result.messages.length, 3);
|
|
assert.equal(result.messages[2].reasoning_content, "plan first");
|
|
assert.deepEqual(result.messages[2].content, [
|
|
{ type: "text", text: "visible text" },
|
|
{ type: "image_url", image_url: { url: "https://example.com/a.png" } },
|
|
{ type: "text", text: "[Tool Result: call_1]\ndone" },
|
|
]);
|
|
assert.equal(result.tools.length, 3);
|
|
assert.equal(result.tools[0].function.name, "claude-tool");
|
|
assert.equal(result.tools[1].function.name, "gemini-tool");
|
|
assert.equal(result.tools[2].function.name, "openai-tool");
|
|
assert.deepEqual(result.tool_choice, { type: "function", function: { name: "forced_tool" } });
|
|
assert.equal("metadata" in result, false);
|
|
assert.equal("anthropic_version" in result, false);
|
|
});
|
|
|
|
test("openaiHelper keeps unmatched tool choices and deletes empty tools arrays", () => {
|
|
const autoChoice = openaiHelper.filterToOpenAIFormat({
|
|
messages: [{ role: "assistant", content: "" }],
|
|
tools: [],
|
|
tool_choice: { type: "auto" },
|
|
});
|
|
assert.equal(autoChoice.tool_choice, "auto");
|
|
assert.equal("tools" in autoChoice, false);
|
|
|
|
const requiredChoice = openaiHelper.filterToOpenAIFormat({
|
|
messages: [{ role: "assistant", content: "" }],
|
|
tool_choice: { type: "any" },
|
|
});
|
|
assert.equal(requiredChoice.tool_choice, "required");
|
|
|
|
const untouched = { metadata: { keep: false } };
|
|
assert.deepEqual(openaiHelper.filterToOpenAIFormat(untouched), {
|
|
metadata: { keep: false },
|
|
});
|
|
});
|
|
|
|
test("claudeHelper validates content, ordering and request preparation branches", () => {
|
|
assert.equal(claudeHelper.hasValidContent({ content: " hello " }), true);
|
|
assert.equal(claudeHelper.hasValidContent({ content: [{ type: "tool_use", id: "call" }] }), true);
|
|
assert.equal(claudeHelper.hasValidContent({ content: [{ type: "text", text: " " }] }), false);
|
|
|
|
assert.deepEqual(claudeHelper.fixToolUseOrdering([{ role: "user", content: "single" }]), [
|
|
{ role: "user", content: "single" },
|
|
]);
|
|
|
|
const reordered = claudeHelper.fixToolUseOrdering([
|
|
{
|
|
role: "assistant",
|
|
content: [
|
|
{ type: "text", text: "before" },
|
|
{ type: "tool_use", id: "call_1", name: "lookup", input: {} },
|
|
{ type: "text", text: "after" },
|
|
],
|
|
},
|
|
{ role: "assistant", content: [{ type: "tool_result", tool_use_id: "call_1", content: [] }] },
|
|
]);
|
|
assert.deepEqual(reordered[0].content, [
|
|
{ type: "tool_result", tool_use_id: "call_1", content: [] },
|
|
{ type: "text", text: "before" },
|
|
{ type: "tool_use", id: "call_1", name: "lookup", input: {} },
|
|
]);
|
|
|
|
// splitMisplacedToolResults: a tool_result whose tool_use_id was already
|
|
// emitted by an earlier assistant turn is moved into the preceding user
|
|
// message. The trailing tool_use survives on the assistant side. (#2815)
|
|
const split = claudeHelper.splitMisplacedToolResults([
|
|
{ role: "user", content: [{ type: "text", text: "q" }] },
|
|
{ role: "assistant", content: [{ type: "tool_use", id: "call_x", name: "Read", input: {} }] },
|
|
{
|
|
role: "assistant",
|
|
content: [
|
|
{ type: "tool_result", tool_use_id: "call_x", content: "ok" },
|
|
{ type: "tool_use", id: "call_y", name: "Read", input: {} },
|
|
],
|
|
},
|
|
]);
|
|
assert.deepEqual(split, [
|
|
{ role: "user", content: [{ type: "text", text: "q" }] },
|
|
{ role: "assistant", content: [{ type: "tool_use", id: "call_x", name: "Read", input: {} }] },
|
|
{ role: "user", content: [{ type: "tool_result", tool_use_id: "call_x", content: "ok" }] },
|
|
{ role: "assistant", content: [{ type: "tool_use", id: "call_y", name: "Read", input: {} }] },
|
|
]);
|
|
|
|
// tool_result whose id has not been seen earlier is dropped — moving it
|
|
// would just shift the 400 to "unexpected tool_use_id".
|
|
const droppedOrphan = claudeHelper.splitMisplacedToolResults([
|
|
{ role: "user", content: [{ type: "text", text: "q" }] },
|
|
{
|
|
role: "assistant",
|
|
content: [
|
|
{ type: "tool_result", tool_use_id: "self-ref", content: "Skill not found" },
|
|
{ type: "tool_use", id: "self-ref", name: "Read", input: {} },
|
|
],
|
|
},
|
|
]);
|
|
assert.deepEqual(droppedOrphan, [
|
|
{ role: "user", content: [{ type: "text", text: "q" }] },
|
|
{ role: "assistant", content: [{ type: "tool_use", id: "self-ref", name: "Read", input: {} }] },
|
|
]);
|
|
|
|
const prepared = claudeHelper.prepareClaudeRequest(
|
|
{
|
|
system: [
|
|
{ type: "text", text: "one", cache_control: { type: "ephemeral" } },
|
|
{ type: "text", text: "two", cache_control: { type: "ephemeral" } },
|
|
],
|
|
messages: [
|
|
{ role: "user", content: [{ type: "text", text: "first question" }] },
|
|
{ role: "assistant", content: "first answer" },
|
|
{ role: "user", content: [{ type: "text", text: "follow up" }] },
|
|
{
|
|
role: "assistant",
|
|
content: [
|
|
{ type: "text", text: "before tool" },
|
|
{ type: "tool_use", id: "call_1", name: "lookup", input: {} },
|
|
{ type: "text", text: "drop after" },
|
|
{ type: "redacted_thinking", text: "keep" },
|
|
],
|
|
},
|
|
{
|
|
role: "user",
|
|
content: [
|
|
{ type: "tool_result", tool_use_id: "call_1", content: "ok" },
|
|
{ type: "tool_result", content: "drop missing id" },
|
|
],
|
|
},
|
|
{
|
|
role: "assistant",
|
|
content: [
|
|
{ type: "thinking", thinking: "old", signature: "replace" },
|
|
{ type: "tool_use", id: "call_2", name: " ", input: {} },
|
|
{ type: "text", text: "" },
|
|
],
|
|
},
|
|
],
|
|
thinking: { type: "enabled" },
|
|
tools: [
|
|
{ name: "", description: "drop me" },
|
|
{ name: "deferred", defer_loading: true, cache_control: { type: "ephemeral" } },
|
|
{ name: "kept-tool", cache_control: { type: "ephemeral" } },
|
|
],
|
|
},
|
|
"claude",
|
|
false
|
|
);
|
|
|
|
assert.deepEqual(prepared.system[0], { type: "text", text: "one" });
|
|
assert.deepEqual(prepared.system[1], {
|
|
type: "text",
|
|
text: "two",
|
|
cache_control: { type: "ephemeral", ttl: "1h" },
|
|
});
|
|
assert.equal(prepared.messages.length, 6);
|
|
assert.equal(prepared.messages[2].content.at(-1).cache_control.type, "ephemeral");
|
|
assert.equal(prepared.messages[4].content[0].type, "tool_result");
|
|
// messages[5] is the latest (and last) assistant message; Anthropic enforces
|
|
// that its thinking blocks must remain verbatim — not rewritten to
|
|
// redacted_thinking. The guard in prepareClaudeRequest preserves them.
|
|
assert.deepEqual(
|
|
prepared.messages[5].content.map((block) => block.type),
|
|
["thinking", "text"]
|
|
);
|
|
assert.equal(prepared.messages[5].content[0].thinking, "old", "thinking text preserved verbatim");
|
|
assert.equal(
|
|
prepared.messages[5].content[0].signature,
|
|
"replace",
|
|
"signature preserved verbatim"
|
|
);
|
|
assert.equal(
|
|
prepared.messages[5].content[0].data,
|
|
undefined,
|
|
"no data field on verbatim thinking"
|
|
);
|
|
assert.equal(prepared.tools.length, 2);
|
|
assert.equal(prepared.tools[0].cache_control, undefined);
|
|
assert.deepEqual(prepared.tools[1].cache_control, { type: "ephemeral", ttl: "1h" });
|
|
|
|
const preserved = claudeHelper.prepareClaudeRequest(
|
|
{
|
|
messages: [
|
|
{
|
|
role: "assistant",
|
|
content: [{ type: "text", text: "keep cache", cache_control: { type: "ephemeral" } }],
|
|
},
|
|
],
|
|
tools: [{ name: "kept", cache_control: { type: "ephemeral" } }],
|
|
},
|
|
"openai",
|
|
true
|
|
);
|
|
assert.deepEqual(preserved.messages[0].content[0].cache_control, { type: "ephemeral" });
|
|
assert.deepEqual(preserved.tools[0].cache_control, { type: "ephemeral" });
|
|
});
|
|
|
|
test("geminiHelper converts content, safely parses JSON and cleans complex schemas", () => {
|
|
assert.deepEqual(geminiHelper.convertOpenAIContentToParts("hello"), [{ text: "hello" }]);
|
|
assert.deepEqual(
|
|
geminiHelper.convertOpenAIContentToParts([
|
|
{ type: "text", text: "hello" },
|
|
{ type: "image_url", image_url: { url: "data:image/png;base64,abc" } },
|
|
{ type: "file_url", file_url: { url: "not-a-data-url" } },
|
|
]),
|
|
[{ text: "hello" }, { inlineData: { mimeType: "image/png", data: "abc" } }]
|
|
);
|
|
|
|
assert.equal(
|
|
geminiHelper.extractTextContent([
|
|
{ type: "text", text: "A" },
|
|
{ type: "image_url", image_url: { url: "https://example.com" } },
|
|
{ type: "text", text: "B" },
|
|
]),
|
|
"AB"
|
|
);
|
|
assert.equal(geminiHelper.extractTextContent({ no: "text" }), "");
|
|
assert.deepEqual(geminiHelper.tryParseJSON('{"ok":true}'), { ok: true });
|
|
assert.equal(geminiHelper.tryParseJSON("{broken"), null);
|
|
assert.equal(geminiHelper.tryParseJSON(42), 42);
|
|
assert.match(geminiHelper.generateRequestId(), /^agent-/);
|
|
assert.match(geminiHelper.generateSessionId(), /^-/);
|
|
|
|
const schema = {
|
|
type: ["null", "object"],
|
|
properties: {},
|
|
required: ["missing"],
|
|
anyOf: [{ type: "null" }, { type: "string", enum: [1, 2] }],
|
|
oneOf: [{ type: "null" }, { type: "array", items: { type: "integer", enum: [1, 2] } }],
|
|
allOf: [
|
|
{ properties: { a: { type: "string", minLength: 2 } }, required: ["a"] },
|
|
{ properties: { b: { const: "fixed" } }, required: ["b"] },
|
|
],
|
|
additionalProperties: false,
|
|
patternProperties: { "^x-": { type: "number" } },
|
|
if: { type: "string" },
|
|
then: { type: "string" },
|
|
else: { type: "string" },
|
|
default: "remove",
|
|
examples: ["remove"],
|
|
};
|
|
|
|
const cleaned = geminiHelper.cleanJSONSchemaForAntigravity(schema);
|
|
assert.equal(cleaned.type, "array");
|
|
assert.deepEqual(cleaned.required.sort(), ["a", "b"]);
|
|
assert.equal(cleaned.properties.a.minLength, undefined);
|
|
assert.deepEqual(cleaned.properties.b.enum, ["fixed"]);
|
|
assert.deepEqual(cleaned.enum, ["1", "2"]);
|
|
assert.equal(cleaned.items.type, "integer");
|
|
assert.equal(cleaned.additionalProperties, undefined);
|
|
assert.equal(cleaned.patternProperties, undefined);
|
|
assert.equal(cleaned.if, undefined);
|
|
assert.equal(cleaned.default, undefined);
|
|
assert.equal(cleaned.examples, undefined);
|
|
|
|
const placeholder = geminiHelper.cleanJSONSchemaForAntigravity({
|
|
type: "object",
|
|
properties: {},
|
|
});
|
|
assert.deepEqual(placeholder.required, ["reason"]);
|
|
assert.equal(placeholder.properties.reason.type, "string");
|
|
});
|
|
|
|
test("toolCallHelper normalizes ids, links tool responses and inserts missing tool results", () => {
|
|
let randomCalls = 0;
|
|
Math.random = () => ((randomCalls++ % 50) + 1) / 100;
|
|
|
|
const body = toolCallHelper.ensureToolCallIds(
|
|
{
|
|
messages: [
|
|
{
|
|
role: "assistant",
|
|
tool_calls: [
|
|
{ function: { name: "first", arguments: { city: "Tokyo" } } },
|
|
{ id: " ", function: { name: "second", arguments: "{}" } },
|
|
],
|
|
},
|
|
{ role: "tool", content: "first result" },
|
|
{ role: "tool", content: "second result" },
|
|
],
|
|
},
|
|
{ use9CharId: true }
|
|
);
|
|
|
|
assert.equal(body.messages[0].tool_calls[0].type, "function");
|
|
assert.equal(typeof body.messages[0].tool_calls[0].function.arguments, "string");
|
|
assert.match(body.messages[0].tool_calls[0].id, /^[a-zA-Z0-9]{9}$/);
|
|
assert.match(body.messages[1].tool_call_id, /^[a-zA-Z0-9]{9}$/);
|
|
assert.match(body.messages[2].tool_call_id, /^[a-zA-Z0-9]{9}$/);
|
|
|
|
const missingResponseFixed = toolCallHelper.fixMissingToolResponses({
|
|
messages: [
|
|
{
|
|
role: "assistant",
|
|
tool_calls: [{ id: "call_a", function: { name: "lookup", arguments: "{}" } }],
|
|
},
|
|
{ role: "user", content: "no tool result here" },
|
|
{
|
|
role: "assistant",
|
|
content: [{ type: "tool_use", id: "call_b", name: "search", input: {} }],
|
|
},
|
|
{
|
|
role: "user",
|
|
content: [{ type: "tool_result", tool_use_id: "call_b", content: "done" }],
|
|
},
|
|
],
|
|
});
|
|
|
|
assert.equal(missingResponseFixed.messages[1].role, "tool");
|
|
assert.equal(missingResponseFixed.messages[1].tool_call_id, "call_a");
|
|
assert.equal(missingResponseFixed.messages[1].content, "");
|
|
assert.deepEqual(
|
|
toolCallHelper.getToolCallIds({
|
|
role: "assistant",
|
|
tool_calls: [{ id: "call_a" }],
|
|
content: [{ type: "tool_use", id: "call_b" }],
|
|
}),
|
|
["call_a", "call_b"]
|
|
);
|
|
assert.equal(toolCallHelper.getToolCallIds({ role: "user" }).length, 0);
|
|
assert.equal(
|
|
toolCallHelper.hasToolResults({ role: "tool", tool_call_id: "call_a" }, ["call_a"]),
|
|
true
|
|
);
|
|
assert.equal(
|
|
toolCallHelper.hasToolResults(
|
|
{ role: "user", content: [{ type: "tool_result", tool_use_id: "call_b" }] },
|
|
["call_b"]
|
|
),
|
|
true
|
|
);
|
|
assert.equal(toolCallHelper.hasToolResults({ role: "user", content: [] }, []), false);
|
|
assert.deepEqual(toolCallHelper.fixMissingToolResponses({ messages: null }), { messages: null });
|
|
});
|
|
|
|
test("fixMissingToolResponses inserts Claude tool_result block when assistant uses Claude shape", () => {
|
|
const fixed = toolCallHelper.fixMissingToolResponses({
|
|
messages: [
|
|
{ role: "user", content: [{ type: "text", text: "do it" }] },
|
|
{
|
|
role: "assistant",
|
|
content: [
|
|
{ type: "tool_use", id: "tool_a", name: "bash", input: { cmd: "ls" } },
|
|
{ type: "tool_use", id: "tool_b", name: "bash", input: { cmd: "pwd" } },
|
|
],
|
|
},
|
|
{ role: "user", content: [{ type: "text", text: "continue" }] },
|
|
],
|
|
});
|
|
|
|
assert.equal(fixed.messages.length, 4);
|
|
const inserted = fixed.messages[2];
|
|
assert.equal(inserted.role, "user");
|
|
assert.ok(Array.isArray(inserted.content));
|
|
assert.equal(inserted.content.length, 2);
|
|
assert.equal(inserted.content[0].type, "tool_result");
|
|
assert.equal(inserted.content[0].tool_use_id, "tool_a");
|
|
assert.equal(inserted.content[0].content, "");
|
|
assert.equal(inserted.content[1].tool_use_id, "tool_b");
|
|
});
|
|
|
|
test("fixMissingToolResponses keeps OpenAI role:tool when assistant uses OpenAI tool_calls", () => {
|
|
const fixed = toolCallHelper.fixMissingToolResponses({
|
|
messages: [
|
|
{
|
|
role: "assistant",
|
|
tool_calls: [
|
|
{ id: "call_a", type: "function", function: { name: "lookup", arguments: "{}" } },
|
|
{ id: "call_b", type: "function", function: { name: "search", arguments: "{}" } },
|
|
],
|
|
},
|
|
{ role: "user", content: "no tool result here" },
|
|
],
|
|
});
|
|
|
|
assert.equal(fixed.messages.length, 4);
|
|
assert.equal(fixed.messages[1].role, "tool");
|
|
assert.equal(fixed.messages[1].tool_call_id, "call_a");
|
|
assert.equal(fixed.messages[2].role, "tool");
|
|
assert.equal(fixed.messages[2].tool_call_id, "call_b");
|
|
});
|
|
|
|
test("fallbackToolCallId returns the right id shape with and without an index", () => {
|
|
const noIndex = toolCallHelper.fallbackToolCallId();
|
|
assert.match(
|
|
noIndex,
|
|
/^call_\d+$/,
|
|
"no-index form must be `call_<ts>` (matches kiro/openai-responses fallback shape)"
|
|
);
|
|
|
|
const withIndex = toolCallHelper.fallbackToolCallId(2);
|
|
assert.match(
|
|
withIndex,
|
|
/^call_2_\d+$/,
|
|
"index form must be `call_<i>_<ts>` (matches indexed fallback shape)"
|
|
);
|
|
|
|
// index 0 is falsy but defined — must still produce the indexed form, not the no-index form.
|
|
const zeroIndex = toolCallHelper.fallbackToolCallId(0);
|
|
assert.match(zeroIndex, /^call_0_\d+$/, "index 0 must use the indexed form, not the bare form");
|
|
});
|
|
|
|
test("translateRequest replays cached reasoning-only messages when interleaved field is reasoning_content", () => {
|
|
clearReasoningCacheAll();
|
|
clearModelsDevCapabilities();
|
|
saveModelsDevCapabilities({
|
|
deepseek: {
|
|
"deepseek-v4-flash": buildCapability({
|
|
interleaved_field: "reasoning_content",
|
|
reasoning: true,
|
|
tool_call: true,
|
|
}),
|
|
},
|
|
});
|
|
cacheReasoningByKey(
|
|
"request:req_reasoning_only:message:0",
|
|
"deepseek",
|
|
"deepseek-v4-flash",
|
|
"cached reasoning only"
|
|
);
|
|
|
|
const result = translateRequest(
|
|
FORMATS.OPENAI,
|
|
FORMATS.OPENAI,
|
|
"deepseek-v4-flash",
|
|
{
|
|
_reasoningCacheRequestId: "req_reasoning_only",
|
|
messages: [
|
|
{ role: "user", content: "solve this" },
|
|
{ role: "assistant", content: "answer", reasoning_content: "" },
|
|
],
|
|
},
|
|
false,
|
|
null,
|
|
"deepseek"
|
|
);
|
|
|
|
assert.equal(result.messages[1].reasoning_content, "cached reasoning only");
|
|
assert.equal(getReasoningCacheServiceStats().replays, 1);
|
|
clearModelsDevCapabilities();
|
|
clearReasoningCacheAll();
|
|
});
|
|
|
|
test("translateRequest does not replay reasoning-only messages for non-DeepSeek models", () => {
|
|
clearReasoningCacheAll();
|
|
cacheReasoningByKey(
|
|
"request:req_kimi_reasoning_only:message:0",
|
|
"kimi",
|
|
"kimi-k2.5",
|
|
"cached kimi reasoning"
|
|
);
|
|
|
|
const result = translateRequest(
|
|
FORMATS.OPENAI,
|
|
FORMATS.OPENAI,
|
|
"kimi-k2.5",
|
|
{
|
|
_reasoningCacheRequestId: "req_kimi_reasoning_only",
|
|
messages: [
|
|
{ role: "user", content: "solve this" },
|
|
{ role: "assistant", content: "answer", reasoning_content: "" },
|
|
],
|
|
},
|
|
false,
|
|
null,
|
|
"kimi"
|
|
);
|
|
|
|
assert.equal(result.messages[1].reasoning_content, undefined);
|
|
assert.equal(getReasoningCacheServiceStats().replays, 0);
|
|
clearReasoningCacheAll();
|
|
});
|
|
|
|
test("translateRequest injects thinking block into Claude-format messages for Kimi K2 reasoning models", () => {
|
|
clearReasoningCacheAll();
|
|
clearModelsDevCapabilities();
|
|
saveModelsDevCapabilities({
|
|
"kimi-coding": {
|
|
"kimi-k2.5": buildCapability({
|
|
interleaved_field: "reasoning_content",
|
|
reasoning: true,
|
|
tool_call: true,
|
|
}),
|
|
},
|
|
});
|
|
cacheReasoningByKey(
|
|
"toolu_kimi_claude",
|
|
"kimi-coding",
|
|
"kimi-k2.5",
|
|
"cached thinking for Kimi tool call"
|
|
);
|
|
|
|
// Claude-format request: assistant has tool_use in content[] but NO thinking block
|
|
// This simulates the scenario that causes infinite loops
|
|
const result = translateRequest(
|
|
FORMATS.OPENAI,
|
|
FORMATS.CLAUDE,
|
|
"kimi-k2.5",
|
|
{
|
|
thinking: { type: "enabled", budget_tokens: 2000 },
|
|
messages: [
|
|
{ role: "user", content: "read the file" },
|
|
{
|
|
role: "assistant",
|
|
content: [
|
|
{
|
|
type: "tool_use",
|
|
id: "toolu_kimi_claude",
|
|
name: "read_file",
|
|
input: { path: "test.ts" },
|
|
},
|
|
],
|
|
},
|
|
{ role: "tool", tool_call_id: "toolu_kimi_claude", content: "file data" },
|
|
],
|
|
},
|
|
false,
|
|
null,
|
|
"kimi-coding"
|
|
);
|
|
|
|
const assistantMsg = result.messages.find((m) => m.role === "assistant");
|
|
assert.ok(assistantMsg, "assistant message should exist");
|
|
assert.ok(Array.isArray(assistantMsg.content), "content should be array");
|
|
|
|
// Should have a thinking block injected before tool_use
|
|
const thinkingBlock = assistantMsg.content.find((b) => b?.type === "thinking");
|
|
assert.ok(thinkingBlock, "thinking block should be injected");
|
|
assert.equal(
|
|
thinkingBlock.thinking,
|
|
"cached thinking for Kimi tool call",
|
|
"should use cached reasoning"
|
|
);
|
|
|
|
// Thinking block should appear before tool_use
|
|
const thinkingIdx = assistantMsg.content.indexOf(thinkingBlock);
|
|
const toolUseIdx = assistantMsg.content.findIndex((b) => b?.type === "tool_use");
|
|
assert.ok(thinkingIdx < toolUseIdx, "thinking block should be before tool_use");
|
|
|
|
assert.equal(getReasoningCacheServiceStats().replays, 1);
|
|
clearModelsDevCapabilities();
|
|
clearReasoningCacheAll();
|
|
});
|
|
|
|
test("translateRequest injects placeholder thinking block for Claude-format Kimi K2 on cache miss", () => {
|
|
clearReasoningCacheAll();
|
|
clearModelsDevCapabilities();
|
|
saveModelsDevCapabilities({
|
|
"kimi-coding": {
|
|
"kimi-k2.6": buildCapability({
|
|
interleaved_field: "reasoning_content",
|
|
reasoning: true,
|
|
tool_call: true,
|
|
}),
|
|
},
|
|
});
|
|
|
|
// No cache seeded - should fall back to placeholder
|
|
const result = translateRequest(
|
|
FORMATS.OPENAI,
|
|
FORMATS.CLAUDE,
|
|
"kimi-k2.6",
|
|
{
|
|
thinking: { type: "enabled", budget_tokens: 2000 },
|
|
messages: [
|
|
{ role: "user", content: "do it" },
|
|
{
|
|
role: "assistant",
|
|
content: [
|
|
{ type: "tool_use", id: "toolu_miss", name: "bash", input: { command: "ls" } },
|
|
],
|
|
},
|
|
{ role: "tool", tool_call_id: "toolu_miss", content: "output" },
|
|
],
|
|
},
|
|
false,
|
|
null,
|
|
"kimi-coding"
|
|
);
|
|
|
|
const assistantMsg = result.messages.find((m) => m.role === "assistant");
|
|
assert.ok(assistantMsg, "assistant message should exist");
|
|
|
|
const thinkingBlock =
|
|
Array.isArray(assistantMsg.content) &&
|
|
assistantMsg.content.find((b) => b?.type === "thinking");
|
|
assert.ok(thinkingBlock, "thinking block should be injected on cache miss");
|
|
// Must be non-empty for kimi-coding
|
|
assert.ok(
|
|
thinkingBlock.thinking && thinkingBlock.thinking.length > 0,
|
|
"placeholder must be non-empty"
|
|
);
|
|
|
|
clearModelsDevCapabilities();
|
|
clearReasoningCacheAll();
|
|
});
|
|
|
|
test("translateRequest does NOT inject duplicate thinking for Claude-format messages with existing thinking block", () => {
|
|
clearReasoningCacheAll();
|
|
clearModelsDevCapabilities();
|
|
saveModelsDevCapabilities({
|
|
"kimi-coding": {
|
|
"kimi-k2.5": buildCapability({
|
|
interleaved_field: "reasoning_content",
|
|
reasoning: true,
|
|
tool_call: true,
|
|
}),
|
|
},
|
|
});
|
|
|
|
const result = translateRequest(
|
|
FORMATS.OPENAI,
|
|
FORMATS.CLAUDE,
|
|
"kimi-k2.5",
|
|
{
|
|
messages: [
|
|
{ role: "user", content: "hi" },
|
|
{
|
|
role: "assistant",
|
|
content: [
|
|
{ type: "thinking", thinking: "I already have this" },
|
|
{ type: "tool_use", id: "toolu_existing", name: "read", input: {} },
|
|
],
|
|
},
|
|
{ role: "tool", tool_call_id: "toolu_existing", content: "data" },
|
|
],
|
|
},
|
|
false,
|
|
null,
|
|
"kimi-coding"
|
|
);
|
|
|
|
const assistantMsg = result.messages.find((m) => m.role === "assistant");
|
|
const thinkingBlocks =
|
|
Array.isArray(assistantMsg.content) &&
|
|
assistantMsg.content.filter((b) => b?.type === "thinking");
|
|
assert.equal(
|
|
thinkingBlocks?.length,
|
|
1,
|
|
"should have exactly one thinking block (no duplicate)"
|
|
);
|
|
assert.equal(
|
|
thinkingBlocks[0].thinking,
|
|
"I already have this",
|
|
"original thinking should be preserved"
|
|
);
|
|
|
|
clearModelsDevCapabilities();
|
|
clearReasoningCacheAll();
|
|
});
|