From a334c9b0cedb425cdd3039836147b0d982a576d0 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Wed, 2 Sep 2026 18:22:42 -0300 Subject: [PATCH 001/143] build(deps): bump github/codeql-action/init from 4.37.8 to 4.37.9 (#12345) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bump de action pinada por SHA para codeql-action v4.37.9. Validado: os pins de `init` e `analyze` (#12345/#12346) apontam para o mesmo commit `cdf488f595d80d6e07e03d4674febd5ab45fa938`, consistente com a tag v4.37.9; nenhum código de aplicação afetado. Obrigado, Dependabot. --- .github/workflows/codeql.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index 6adc2d6669..b58efdb152 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -22,7 +22,7 @@ jobs: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: persist-credentials: false - - uses: github/codeql-action/init@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v4.37.8 + - uses: github/codeql-action/init@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 with: languages: javascript-typescript queries: security-extended From b1fd07df28fbf2bcb9fde31f645f045a33af4c26 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Wed, 2 Sep 2026 18:22:47 -0300 Subject: [PATCH 002/143] build(deps): bump github/codeql-action/analyze from 4.37.8 to 4.37.9 (#12346) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bump de action pinada por SHA para codeql-action v4.37.9. Validado: os pins de `init` e `analyze` (#12345/#12346) apontam para o mesmo commit `cdf488f595d80d6e07e03d4674febd5ab45fa938`, consistente com a tag v4.37.9; nenhum código de aplicação afetado. Obrigado, Dependabot. --- .github/workflows/codeql.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index b58efdb152..4ba95c5652 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -26,6 +26,6 @@ jobs: with: languages: javascript-typescript queries: security-extended - - uses: github/codeql-action/analyze@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v4.37.8 + - uses: github/codeql-action/analyze@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 with: category: "/language:javascript-typescript" From 51cd154da1b9c7cb9b06146ef1e94134243a9e00 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Wed, 2 Sep 2026 18:22:52 -0300 Subject: [PATCH 003/143] build(deps): bump github/codeql-action from 4.37.8 to 4.37.9 (#12349) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bump de action pinada por SHA para codeql-action v4.37.9. Validado: os pins de `init` e `analyze` (#12345/#12346) apontam para o mesmo commit `cdf488f595d80d6e07e03d4674febd5ab45fa938`, consistente com a tag v4.37.9; nenhum código de aplicação afetado. Obrigado, Dependabot. --- .github/workflows/docker-publish.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/docker-publish.yml b/.github/workflows/docker-publish.yml index 917876b361..77e722b318 100644 --- a/.github/workflows/docker-publish.yml +++ b/.github/workflows/docker-publish.yml @@ -535,7 +535,7 @@ jobs: - name: Upload Trivy SARIF to Security tab if: needs.prepare.outputs.version != 'main' continue-on-error: true - uses: github/codeql-action/upload-sarif@v4.37.8 + uses: github/codeql-action/upload-sarif@v4.37.9 with: sarif_file: trivy-results.sarif category: trivy-image From 7448590b8179f356bf91e95a3854611d09e15034 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Wed, 2 Sep 2026 18:23:34 -0300 Subject: [PATCH 004/143] deps: bump @xmldom/xmldom (#12500) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bump do override `@xmldom/xmldom` 0.9.10 → 0.9.12 no workspace `electron/` (grupo npm_and_yarn). Escopo isolado: só `electron/package.json` + lock, sem interseção com o install da raiz; a redução de ~199 linhas no lock é dedupe da própria resolução. Obrigado, Dependabot. --- electron/package-lock.json | 201 +------------------------------------ electron/package.json | 2 +- 2 files changed, 4 insertions(+), 199 deletions(-) diff --git a/electron/package-lock.json b/electron/package-lock.json index 61b1da8b88..1aeabc7edc 100644 --- a/electron/package-lock.json +++ b/electron/package-lock.json @@ -297,45 +297,6 @@ "url": "https://github.com/sponsors/isaacs" } }, - "node_modules/@electron/windows-sign": { - "version": "1.2.2", - "resolved": "https://registry.npmjs.org/@electron/windows-sign/-/windows-sign-1.2.2.tgz", - "integrity": "sha512-dfZeox66AvdPtb2lD8OsIIQh12Tp0GNCRUDfBHIKGpbmopZto2/A8nSpYYLoedPIHpqkeblZ/k8OV0Gy7PYuyQ==", - "dev": true, - "license": "BSD-2-Clause", - "optional": true, - "peer": true, - "dependencies": { - "cross-dirname": "^0.1.0", - "debug": "^4.3.4", - "fs-extra": "^11.1.1", - "minimist": "^1.2.8", - "postject": "^1.0.0-alpha.6" - }, - "bin": { - "electron-windows-sign": "bin/electron-windows-sign.js" - }, - "engines": { - "node": ">=14.14" - } - }, - "node_modules/@electron/windows-sign/node_modules/fs-extra": { - "version": "11.4.0", - "resolved": "https://registry.npmjs.org/fs-extra/-/fs-extra-11.4.0.tgz", - "integrity": "sha512-EQsFzMUJkCKGr1ePqlYADkIUmHW1s3ZXr5Yqy6wbGrfUCphpl2maM/kyOIRA2HpP3AaFQTZXD4ldjek+nccddA==", - "dev": true, - "license": "MIT", - "optional": true, - "peer": true, - "dependencies": { - "graceful-fs": "^4.2.0", - "jsonfile": "^6.0.1", - "universalify": "^2.0.0" - }, - "engines": { - "node": ">=14.14" - } - }, "node_modules/@isaacs/fs-minipass": { "version": "4.0.1", "resolved": "https://registry.npmjs.org/@isaacs/fs-minipass/-/fs-minipass-4.0.1.tgz", @@ -573,9 +534,9 @@ } }, "node_modules/@xmldom/xmldom": { - "version": "0.9.10", - "resolved": "https://registry.npmjs.org/@xmldom/xmldom/-/xmldom-0.9.10.tgz", - "integrity": "sha512-A9gOqLdi6cV4ibazAjcQufGj0B1y/vDqYrcuP6d/6x8P27gRS8643Dj9o1dEKtB6O7fwxb2FgBmJS2mX7gpvdw==", + "version": "0.9.12", + "resolved": "https://registry.npmjs.org/@xmldom/xmldom/-/xmldom-0.9.12.tgz", + "integrity": "sha512-5AXjrcMClTryPe9LgZrygpB1lj7s0S9E0+W+AHaVKAVyHanafK86iPSvG5xHVSp/jC+VH1UXu0TAEmY279xH7A==", "dev": true, "license": "MIT", "engines": { @@ -1130,15 +1091,6 @@ "dev": true, "license": "MIT" }, - "node_modules/cross-dirname": { - "version": "0.1.0", - "resolved": "https://registry.npmjs.org/cross-dirname/-/cross-dirname-0.1.0.tgz", - "integrity": "sha512-+R08/oI0nl3vfPcqftZRpytksBXDzOUveBq/NBVx0sUp1axwzPQrKinNx5yd5sxPu8j1wIy8AfnVQ+5eFdha6Q==", - "dev": true, - "license": "MIT", - "optional": true, - "peer": true - }, "node_modules/cross-spawn": { "version": "7.0.6", "resolved": "https://registry.npmjs.org/cross-spawn/-/cross-spawn-7.0.6.tgz", @@ -1459,19 +1411,6 @@ "node": ">=14.0.0" } }, - "node_modules/electron-builder-squirrel-windows": { - "version": "26.15.3", - "resolved": "https://registry.npmjs.org/electron-builder-squirrel-windows/-/electron-builder-squirrel-windows-26.15.3.tgz", - "integrity": "sha512-Jc19XPV9y9+2bAdZPkXuVNGNIEFBq9poHC61l8Kv6FdK7DRG3+Ic0rerC0DXOaeHNz8yW0fg/JnF8GQROOF5MA==", - "dev": true, - "license": "MIT", - "peer": true, - "dependencies": { - "app-builder-lib": "26.15.3", - "builder-util": "26.15.3", - "electron-winstaller": "5.4.0" - } - }, "node_modules/electron-publish": { "version": "26.15.3", "resolved": "https://registry.npmjs.org/electron-publish/-/electron-publish-26.15.3.tgz", @@ -1506,66 +1445,6 @@ "tiny-typed-emitter": "^2.1.0" } }, - "node_modules/electron-winstaller": { - "version": "5.4.0", - "resolved": "https://registry.npmjs.org/electron-winstaller/-/electron-winstaller-5.4.0.tgz", - "integrity": "sha512-bO3y10YikuUwUuDUQRM4KfwNkKhnpVO7IPdbsrejwN9/AABJzzTQ4GeHwyzNSrVO+tEH3/Np255a3sVZpZDjvg==", - "dev": true, - "hasInstallScript": true, - "license": "MIT", - "peer": true, - "dependencies": { - "@electron/asar": "^3.2.1", - "debug": "^4.1.1", - "fs-extra": "^7.0.1", - "lodash": "^4.17.21", - "temp": "^0.9.0" - }, - "engines": { - "node": ">=8.0.0" - }, - "optionalDependencies": { - "@electron/windows-sign": "^1.1.2" - } - }, - "node_modules/electron-winstaller/node_modules/fs-extra": { - "version": "7.0.1", - "resolved": "https://registry.npmjs.org/fs-extra/-/fs-extra-7.0.1.tgz", - "integrity": "sha512-YJDaCJZEnBmcbw13fvdAM9AwNOJwOzrE4pqMqBq5nFiEqXUqHwlK4B+3pUw6JNvfSPtX05xFHtYy/1ni01eGCw==", - "dev": true, - "license": "MIT", - "peer": true, - "dependencies": { - "graceful-fs": "^4.1.2", - "jsonfile": "^4.0.0", - "universalify": "^0.1.0" - }, - "engines": { - "node": ">=6 <7 || >=8" - } - }, - "node_modules/electron-winstaller/node_modules/jsonfile": { - "version": "4.0.0", - "resolved": "https://registry.npmjs.org/jsonfile/-/jsonfile-4.0.0.tgz", - "integrity": "sha512-m6F1R3z8jjlf2imQHS2Qez5sjKWQzbuuhuJ/FKYFRZvPE3PuHcSMVZzfsLhGVOkfd20obL5SWEBew5ShlquNxg==", - "dev": true, - "license": "MIT", - "peer": true, - "optionalDependencies": { - "graceful-fs": "^4.1.6" - } - }, - "node_modules/electron-winstaller/node_modules/universalify": { - "version": "0.1.2", - "resolved": "https://registry.npmjs.org/universalify/-/universalify-0.1.2.tgz", - "integrity": "sha512-rBJeI5CXAlmy1pV+617WB9J63U6XcazHHF2f2dbJix4XzpUF0RS3Zbj0FGIOCAva5P/d/GBOYaACQ1w+0azUkg==", - "dev": true, - "license": "MIT", - "peer": true, - "engines": { - "node": ">= 4.0.0" - } - }, "node_modules/emoji-regex": { "version": "8.0.0", "resolved": "https://registry.npmjs.org/emoji-regex/-/emoji-regex-8.0.0.tgz", @@ -2480,20 +2359,6 @@ "node": ">= 18" } }, - "node_modules/mkdirp": { - "version": "0.5.6", - "resolved": "https://registry.npmjs.org/mkdirp/-/mkdirp-0.5.6.tgz", - "integrity": "sha512-FP+p8RB8OWpF3YZBCrP5gtADmtXApB5AMLn+vdyA+PyxCjrCs00mjyUozssO33cwDeT3wNGdLxJ5M//YqtHAJw==", - "dev": true, - "license": "MIT", - "peer": true, - "dependencies": { - "minimist": "^1.2.6" - }, - "bin": { - "mkdirp": "bin/cmd.js" - } - }, "node_modules/ms": { "version": "2.1.3", "resolved": "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz", @@ -2757,36 +2622,6 @@ "node": ">=18" } }, - "node_modules/postject": { - "version": "1.0.0-alpha.6", - "resolved": "https://registry.npmjs.org/postject/-/postject-1.0.0-alpha.6.tgz", - "integrity": "sha512-b9Eb8h2eVqNE8edvKdwqkrY6O7kAwmI8kcnBv1NScolYJbo59XUF0noFq+lxbC1yN20bmC0WBEbDC5H/7ASb0A==", - "dev": true, - "license": "MIT", - "optional": true, - "peer": true, - "dependencies": { - "commander": "^9.4.0" - }, - "bin": { - "postject": "dist/cli.js" - }, - "engines": { - "node": ">=14.0.0" - } - }, - "node_modules/postject/node_modules/commander": { - "version": "9.5.0", - "resolved": "https://registry.npmjs.org/commander/-/commander-9.5.0.tgz", - "integrity": "sha512-KRs7WVDKg86PWiuAqhDrAQnTXZKraVcCc6vFdL14qrZ/DcWwuRo7VoiYXalXO7S5GKpqYiVEwCbgFDfxNHKJBQ==", - "dev": true, - "license": "MIT", - "optional": true, - "peer": true, - "engines": { - "node": "^12.20.0 || >=14" - } - }, "node_modules/proc-log": { "version": "6.1.0", "resolved": "https://registry.npmjs.org/proc-log/-/proc-log-6.1.0.tgz", @@ -2981,21 +2816,6 @@ "node": ">= 4" } }, - "node_modules/rimraf": { - "version": "2.6.3", - "resolved": "https://registry.npmjs.org/rimraf/-/rimraf-2.6.3.tgz", - "integrity": "sha512-mwqeW5XsA2qAejG46gYdENaxXjx9onRNCfn7L0duuP4hCuTIi/QO7PDK07KJfp1d+izWPrzEJDcSqBa0OZQriA==", - "deprecated": "Rimraf versions prior to v4 are no longer supported", - "dev": true, - "license": "ISC", - "peer": true, - "dependencies": { - "glob": "^7.1.3" - }, - "bin": { - "rimraf": "bin.js" - } - }, "node_modules/roarr": { "version": "2.15.4", "resolved": "https://registry.npmjs.org/roarr/-/roarr-2.15.4.tgz", @@ -3251,21 +3071,6 @@ "node": ">=18" } }, - "node_modules/temp": { - "version": "0.9.4", - "resolved": "https://registry.npmjs.org/temp/-/temp-0.9.4.tgz", - "integrity": "sha512-yYrrsWnrXMcdsnu/7YMYAofM1ktpL5By7vZhf15CrXijWWrEYZks5AXBudalfSWJLlnen/QUJUB5aoB0kqZUGA==", - "dev": true, - "license": "MIT", - "peer": true, - "dependencies": { - "mkdirp": "^0.5.1", - "rimraf": "~2.6.2" - }, - "engines": { - "node": ">=6.0.0" - } - }, "node_modules/temp-file": { "version": "3.4.0", "resolved": "https://registry.npmjs.org/temp-file/-/temp-file-3.4.0.tgz", diff --git a/electron/package.json b/electron/package.json index bfd803a045..d4fd2bd1b2 100644 --- a/electron/package.json +++ b/electron/package.json @@ -32,7 +32,7 @@ "electron-builder": "^26.15.3" }, "overrides": { - "@xmldom/xmldom": "^0.9.10", + "@xmldom/xmldom": "^0.9.12", "fast-uri": "^3.1.3", "plist": "^4.0.0", "form-data": "^4.0.6", From 86e83d4138441c3c842c2a3bc5c20d1b06db41d1 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Wed, 2 Sep 2026 18:32:51 -0300 Subject: [PATCH 005/143] fix(video): re-anchor transcript log-redaction so PII/credential maskers can't reopen the leak (#12150 P1 follow-up) (#12503) * fix(video): re-anchor log-redaction fullText from the finished guardrail payload so PII/credential maskers can't reopen the transcript leak * docs(video): clarify P1 transcript-retention scope and re-anchor; list P2 surfaces (#12430) --- docs/security/GUARDRAILS.md | 39 +++++++++------ src/lib/guardrails/videoBridge.ts | 32 ++++++++++++ src/sse/handlers/chat.ts | 26 +++++++--- tests/unit/guardrails/videoBridge.test.ts | 61 ++++++++++++++++++++++- 4 files changed, 134 insertions(+), 24 deletions(-) diff --git a/docs/security/GUARDRAILS.md b/docs/security/GUARDRAILS.md index e4b956e43f..6d2e21c61e 100644 --- a/docs/security/GUARDRAILS.md +++ b/docs/security/GUARDRAILS.md @@ -476,23 +476,32 @@ fusion counters. The default Video Bridge path does not invoke speech-to-text or download a second media copy; without that explicit track, it remains video-only. -**Transcript retention (opt-in feature, #12150 P1).** When a request renders any -transcript cue (a caller-declared `transcript` or a fused `audioTranscript`), the -guardrail marks it `videoBridgeObserved` and produces a redacted shadow of the -video description — an identical rendering in which every cue's free-text body is -replaced by `[redacted-video-transcript]`, built by substituting the structured -cue field before the string is assembled (never by parsing the flattened text, so -no cue content — adversarial or ordinary, including bodies containing `]` such as +**Transcript retention (#12150 P1).** This applies automatically whenever the +Video Bridge (itself opt-in) renders a transcript cue — there is no separate +retention flag. When a request renders any transcript cue (a caller-declared +`transcript` or a fused `audioTranscript`), the guardrail marks it +`videoBridgeObserved` and produces a redacted shadow of the video description — +an identical rendering in which every cue's free-text body is replaced by +`[redacted-video-transcript]`, built by substituting the structured cue field +before the string is assembled (never by parsing the flattened text, so no cue +content — adversarial or ordinary, including bodies containing `]` such as `[inaudible]`/`[music]` — can survive). The persisted call-log request body swaps each video-derived text part for that redacted shadow, matched by content -equality (so it stays correct even after system-prompt/handoff/memory injection -reshapes the message array); the body sent upstream to the model is unchanged. -An observed request also populates no durable Memory (both request- and -response-derived extraction are skipped), so the model's own reply cannot echo -transcript text into Memory. Two further retention surfaces — the raw -pre-guardrail client-request snapshot in the detailed-log artifact and -`previous_response_id` continuation fail-closed — are tracked for a follow-up -(P2) and are not yet closed. +equality; the `fullText` anchor is re-read from the finished pre-call guardrail +payload, so the match still succeeds after later chain guardrails (the PII and +credential maskers, priorities 10/95) rewrite the description text in place and +after system-prompt/handoff/memory injection reshapes the message array. The +body sent upstream to the model is unchanged. An observed request also populates +no durable Memory (both request- and response-derived extraction are skipped), +so the model's own reply cannot echo transcript text into Memory. + +Retention surfaces still open, tracked for a follow-up (**P2**, #12430): the raw +pre-guardrail client-request snapshot in the detailed-log artifact; +`previous_response_id` continuation fail-closed; derived-prompt internal +dispatches that embed the transcript inside a synthesized string prompt +(pipeline stages, context-handoff); and the response body / semantic-cache copy +of a model reply that quotes the transcript. These are raw/response-class or +opt-in surfaces outside P1's persisted-request-body + Memory scope. The internal `/api/modality-bridge/video/drilldown` lifecycle is a separate, loopback/token-authenticated cache substrate. Every operation also requires a diff --git a/src/lib/guardrails/videoBridge.ts b/src/lib/guardrails/videoBridge.ts index f8425a188c..216c525365 100644 --- a/src/lib/guardrails/videoBridge.ts +++ b/src/lib/guardrails/videoBridge.ts @@ -66,6 +66,38 @@ export interface VideoBridgeLogRedactionEntry { redactedText: string; } +/** + * #12150 P1 final-review fix: re-anchor each redaction entry's `fullText` from + * the FINAL pre-call guardrail payload. The video-bridge guardrail runs at + * priority 7, but the PII masker (10) and credential masker (95) rewrite the + * SAME chained payload afterward, in place — so by the end of the chain the + * replaced part's text may differ from what video-bridge recorded, and the log + * sink's content-match (`part.text === fullText`) would miss (fail open). Chain + * guardrails only rewrite text in place — they never splice the message array — + * so the advisory `(container, messageIndex, partIndex)` still resolves inside + * the finished chain payload; reading the part text there yields the true + * post-chain text the log sink will see. Falls back to the original `fullText` + * when the index no longer resolves. Returns new entries; never mutates the + * shared guardrail `meta` array. `redactedText` is unchanged (it is rendered + * from the structured cues, independent of any masker rewrite). + */ +export function reanchorVideoBridgeRedaction( + entries: readonly VideoBridgeLogRedactionEntry[], + finalBody: unknown +): VideoBridgeLogRedactionEntry[] { + const body = finalBody as Record | null | undefined; + return entries.map((entry) => { + const container = body?.[entry.container]; + if (!Array.isArray(container)) return { ...entry }; + const message = container[entry.messageIndex] as { content?: unknown } | undefined; + const content = message?.content; + if (!Array.isArray(content)) return { ...entry }; + const part = content[entry.partIndex] as { text?: unknown } | undefined; + if (!part || typeof part.text !== "string") return { ...entry }; + return { ...entry, fullText: part.text }; + }); +} + type VideoBridgeBody = { model?: string; messages?: Array<{ role?: string; content?: unknown }>; diff --git a/src/sse/handlers/chat.ts b/src/sse/handlers/chat.ts index 7747bc00c1..6c426bcfce 100644 --- a/src/sse/handlers/chat.ts +++ b/src/sse/handlers/chat.ts @@ -104,6 +104,7 @@ import { } from "./chatHelpers"; import { buildModalityBridgeHeader } from "@/lib/guardrails/modalityBridge/bridgeStats"; import type { VideoBridgeLogRedactionEntry } from "@/lib/guardrails/videoBridge"; +import { reanchorVideoBridgeRedaction } from "@/lib/guardrails/videoBridge"; import { resolveConversationId } from "@omniroute/open-sse/services/conversationTracker.ts"; import { classifyProviderBreakerResult, @@ -315,11 +316,18 @@ type VideoBridgeLog = { observed: boolean; redaction: VideoBridgeLogRedactionEnt /** * #12150 P1b: derive the video-bridge log/Memory shadow from - * preCallGuardrails.results. Returns undefined when the video-bridge - * guardrail did not run (disabled, no video parts) or ran but rendered no - * transcript cue (ordinary video, or the request was blocked/failed before - * meta was set) — so every non-video request threads `undefined` through the - * dispatch chain, byte-identical to before this param existed. + * preCallGuardrails.results. Returns undefined only when the video-bridge + * guardrail did not run (disabled, no video parts, or the request was + * blocked/failed before meta was set); a replaced ordinary video returns + * `{ observed: false, redaction: [] }`. So every non-video request threads + * `undefined` through the dispatch chain, byte-identical to before this param + * existed. + * + * `finalBody` is the payload AFTER the whole pre-call chain + * (`preCallGuardrails.payload`): #12150 P1 final-review fix re-anchors each + * redaction entry's `fullText` from it so the log sink's content-match still + * finds the part after the PII/credential maskers (priorities 10/95) rewrote + * the description text in place. * * `results` is typed as a structural subset of GuardrailExecutionResult * (src/lib/guardrails/base.ts), the same "no type dependency on the @@ -327,14 +335,16 @@ type VideoBridgeLog = { observed: boolean; redaction: VideoBridgeLogRedactionEnt * (modalityBridge/bridgeStats.ts). */ function deriveVideoBridgeLog( - results: Array<{ guardrail: string; meta?: Record | null }> + results: Array<{ guardrail: string; meta?: Record | null }>, + finalBody: unknown ): VideoBridgeLog | undefined { const entry = results.find((r) => r.guardrail === "video-bridge"); const meta = entry?.meta; if (!meta || typeof meta.videoBridgeObserved !== "boolean") return undefined; - const redaction = Array.isArray(meta.videoBridgeLogRedaction) + const rawRedaction = Array.isArray(meta.videoBridgeLogRedaction) ? (meta.videoBridgeLogRedaction as VideoBridgeLogRedactionEntry[]) : []; + const redaction = reanchorVideoBridgeRedaction(rawRedaction, finalBody); return { observed: meta.videoBridgeObserved, redaction }; } @@ -774,7 +784,7 @@ async function handleChatImplementation( // #12150 P1b: video-bridge log/Memory shadow — undefined on every // non-video request. Threaded through handleSingleModelChat's // runtimeOptions -> executeChatWithBreaker -> handleChatCore. - const videoBridgeLog = deriveVideoBridgeLog(preCallGuardrails.results); + const videoBridgeLog = deriveVideoBridgeLog(preCallGuardrails.results, body); telemetry.endPhase(); // Agentic conversation tracking (X-ConversationId): resolved once per diff --git a/tests/unit/guardrails/videoBridge.test.ts b/tests/unit/guardrails/videoBridge.test.ts index fa62495eba..0aaafbffe6 100644 --- a/tests/unit/guardrails/videoBridge.test.ts +++ b/tests/unit/guardrails/videoBridge.test.ts @@ -1,7 +1,10 @@ import assert from "node:assert/strict"; import test from "node:test"; -import { VideoBridgeGuardrail } from "../../../src/lib/guardrails/videoBridge.ts"; +import { + VideoBridgeGuardrail, + reanchorVideoBridgeRedaction, +} from "../../../src/lib/guardrails/videoBridge.ts"; import { callVisionModel } from "../../../src/lib/guardrails/visionBridgeHelpers.ts"; import { buildModalityBridgeHeader, @@ -821,3 +824,59 @@ test("audio/video fusion telemetry reaches guardrail meta, bridge stats, and cac assert.equal(after.fusionRuns - before.fusionRuns, 2); assert.equal(after.fusionPartials - before.fusionPartials, 2); }); + +test("reanchorVideoBridgeRedaction re-reads fullText from the post-guardrail body (PII/credential masker interaction)", () => { + // A later chain guardrail (PII masker @10, credential masker @95) rewrote the + // description text IN PLACE after video-bridge@7 built the redaction map, so + // the map's fullText is stale. Re-anchoring at the advisory indices must pick + // up the post-masker text so the log sink's content-match still finds the part. + const entries = [ + { + container: "messages" as const, + messageIndex: 0, + partIndex: 1, + fullText: + "[Video description: transcript[source=client;confidence=0.90;interval=00:01.000-00:02.000] my name is Alice]", + redactedText: + "[Video description: transcript[source=client;confidence=0.90;interval=00:01.000-00:02.000] [redacted-video-transcript]]", + }, + ]; + const finalBody = { + messages: [ + { + role: "user", + content: [ + { type: "text", text: "hello" }, + { + type: "text", + // PII masker replaced "Alice" with a token in place: + text: "[Video description: transcript[source=client;confidence=0.90;interval=00:01.000-00:02.000] my name is [NAME_1]]", + }, + ], + }, + ], + }; + const reanchored = reanchorVideoBridgeRedaction(entries, finalBody); + assert.equal( + reanchored[0].fullText, + (finalBody.messages[0].content[1] as { text: string }).text, + "fullText must equal the post-masker part text so the sink match succeeds" + ); + assert.equal(reanchored[0].redactedText, entries[0].redactedText, "redactedText is unchanged"); + // Original entries object is not mutated (meta is shared). + assert.equal(entries[0].fullText.includes("Alice"), true); +}); + +test("reanchorVideoBridgeRedaction keeps original fullText when the advisory index no longer resolves", () => { + const entries = [ + { + container: "messages" as const, + messageIndex: 5, + partIndex: 9, + fullText: "[Video description: original]", + redactedText: "[Video description: [redacted-video-transcript]]", + }, + ]; + const reanchored = reanchorVideoBridgeRedaction(entries, { messages: [] }); + assert.equal(reanchored[0].fullText, "[Video description: original]"); +}); From a628d28898bb499e089602edc8a76fcb7e6b74ec Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Wed, 2 Sep 2026 20:37:02 -0300 Subject: [PATCH 006/143] =?UTF-8?q?feat(dashboard):=20orchestration=20canv?= =?UTF-8?q?as=20fase=202=20=E2=80=94=20repeat=20action=20+=20A2A=20memory?= =?UTF-8?q?=20hits=20(2.6/2.7)=20(#12508)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(api): conductor task creation route (repeat support) * feat(a2a): record memoryHits consulted per task (observability, 2.7) * feat(dashboard): repeat action in orchestration drawer (2.6 — repeat only) * fix(dashboard): require every field each repeat contract needs before enabling the action * feat(dashboard): memory-used drawer section + fase2 i18n/changelog (2.7) * fix(dashboard): validate memoryHits shape before rendering + locale wording fixes * fix(dashboard,a2a): stop memoryHits leaking into repeats, harden drawer guard + status clamp Final whole-branch review fix wave for the Orchestration Canvas Fase 2 PR-C. - a2a: `createTask` stores a COPY of `input.metadata` instead of aliasing it, so the observability `memoryHits` written by `executeA2ATaskWithState` no longer leak into `task.input.metadata`, into the persisted `a2a_tasks.input_json`, or into the drawer's "Repeat" body (a repeated task was born carrying the previous run's memory snippets, even with the `OMNIROUTE_A2A_MEMORY_HITS=0` kill-switch on). - dashboard: `repeatReqFor` strips `memoryHits` from the a2a repeat metadata, so tasks persisted before the copy-fix do not propagate them either. - dashboard: the "Memory used" section now requires all four rendered fields (id, key, type, snippet) to be strings — `{ id: "x", key: { a: 1 } }` used to throw "Objects are not valid as a React child" and take the whole drawer down. - dashboard: an `/a2a` action answered with a JSON-RPC error under HTTP 200 is reported as a failure (`RPC `, code only — never the upstream message) instead of a success toast; the secured-deployment rejection keeps surfacing the sanitized `HTTP 400`. - api: the conductor task-creation route clamps a hub status outside 400-599 to 502, so an out-of-range status can no longer turn a hub refusal into a `RangeError`. - dashboard: the History tab's `onActionDone` keeps the drawer mounted (and refreshes the range) instead of closing it, so the repeat/cancel confirmation is actually visible. - a2a: documented the recall owner-id limitation — `task.owner` is a SHA-256 key prefix while memory rows are keyed by the DB api-key id, and no hash-to-id lookup exists today, so recall only resolves under the keyless posture. * refactor(dashboard): split drawer repeat helpers and test file under the size/complexity gates --------- Co-authored-by: Markus Hartung --- .env.example | 6 + .../features/orchestration-repeat-memory.md | 24 + docs/openapi.yaml | 8 + docs/reference/ENVIRONMENT.md | 1 + .../drawer/OrchestrationDrawer.tsx | 159 +++- .../orchestration/drawer/useDrawerDetail.ts | 144 +++- .../orchestration/tabs/HistoryTab.tsx | 22 +- src/app/api/conductor/tasks/route.ts | 66 ++ src/i18n/messages/ar.json | 7 +- src/i18n/messages/az.json | 7 +- src/i18n/messages/bg.json | 7 +- src/i18n/messages/bn.json | 7 +- src/i18n/messages/cs.json | 7 +- src/i18n/messages/da.json | 7 +- src/i18n/messages/de.json | 7 +- src/i18n/messages/en.json | 7 +- src/i18n/messages/es.json | 7 +- src/i18n/messages/fa.json | 7 +- src/i18n/messages/fi.json | 7 +- src/i18n/messages/fr.json | 7 +- src/i18n/messages/gu.json | 7 +- src/i18n/messages/he.json | 7 +- src/i18n/messages/hi.json | 7 +- src/i18n/messages/hu.json | 7 +- src/i18n/messages/id.json | 7 +- src/i18n/messages/it.json | 7 +- src/i18n/messages/ja.json | 7 +- src/i18n/messages/ko.json | 7 +- src/i18n/messages/mr.json | 7 +- src/i18n/messages/ms.json | 7 +- src/i18n/messages/nl.json | 7 +- src/i18n/messages/no.json | 7 +- src/i18n/messages/phi.json | 7 +- src/i18n/messages/pl.json | 7 +- src/i18n/messages/pt-BR.json | 7 +- src/i18n/messages/pt.json | 7 +- src/i18n/messages/ro.json | 7 +- src/i18n/messages/ru.json | 7 +- src/i18n/messages/sk.json | 7 +- src/i18n/messages/sv.json | 7 +- src/i18n/messages/sw.json | 7 +- src/i18n/messages/ta.json | 7 +- src/i18n/messages/te.json | 7 +- src/i18n/messages/th.json | 7 +- src/i18n/messages/tr.json | 7 +- src/i18n/messages/uk-UA.json | 7 +- src/i18n/messages/ur.json | 7 +- src/i18n/messages/vi.json | 7 +- src/i18n/messages/zh-CN.json | 7 +- src/i18n/messages/zh-TW.json | 7 +- src/lib/a2a/taskExecution.ts | 99 ++- src/lib/a2a/taskManager.ts | 15 +- tests/unit/a2a-memory-hits.test.ts | 331 ++++++++ tests/unit/conductor-create-route.test.ts | 155 ++++ .../ui/orchestrationDrawerRepeat.test.tsx | 786 ++++++++++++++++++ .../unit/ui/orchestrationHistoryTab.test.tsx | 30 + 56 files changed, 2078 insertions(+), 62 deletions(-) create mode 100644 changelog.d/features/orchestration-repeat-memory.md create mode 100644 src/app/api/conductor/tasks/route.ts create mode 100644 tests/unit/a2a-memory-hits.test.ts create mode 100644 tests/unit/conductor-create-route.test.ts create mode 100644 tests/unit/ui/orchestrationDrawerRepeat.test.tsx diff --git a/.env.example b/.env.example index 187a1049d3..4e6438b593 100644 --- a/.env.example +++ b/.env.example @@ -869,6 +869,12 @@ NEXT_PUBLIC_ENABLE_SOCKS5_PROXY=true # or <= 0 falls back to the default. # OMNIROUTE_A2A_HISTORY_RETENTION_DAYS=30 +# Kill-switch for the A2A memory-hits observability feature (Orchestration Canvas +# Fase 2). Set to "0" to skip the memory recall lookup entirely; any other value +# (including unset) keeps it enabled. +# Used by: src/lib/a2a/taskExecution.ts (collectMemoryHits). +# OMNIROUTE_A2A_MEMORY_HITS=1 + # Enable the offline/local Issue Agent recorded-triage endpoint. # Used by: src/app/api/issue-agent/runs/route.ts. Default: disabled. # OMNIROUTE_ISSUE_AGENT_ENABLED=false diff --git a/changelog.d/features/orchestration-repeat-memory.md b/changelog.d/features/orchestration-repeat-memory.md new file mode 100644 index 0000000000..b1daddb0aa --- /dev/null +++ b/changelog.d/features/orchestration-repeat-memory.md @@ -0,0 +1,24 @@ +- **feat(dashboard):** the orchestration detail drawer gained a "Repeat" action for Cloud Agent, + A2A and Conductor tasks — a two-click confirm (click once to arm, click again within the + confirm window to fire) re-submits the original prompt/input as a new run. The button is + disabled with an explanatory tooltip whenever the original input can't be recovered from the + loaded task detail (e.g. it never carried a prompt, or the detail failed to load). + Two limitations of the A2A variant, by design: it targets the `/a2a` JSON-RPC endpoint, which + authenticates with an API key only (`REQUIRE_API_KEY=true` or a configured `OMNIROUTE_API_KEY` + makes a dashboard-session repeat answer `HTTP 400` — surfaced verbatim in the drawer's error + line, never as a success), and the `message/send` call is SYNCHRONOUS: the POST blocks for the + whole skill run, so the success confirmation only appears once the repeated task finishes. + A dashboard-authenticated A2A creation path is deliberately left to a follow-up — widening the + endpoint's auth posture is an operator decision, not a side effect of this feature. +- **feat(a2a):** A2A task execution now records which memories were consulted for the task's + last user message as `metadata.memoryHits` (id/key/type/content-snippet) plus a `memory_hits` + history event, purely for observability — the retrieved memory is never injected into a + skill's prompt or behavior. Gated by the `OMNIROUTE_A2A_MEMORY_HITS` kill-switch (default + enabled; set to `0` to skip the recall lookup entirely). The drawer's new "Memory used" + section lists these hits for a2a tasks and is omitted whenever there are none. Known + limitation: recall only resolves under the keyless posture — a keyed caller's task owner is a + SHA-256 prefix of the API key, while memory rows are keyed by the database api-key id, and no + hash→id lookup exists today, so the hit list stays empty for keyed callers. The recorded hits + are also kept out of the task's own `input` (and therefore out of the persisted input and of + the "Repeat" request body), so repeating a task never re-sends the previous run's memory + snippets. diff --git a/docs/openapi.yaml b/docs/openapi.yaml index a8d679f8dc..2f83d88080 100644 --- a/docs/openapi.yaml +++ b/docs/openapi.yaml @@ -8953,6 +8953,14 @@ paths: responses: "200": description: OK + /api/conductor/tasks: + post: + tags: + - Conductor + summary: "POST conductor › tasks" + responses: + "201": + description: Created /api/conductor/tasks/{id}: get: tags: diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index 2dded98167..ace16cf6c9 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -502,6 +502,7 @@ detection above). | `OMNIROUTE_API_KEY_ID` | _(unset)_ | `open-sse/mcp-server/audit.ts` | Key ID for MCP audit log attribution. | | `ROUTER_API_KEY` | _(unset)_ | Legacy | Legacy alias for `OMNIROUTE_API_KEY`. | | `OMNIROUTE_A2A_HISTORY_RETENTION_DAYS` | `30` | `src/lib/a2a/taskManager.ts` | Days of A2A task history kept in the local database before the daily purge deletes a row. Unset, non-numeric, or `<= 0` falls back to `30`. | +| `OMNIROUTE_A2A_MEMORY_HITS` | `1` | `src/lib/a2a/taskExecution.ts` | Kill-switch for the A2A memory-hits observability feature. Set to `0` to skip the memory recall lookup for a task entirely; any other value (including unset) keeps it enabled. | | `OMNIROUTE_ISSUE_AGENT_ENABLED` | `false` | `src/app/api/issue-agent/runs/route.ts` | Enables the offline/local Issue Agent recorded-triage endpoint. Leave disabled unless explicitly running local recorded-triage workflows. | | `OMNIROUTE_ISSUE_AGENT_TIMEOUT_MS` | _(unset)_ | `src/lib/issueAgent/execution.ts` | Timeout (ms) for a single Issue Agent recorded-triage run. Clamped to an internal maximum; falls back to the built-in default when unset or invalid. | | `OMNIROUTE_CONTEXT` | _(active context)_ | `bin/cli/program.mjs`, `bin/cli/api.mjs` | CLI remote-mode context/profile for `omniroute` commands; overrides the active context in the local contexts store. Equivalent to `--context `. | diff --git a/src/app/(dashboard)/dashboard/orchestration/drawer/OrchestrationDrawer.tsx b/src/app/(dashboard)/dashboard/orchestration/drawer/OrchestrationDrawer.tsx index 39f5edcef0..1480e4e3d5 100644 --- a/src/app/(dashboard)/dashboard/orchestration/drawer/OrchestrationDrawer.tsx +++ b/src/app/(dashboard)/dashboard/orchestration/drawer/OrchestrationDrawer.tsx @@ -9,6 +9,7 @@ import type { CloudAgentTask } from "@/lib/cloudAgent/types"; import type { A2ATask } from "@/lib/a2a/taskManager"; const TOAST_MS = 2500; +const REPEAT_CONFIRM_MS = 3000; /** Timeline normalized by source — the same data the Timeline component displays. */ function normalizedTimeline(node: OrchNode, detail: unknown): unknown { @@ -256,27 +257,150 @@ function DrawerResult({ ); } -/** Approve/cancel action buttons — omitted when neither action is available. */ +/** + * "Memory used" section (Task D4, PR-C): lists the memories consulted for an a2a task's + * last user message (`task.metadata.memoryHits`, written by `collectMemoryHits` in + * `src/lib/a2a/taskExecution.ts` — observability only, never injected into a skill's + * behavior). `metadata` is caller-supplied and unvalidated end to end + * (`src/app/a2a/route.ts` passes `params?.metadata` straight through, and + * `collectMemoryHits` only overwrites it when it finds hits), so `memoryHits` cannot be + * trusted at the `A2ATask` type — a malicious/buggy A2A client could post + * `metadata: { memoryHits: "boom" }` (a string's `.length` is truthy, so a plain + * `!hits || hits.length === 0` guard would let it through) or an array containing + * malformed entries. Every entry is validated defensively before it is ever rendered, so a + * bad payload silently drops that entry instead of crashing the drawer — and the check + * covers ALL FOUR fields, not just `id`: `type`, `key` and `snippet` are rendered as React + * children, so `{ id: "x", key: { a: 1 } }` (a validated id next to an object field) would + * throw "Objects are not valid as a React child" and take the whole drawer down. + */ +const MEMORY_HIT_FIELDS = ["id", "key", "type", "snippet"] as const; + +function DrawerMemory({ a2a, t }: { a2a: A2ATask | null; t: Translate }) { + const raw = a2a?.metadata?.memoryHits; + const hits = (Array.isArray(raw) ? raw : []).filter( + (h): h is { id: string; key: string; type: string; snippet: string } => + !!h && + typeof h === "object" && + MEMORY_HIT_FIELDS.every((f) => typeof (h as Record)[f] === "string") + ); + if (hits.length === 0) return null; + return ( +
+
    + {hits.map((h) => ( +
  • + {h.type} + {h.key} +
    {h.snippet}
    +
  • + ))} +
+
+ ); +} + +/** + * Two-click confirm for a single non-idempotent action — deliberately NOT + * `window.confirm`, since modal dialogs block browser automation. The first click + * arms `confirming` for `REPEAT_CONFIRM_MS`; a second click within that window runs + * `onConfirm`. The timer is cleared before it can fire again (a stale timeout must + * never flip an already-fired confirmation back) and on unmount. + */ +function useTwoClickConfirm(onConfirm: () => void) { + const [confirming, setConfirming] = useState(false); + const timerRef = useRef | null>(null); + + const clearTimer = () => { + if (timerRef.current) clearTimeout(timerRef.current); + timerRef.current = null; + }; + + const onClick = () => { + if (confirming) { + clearTimer(); + setConfirming(false); + onConfirm(); + return; + } + setConfirming(true); + timerRef.current = setTimeout(() => setConfirming(false), REPEAT_CONFIRM_MS); + }; + + useEffect(() => clearTimer, []); + + return { confirming, onClick }; +} + +/** + * Repeat button: two-click confirm, disabled + tooltip when the input isn't recoverable. + * `canRepeat` already folds in `!busy` (so the button is disabled while a repeat POST is + * in flight), but the tooltip must not claim the input is unrecoverable in that case — a + * successful repeat is legitimately busy, not unavailable. `busy` is threaded through + * separately so the title can tell the two apart: silent (no title) while busy, the real + * `repeatUnavailable` message only when the input truly can't be recovered. + */ +function RepeatButton({ + canRepeat, + busy, + repeat, + onActionDone, + onToast, + t, +}: { + canRepeat: boolean; + busy: boolean; + repeat: () => Promise; + onActionDone: () => void; + onToast: (text: string) => void; + t: Translate; +}) { + const { confirming, onClick } = useTwoClickConfirm(() => { + void (async () => { + if (await repeat()) { + onActionDone(); + onToast(t("repeatDone")); + } + })(); + }); + return ( + + ); +} + +/** Approve/cancel/repeat action buttons — omitted when none of them apply. */ function DrawerActions({ canApprove, canCancel, + showRepeat, + canRepeat, busy, approve, cancel, + repeat, onActionDone, onToast, t, }: { canApprove: boolean; canCancel: boolean; + showRepeat: boolean; + canRepeat: boolean; busy: boolean; approve: () => Promise; cancel: () => Promise; + repeat: () => Promise; onActionDone: () => void; onToast: (text: string) => void; t: Translate; }) { - if (!canApprove && !canCancel) return null; + if (!canApprove && !canCancel && !showRepeat) return null; const run = async (fn: () => Promise) => { if (await fn()) { onActionDone(); @@ -304,6 +428,16 @@ function DrawerActions({ {t("actionCancel")} )} + {showRepeat && ( + + )} ); @@ -357,14 +491,27 @@ export function OrchestrationDrawer({ onActionDone: () => void; }) { const t = useTranslations("orchestration"); - const { detail, isLoading, busy, error, errorKind, canApprove, canCancel, approve, cancel } = - useDrawerDetail(node); + const { + detail, + isLoading, + busy, + error, + errorKind, + canApprove, + canCancel, + canRepeat, + approve, + cancel, + repeat, + } = useDrawerDetail(node); useCloseOnEscape(node, onClose); const { toast, showToast } = useDrawerToast(); if (!node) return null; const state = node.state ?? "queued"; const { ca, a2a } = narrowDetail(node, detail); + const showRepeat = + node.source === "cloud-agent" || node.source === "a2a" || node.source === "conductor"; return ( <> @@ -397,12 +544,16 @@ export function OrchestrationDrawer({ + ` errors and AbortError pass through verbatim, everything else collapses to a generic string. +// Client-safe stand-in for sanitizeErrorMessage (server-only, breaks the client bundle — #10692): only our own `HTTP ` / `RPC ` errors and AbortError pass through verbatim, everything else collapses to a generic string. `RPC ` carries the JSON-RPC error CODE only — never the upstream `error.message`, which is attacker/upstream-controlled text (Hard Rule #12). function toSafeErrorText(err: unknown): string { if (err instanceof Error) { if (/^HTTP \d{3}$/.test(err.message)) return err.message; + if (/^RPC -?\d{1,6}$/.test(err.message)) return err.message; if (err.name === "AbortError") return "Request cancelled"; } return "Request failed"; @@ -55,6 +59,115 @@ function routeFor(node: OrchNode): SourceRoute { return { detailUrl: null, cancelReq: null, approveReq: null }; // runners/overflow: raw only } +/** + * Builds the POST request that recreates a task with the same input, from the LOADED + * DETAIL — never from `node` (the node only carries display fields, not the full + * original request). Returns `null` when the original input cannot be recovered, so + * the caller can render the "Repeat" action disabled instead of firing a bad request. + * Contracts, verified against the live routes (not assumed) — the null-guard requires + * EVERY field the target route treats as mandatory, not merely one of them (a partially + * recoverable detail is not recoverable: a POST missing one required field 400s, which is + * an enabled button that cannot work): + * - cloud-agent → `POST /api/v1/agents/tasks`, `CreateCloudAgentTaskSchema` shape + * (`src/lib/cloudAgent/types.ts`) — `providerId`, `prompt` and `source` are all + * required there; `options` is optional. + * - a2a → `POST /a2a`, JSON-RPC `message/send` (`src/app/a2a/route.ts`) — only + * `messages` is required (`skill` defaults to `"smart-routing"`, `metadata` is + * optional), so that is the only field guarded here. + * - conductor → `POST /api/conductor/tasks` (D1, `src/app/api/conductor/tasks/route.ts`) + * — `repoUrl` and `prompt` are both `z.string().min(1)` (required); `ConductorTaskDetail` + * leaves `repo`/`prompt` independently nullable, so either one missing must null out + * the whole request. + */ +/** + * Strips `memoryHits` from the metadata a repeat re-sends. `metadata.memoryHits` is + * OBSERVABILITY written by the previous run (`src/lib/a2a/taskExecution.ts`) — never + * caller input — so echoing it back would make the new task be born carrying the old + * run's memory snippets, and would keep showing them in the drawer even with the + * `OMNIROUTE_A2A_MEMORY_HITS=0` kill-switch on. `taskManager.createTask` no longer aliases + * `metadata` into `input`, but historical tasks persisted before that fix still carry the + * hits inside `input.metadata`, so the repeat path must drop them too. + * Returns `undefined` for a missing/non-object metadata so the JSON body omits the field + * entirely (the route treats `params.metadata` as optional). + */ +function withoutMemoryHits(metadata: unknown): Record | undefined { + if (!metadata || typeof metadata !== "object" || Array.isArray(metadata)) return undefined; + const rest = { ...(metadata as Record) }; + delete rest.memoryHits; + return rest; +} + +/** Builds the JSON-body `RequestInit` shared by every `repeatReqFor*` source builder below. */ +function postJson(body: unknown): RequestInit { + return { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify(body), + }; +} + +/** cloud-agent repeat builder — `POST /api/v1/agents/tasks`, `CreateCloudAgentTaskSchema` shape. */ +function repeatReqForCloudAgent(detail: unknown): { url: string; init: RequestInit } | null { + const d = detail as CloudAgentTask | null; + if (!d?.providerId || !d?.prompt || !d?.source) return null; + return { + url: "/api/v1/agents/tasks", + init: postJson({ + providerId: d.providerId, + prompt: d.prompt, + source: d.source, + options: d.options, + }), + }; +} + +/** a2a repeat builder — `POST /a2a`, JSON-RPC `message/send` from `detail.input`. */ +function repeatReqForA2a( + nodeId: string, + detail: unknown +): { url: string; init: RequestInit } | null { + const d = detail as A2ATask | null; + if (!d?.input?.messages?.length) return null; + return { + url: "/a2a", + init: postJson({ + jsonrpc: "2.0", + id: nodeId, + method: "message/send", + params: { + skill: d.input.skill, + messages: d.input.messages, + metadata: withoutMemoryHits(d.input.metadata), + }, + }), + }; +} + +/** conductor repeat builder — `POST /api/conductor/tasks` (D1 task-creation route). */ +function repeatReqForConductor(detail: unknown): { url: string; init: RequestInit } | null { + const d = detail as ConductorTaskDetail | null; + if (!d?.repo || !d?.prompt) return null; + return { + url: "/api/conductor/tasks", + init: postJson({ + repoUrl: d.repo, + prompt: d.prompt, + baseRef: d.base_ref ?? undefined, + mode: d.mode, + }), + }; +} + +export function repeatReqFor( + node: OrchNode, + detail: unknown +): { url: string; init: RequestInit } | null { + if (node.id.startsWith("cloud-agent:")) return repeatReqForCloudAgent(detail); + if (node.id.startsWith("a2a:")) return repeatReqForA2a(node.id, detail); + if (node.id.startsWith("conductor:task:")) return repeatReqForConductor(detail); + return null; +} + /** * Unwraps a task-detail GET response to the actual task payload. Each source's * route has its own envelope — verified against the live handlers, not assumed: @@ -138,6 +251,26 @@ function useFetchDetail( }, [node?.id]); } +/** + * A JSON-RPC endpoint can report a failure with an HTTP 200: `/a2a`'s `jsonRpcError()` + * only maps a few codes to 4xx/5xx and defaults to `status: 200` + * (`src/app/a2a/route.ts`). `res.ok` alone would then render the success toast for a run + * that never happened, so the `/a2a` action also inspects the envelope. Only the numeric + * `error.code` is surfaced (`RPC `) — never the upstream `error.message`. + */ +async function jsonRpcErrorCode(res: { + json?: () => Promise; +}): Promise { + try { + const body = (await res.json?.()) as { error?: { code?: unknown } } | undefined; + const code = body?.error?.code; + return typeof code === "number" ? code : body?.error ? -32603 : undefined; + } catch { + // A non-JSON / already-consumed body is not evidence of failure — the status stands. + return undefined; + } +} + async function performAction( req: { url: string; init: RequestInit } | null, setActionError: (text: string) => void @@ -146,6 +279,10 @@ async function performAction( try { const res = await fetch(req.url, req.init); if (!res.ok) throw new Error(`HTTP ${res.status}`); + if (req.url === "/a2a") { + const code = await jsonRpcErrorCode(res); + if (code !== undefined) throw new Error(`RPC ${code}`); + } return true; } catch (err) { setActionError(toSafeErrorText(err)); @@ -167,6 +304,7 @@ export function useDrawerDetail(node: OrchNode | null) { useFetchDetail(node, route, setDetail, setDetailError, setIsLoading); const { canApprove, canCancel } = deriveActionAvailability(route, node); + const repeatReq = node ? repeatReqFor(node, detail) : null; const runAction = async (req: { url: string; init: RequestInit } | null): Promise => { if (busy) return false; @@ -186,7 +324,9 @@ export function useDrawerDetail(node: OrchNode | null) { errorKind: error?.kind ?? null, canApprove, canCancel, + canRepeat: !!repeatReq && !busy, approve: () => runAction(route?.approveReq ?? null), cancel: () => runAction(route?.cancelReq ?? null), + repeat: () => runAction(repeatReq), }; } diff --git a/src/app/(dashboard)/dashboard/orchestration/tabs/HistoryTab.tsx b/src/app/(dashboard)/dashboard/orchestration/tabs/HistoryTab.tsx index bd17c46e7f..d8939cf7f6 100644 --- a/src/app/(dashboard)/dashboard/orchestration/tabs/HistoryTab.tsx +++ b/src/app/(dashboard)/dashboard/orchestration/tabs/HistoryTab.tsx @@ -191,9 +191,7 @@ function PresetButtons({ type="button" aria-pressed={preset === p} className={`px-2 py-1 text-xs rounded border ${ - preset === p - ? "border-primary bg-primary/10 font-medium" - : "border-border text-muted" + preset === p ? "border-primary bg-primary/10 font-medium" : "border-border text-muted" }`} onClick={() => onSelect(p)} > @@ -265,9 +263,7 @@ function HistoryGridTable({ className="text-left px-2 py-1 sticky left-0 bg-surface whitespace-nowrap font-normal" > {row.identity}{" "} - - {t(SOURCE_KEY[row.source])} - + {t(SOURCE_KEY[row.source])} {row.cells.map((cell, i) => ( @@ -350,10 +346,22 @@ export function HistoryTab() { )} + {/* `onActionDone` must NOT close the drawer: the drawer renders its own success toast + right after calling it, so unmounting here threw the confirmation away and the + operator saw a repeat/cancel silently do nothing. Re-sampling `nowMs` instead + keeps the drawer mounted (the toast lands) and refreshes the grid through the new + range — the same "refetch, don't close" contract `OrchestrationPageClient` uses. + `Date.now()` is sampled inside a real event-driven callback, never during render + (see the `nowMs` note above), and the updater is pure — it only picks the larger of + the sampled clock and `prev + 1`, so the range always changes (and the refetch + always happens) even when two samples land in the same millisecond. */} setSelected(null)} - onActionDone={() => setSelected(null)} + onActionDone={() => { + const sampled = Date.now(); + setNowMs((prev) => (sampled > prev ? sampled : prev + 1)); + }} /> ); diff --git a/src/app/api/conductor/tasks/route.ts b/src/app/api/conductor/tasks/route.ts new file mode 100644 index 0000000000..02168a69c8 --- /dev/null +++ b/src/app/api/conductor/tasks/route.ts @@ -0,0 +1,66 @@ +/** + * POST /api/conductor/tasks — creates a task on the Conductor hub (Orchestration Canvas + * Fase 2, "Repeat" action on the drawer). Thin creation route: validate → auth → delegate + * to `createConductorTask`. A hub refusal comes back as the hub's status with a sanitized + * body (never the raw upstream body — Hard Rule #12). + */ + +import { NextResponse } from "next/server"; +import { z } from "zod"; + +import { createErrorResponse } from "@/lib/api/errorResponse"; +import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; +import { createConductorTask } from "@/lib/conductor/hubProxy"; + +/** + * Clamps a hub status before it is used as OUR response status. `createConductorTask` + * mirrors whatever the hub answered, and `Response.json()` throws a `RangeError` for any + * status outside 200-599 — a hub (or a stubbed fetch) answering `0`/`600` would turn a + * hub refusal into a 500 from an unhandled throw. A 3xx/2xx reaching this branch is + * equally meaningless as an error status, so anything outside 400-599 becomes 502 + * (Bad Gateway — the honest description of "the upstream hub answered something we + * cannot forward"). + */ +function clampErrorStatus(status: unknown): number { + const s = Number(status); + return Number.isInteger(s) && s >= 400 && s <= 599 ? s : 502; +} + +const createTaskSchema = z.object({ + repoUrl: z.string().min(1), + prompt: z.string().min(1), + baseRef: z.string().optional(), + mode: z.string().optional(), + cli: z.string().optional(), + model: z.string().optional(), +}); + +export async function POST(request: Request) { + const authError = await requireManagementAuth(request); + if (authError) return authError; + + let rawBody: unknown; + try { + rawBody = await request.json(); + } catch { + return createErrorResponse({ status: 400, message: "Invalid JSON body" }); + } + + const parsed = createTaskSchema.safeParse(rawBody); + if (!parsed.success) { + return createErrorResponse({ + status: 400, + message: "Invalid request body", + details: parsed.error.flatten(), + }); + } + + const result = await createConductorTask(parsed.data); + if (!result.ok || !result.task_id) { + return createErrorResponse({ + status: clampErrorStatus(result.status), + message: `Conductor hub refused the task creation (HTTP ${result.status})`, + }); + } + return NextResponse.json({ task_id: result.task_id }, { status: 201 }); +} diff --git a/src/i18n/messages/ar.json b/src/i18n/messages/ar.json index fdcaf03dba..d33a7633bf 100644 --- a/src/i18n/messages/ar.json +++ b/src/i18n/messages/ar.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "الجدول الزمني", "drawerMetrics": "المقاييس", "drawerResult": "النتيجة", + "drawerMemory": "الذاكرة المستخدمة", "drawerActions": "الإجراءات", "drawerClose": "إغلاق", "copyTrace": "نسخ أثر التتبع (JSON)", @@ -13997,7 +13998,11 @@ "actionDone": "تم تطبيق الإجراء", "actionFailed": "فشل الإجراء: {error}", "detailFailed": "فشل تحميل التفاصيل: {error}", - "mirroredInA2A": "معكوس في A2A" + "mirroredInA2A": "معكوس في A2A", + "actionRepeat": "تكرار", + "repeatConfirm": "انقر مرة أخرى للتأكيد", + "repeatDone": "تم إنشاء تشغيل جديد", + "repeatUnavailable": "المُدخل الأصلي غير متاح" }, "cliproxyProviderExposure": { "title": "تعرض المزود", diff --git a/src/i18n/messages/az.json b/src/i18n/messages/az.json index 1dd62f5f51..a50b88762f 100644 --- a/src/i18n/messages/az.json +++ b/src/i18n/messages/az.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Zaman xətti", "drawerMetrics": "Metriklər", "drawerResult": "Nəticə", + "drawerMemory": "İstifadə olunan yaddaş", "drawerActions": "Əməliyyatlar", "drawerClose": "Bağla", "copyTrace": "İzləmə JSON-unu kopyala", @@ -13997,7 +13998,11 @@ "actionDone": "Əməliyyat tətbiq edildi", "actionFailed": "Əməliyyat uğursuz oldu: {error}", "detailFailed": "Detallar yüklənmədi: {error}", - "mirroredInA2A": "A2A-da əks olunub" + "mirroredInA2A": "A2A-da əks olunub", + "actionRepeat": "Təkrarla", + "repeatConfirm": "Təsdiqləmək üçün yenidən klikləyin", + "repeatDone": "Yeni icra yaradıldı", + "repeatUnavailable": "Orijinal giriş mövcud deyil" }, "cliproxyProviderExposure": { "title": "Təchizatçı Məlumatı", diff --git a/src/i18n/messages/bg.json b/src/i18n/messages/bg.json index a686606a62..8086f96fc5 100644 --- a/src/i18n/messages/bg.json +++ b/src/i18n/messages/bg.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Хронология", "drawerMetrics": "Метрики", "drawerResult": "Резултат", + "drawerMemory": "Използвана памет", "drawerActions": "Действия", "drawerClose": "Затваряне", "copyTrace": "Копиране на трасето (JSON)", @@ -13997,7 +13998,11 @@ "actionDone": "Действието е приложено", "actionFailed": "Действието се провали: {error}", "detailFailed": "Неуспешно зареждане на детайлите: {error}", - "mirroredInA2A": "Отразено в A2A" + "mirroredInA2A": "Отразено в A2A", + "actionRepeat": "Повтори", + "repeatConfirm": "Кликнете отново, за да потвърдите", + "repeatDone": "Създадено е ново изпълнение", + "repeatUnavailable": "Оригиналният вход не е наличен" }, "cliproxyProviderExposure": { "title": "Излагане на доставчика", diff --git a/src/i18n/messages/bn.json b/src/i18n/messages/bn.json index e7c9c57add..507007dc7b 100644 --- a/src/i18n/messages/bn.json +++ b/src/i18n/messages/bn.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "টাইমলাইন", "drawerMetrics": "মেট্রিক্স", "drawerResult": "ফলাফল", + "drawerMemory": "ব্যবহৃত মেমরি", "drawerActions": "কার্যক্রম", "drawerClose": "বন্ধ করুন", "copyTrace": "ট্রেস JSON কপি করুন", @@ -13997,7 +13998,11 @@ "actionDone": "কার্যক্রম প্রয়োগ করা হয়েছে", "actionFailed": "কার্যক্রম ব্যর্থ হয়েছে: {error}", "detailFailed": "বিবরণ লোড করতে ব্যর্থ: {error}", - "mirroredInA2A": "A2A-তে প্রতিফলিত" + "mirroredInA2A": "A2A-তে প্রতিফলিত", + "actionRepeat": "পুনরাবৃত্তি করুন", + "repeatConfirm": "নিশ্চিত করতে আবার ক্লিক করুন", + "repeatDone": "নতুন রান তৈরি হয়েছে", + "repeatUnavailable": "মূল ইনপুট উপলব্ধ নেই" }, "cliproxyProviderExposure": { "title": "প্রদানকারী এক্সপোজার", diff --git a/src/i18n/messages/cs.json b/src/i18n/messages/cs.json index cde44ef0ca..f68aa4b51d 100644 --- a/src/i18n/messages/cs.json +++ b/src/i18n/messages/cs.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Časová osa", "drawerMetrics": "Metriky", "drawerResult": "Výsledek", + "drawerMemory": "Použitá paměť", "drawerActions": "Akce", "drawerClose": "Zavřít", "copyTrace": "Kopírovat trasování (JSON)", @@ -13997,7 +13998,11 @@ "actionDone": "Akce provedena", "actionFailed": "Akce selhala: {error}", "detailFailed": "Nepodařilo se načíst podrobnosti: {error}", - "mirroredInA2A": "Zrcadleno v A2A" + "mirroredInA2A": "Zrcadleno v A2A", + "actionRepeat": "Opakovat", + "repeatConfirm": "Kliknutím znovu potvrdíte", + "repeatDone": "Vytvořeno nové spuštění", + "repeatUnavailable": "Původní vstup není k dispozici" }, "cliproxyProviderExposure": { "title": "Expozice Poskytovatele", diff --git a/src/i18n/messages/da.json b/src/i18n/messages/da.json index d763f93dd6..f326e8f313 100644 --- a/src/i18n/messages/da.json +++ b/src/i18n/messages/da.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Tidslinje", "drawerMetrics": "Målinger", "drawerResult": "Resultat", + "drawerMemory": "Anvendt hukommelse", "drawerActions": "Handlinger", "drawerClose": "Luk", "copyTrace": "Kopiér sporing (JSON)", @@ -13997,7 +13998,11 @@ "actionDone": "Handling udført", "actionFailed": "Handling mislykkedes: {error}", "detailFailed": "Kunne ikke indlæse detaljer: {error}", - "mirroredInA2A": "Spejlet i A2A" + "mirroredInA2A": "Spejlet i A2A", + "actionRepeat": "Gentag", + "repeatConfirm": "Klik igen for at bekræfte", + "repeatDone": "Ny kørsel oprettet", + "repeatUnavailable": "Oprindeligt input ikke tilgængeligt" }, "cliproxyProviderExposure": { "title": "Udbyder Eksponering", diff --git a/src/i18n/messages/de.json b/src/i18n/messages/de.json index ae2cd0724f..4b51536213 100644 --- a/src/i18n/messages/de.json +++ b/src/i18n/messages/de.json @@ -13995,6 +13995,7 @@ "drawerTimeline": "Zeitachse", "drawerMetrics": "Metriken", "drawerResult": "Ergebnis", + "drawerMemory": "Genutzte Erinnerungen", "drawerActions": "Aktionen", "drawerClose": "Schließen", "copyTrace": "Trace-JSON kopieren", @@ -14004,7 +14005,11 @@ "actionDone": "Aktion angewendet", "actionFailed": "Aktion fehlgeschlagen: {error}", "detailFailed": "Details konnten nicht geladen werden: {error}", - "mirroredInA2A": "In A2A gespiegelt" + "mirroredInA2A": "In A2A gespiegelt", + "actionRepeat": "Wiederholen", + "repeatConfirm": "Zum Bestätigen erneut klicken", + "repeatDone": "Neuer Lauf erstellt", + "repeatUnavailable": "Ursprüngliche Eingabe nicht verfügbar" }, "cliproxyProviderExposure": { "title": "Anbieterexposition", diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index 7c9d7439d4..292d76c86d 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -13995,16 +13995,21 @@ "drawerTimeline": "Timeline", "drawerMetrics": "Metrics", "drawerResult": "Result", + "drawerMemory": "Memory used", "drawerActions": "Actions", "drawerClose": "Close", "copyTrace": "Copy trace JSON", "actionApprove": "Approve plan", "actionCancel": "Cancel", "actionSeeInGraph": "See in graph", + "actionRepeat": "Repeat", "actionDone": "Action applied", "actionFailed": "Action failed: {error}", "detailFailed": "Failed to load details: {error}", - "mirroredInA2A": "Mirrored in A2A" + "mirroredInA2A": "Mirrored in A2A", + "repeatConfirm": "Click again to confirm", + "repeatDone": "New run created", + "repeatUnavailable": "Original input not available" }, "cliproxyProviderExposure": { "title": "Provider Exposure", diff --git a/src/i18n/messages/es.json b/src/i18n/messages/es.json index 4ce646e653..b1fbcc4fe6 100644 --- a/src/i18n/messages/es.json +++ b/src/i18n/messages/es.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Cronología", "drawerMetrics": "Métricas", "drawerResult": "Resultado", + "drawerMemory": "Memoria utilizada", "drawerActions": "Acciones", "drawerClose": "Cerrar", "copyTrace": "Copiar traza JSON", @@ -13997,7 +13998,11 @@ "actionDone": "Acción aplicada", "actionFailed": "Error en la acción: {error}", "detailFailed": "No se pudieron cargar los detalles: {error}", - "mirroredInA2A": "Reflejado en A2A" + "mirroredInA2A": "Reflejado en A2A", + "actionRepeat": "Repetir", + "repeatConfirm": "Hacer clic de nuevo para confirmar", + "repeatDone": "Nueva ejecución creada", + "repeatUnavailable": "Entrada original no disponible" }, "cliproxyProviderExposure": { "title": "Exposición del Proveedor", diff --git a/src/i18n/messages/fa.json b/src/i18n/messages/fa.json index 5c5112f3cc..b57be44976 100644 --- a/src/i18n/messages/fa.json +++ b/src/i18n/messages/fa.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "جدول زمانی", "drawerMetrics": "معیارها", "drawerResult": "نتیجه", + "drawerMemory": "حافظه استفاده‌شده", "drawerActions": "اقدامات", "drawerClose": "بستن", "copyTrace": "کپی ردیابی JSON", @@ -13997,7 +13998,11 @@ "actionDone": "اقدام اعمال شد", "actionFailed": "اقدام ناموفق بود: {error}", "detailFailed": "بارگذاری جزئیات ناموفق بود: {error}", - "mirroredInA2A": "بازتاب‌یافته در A2A" + "mirroredInA2A": "بازتاب‌یافته در A2A", + "actionRepeat": "تکرار", + "repeatConfirm": "برای تأیید دوباره کلیک کنید", + "repeatDone": "اجرای جدید ایجاد شد", + "repeatUnavailable": "ورودی اصلی در دسترس نیست" }, "cliproxyProviderExposure": { "title": "قرار گرفتن در معرض ارائه‌دهنده", diff --git a/src/i18n/messages/fi.json b/src/i18n/messages/fi.json index 9124a16634..0b220d98b7 100644 --- a/src/i18n/messages/fi.json +++ b/src/i18n/messages/fi.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Aikajana", "drawerMetrics": "Mittarit", "drawerResult": "Tulos", + "drawerMemory": "Käytetty muisti", "drawerActions": "Toiminnot", "drawerClose": "Sulje", "copyTrace": "Kopioi jäljitys-JSON", @@ -13997,7 +13998,11 @@ "actionDone": "Toiminto suoritettu", "actionFailed": "Toiminto epäonnistui: {error}", "detailFailed": "Tietojen lataus epäonnistui: {error}", - "mirroredInA2A": "Peilattu A2A:ssa" + "mirroredInA2A": "Peilattu A2A:ssa", + "actionRepeat": "Toista", + "repeatConfirm": "Vahvista napsauttamalla uudelleen", + "repeatDone": "Uusi ajo luotu", + "repeatUnavailable": "Alkuperäistä syötettä ei ole saatavilla" }, "cliproxyProviderExposure": { "title": "Palveluntarjoajan Altistus", diff --git a/src/i18n/messages/fr.json b/src/i18n/messages/fr.json index 34577a4dbb..59cae55660 100644 --- a/src/i18n/messages/fr.json +++ b/src/i18n/messages/fr.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Chronologie", "drawerMetrics": "Métriques", "drawerResult": "Résultat", + "drawerMemory": "Mémoire utilisée", "drawerActions": "Actions", "drawerClose": "Fermer", "copyTrace": "Copier la trace JSON", @@ -13997,7 +13998,11 @@ "actionDone": "Action appliquée", "actionFailed": "Échec de l'action : {error}", "detailFailed": "Échec du chargement des détails : {error}", - "mirroredInA2A": "Reflété dans A2A" + "mirroredInA2A": "Reflété dans A2A", + "actionRepeat": "Répéter", + "repeatConfirm": "Cliquez à nouveau pour confirmer", + "repeatDone": "Nouvelle exécution créée", + "repeatUnavailable": "Entrée d'origine non disponible" }, "cliproxyProviderExposure": { "title": "Exposition du Fournisseur", diff --git a/src/i18n/messages/gu.json b/src/i18n/messages/gu.json index 1f278c1f98..152237a606 100644 --- a/src/i18n/messages/gu.json +++ b/src/i18n/messages/gu.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "સમયરેખા", "drawerMetrics": "મેટ્રિક્સ", "drawerResult": "પરિણામ", + "drawerMemory": "વપરાયેલ મેમરી", "drawerActions": "ક્રિયાઓ", "drawerClose": "બંધ કરો", "copyTrace": "ટ્રેસ JSON કૉપિ કરો", @@ -13997,7 +13998,11 @@ "actionDone": "ક્રિયા લાગુ કરાઈ", "actionFailed": "ક્રિયા નિષ્ફળ: {error}", "detailFailed": "વિગતો લોડ કરવામાં નિષ્ફળ: {error}", - "mirroredInA2A": "A2A માં પ્રતિબિંબિત" + "mirroredInA2A": "A2A માં પ્રતિબિંબિત", + "actionRepeat": "પુનરાવર્તન કરો", + "repeatConfirm": "પુષ્ટિ કરવા માટે ફરીથી ક્લિક કરો", + "repeatDone": "નવું રન બનાવવામાં આવ્યું", + "repeatUnavailable": "મૂળ ઇનપુટ ઉપલબ્ધ નથી" }, "cliproxyProviderExposure": { "title": "પ્રદાતા એક્સપોઝર", diff --git a/src/i18n/messages/he.json b/src/i18n/messages/he.json index e720fe27be..7f45b461c0 100644 --- a/src/i18n/messages/he.json +++ b/src/i18n/messages/he.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "ציר זמן", "drawerMetrics": "מדדים", "drawerResult": "תוצאה", + "drawerMemory": "זיכרון בשימוש", "drawerActions": "פעולות", "drawerClose": "סגירה", "copyTrace": "העתקת מעקב JSON", @@ -13997,7 +13998,11 @@ "actionDone": "הפעולה בוצעה", "actionFailed": "הפעולה נכשלה: {error}", "detailFailed": "טעינת הפרטים נכשלה: {error}", - "mirroredInA2A": "משוקף ב-A2A" + "mirroredInA2A": "משוקף ב-A2A", + "actionRepeat": "הרץ שוב", + "repeatConfirm": "לחץ שוב כדי לאשר", + "repeatDone": "נוצרה הרצה חדשה", + "repeatUnavailable": "הקלט המקורי אינו זמין" }, "cliproxyProviderExposure": { "title": "חשיפת ספק", diff --git a/src/i18n/messages/hi.json b/src/i18n/messages/hi.json index 1bd1536902..3867eeb43a 100644 --- a/src/i18n/messages/hi.json +++ b/src/i18n/messages/hi.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "समयरेखा", "drawerMetrics": "मेट्रिक्स", "drawerResult": "परिणाम", + "drawerMemory": "उपयोग की गई मेमोरी", "drawerActions": "कार्रवाइयां", "drawerClose": "बंद करें", "copyTrace": "ट्रेस JSON कॉपी करें", @@ -13997,7 +13998,11 @@ "actionDone": "कार्रवाई लागू की गई", "actionFailed": "कार्रवाई विफल: {error}", "detailFailed": "विवरण लोड करने में विफल: {error}", - "mirroredInA2A": "A2A में प्रतिबिंबित" + "mirroredInA2A": "A2A में प्रतिबिंबित", + "actionRepeat": "दोहराएं", + "repeatConfirm": "पुष्टि के लिए फिर से क्लिक करें", + "repeatDone": "नया रन बनाया गया", + "repeatUnavailable": "मूल इनपुट उपलब्ध नहीं है" }, "cliproxyProviderExposure": { "title": "प्रदाता एक्सपोजर", diff --git a/src/i18n/messages/hu.json b/src/i18n/messages/hu.json index 66459aa3af..5334482837 100644 --- a/src/i18n/messages/hu.json +++ b/src/i18n/messages/hu.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Idővonal", "drawerMetrics": "Metrikák", "drawerResult": "Eredmény", + "drawerMemory": "Felhasznált memória", "drawerActions": "Műveletek", "drawerClose": "Bezárás", "copyTrace": "Nyomkövetési JSON másolása", @@ -13997,7 +13998,11 @@ "actionDone": "Művelet végrehajtva", "actionFailed": "A művelet sikertelen: {error}", "detailFailed": "A részletek betöltése sikertelen: {error}", - "mirroredInA2A": "Tükrözve az A2A-ban" + "mirroredInA2A": "Tükrözve az A2A-ban", + "actionRepeat": "Ismétlés", + "repeatConfirm": "Kattintson újra a megerősítéshez", + "repeatDone": "Új futás létrehozva", + "repeatUnavailable": "Az eredeti bemenet nem érhető el" }, "cliproxyProviderExposure": { "title": "Szolgáltató Expozíció", diff --git a/src/i18n/messages/id.json b/src/i18n/messages/id.json index 3369790cfd..545e9a5a41 100644 --- a/src/i18n/messages/id.json +++ b/src/i18n/messages/id.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Linimasa", "drawerMetrics": "Metrik", "drawerResult": "Hasil", + "drawerMemory": "Memori yang digunakan", "drawerActions": "Tindakan", "drawerClose": "Tutup", "copyTrace": "Salin trace JSON", @@ -13997,7 +13998,11 @@ "actionDone": "Tindakan diterapkan", "actionFailed": "Tindakan gagal: {error}", "detailFailed": "Gagal memuat detail: {error}", - "mirroredInA2A": "Dicerminkan di A2A" + "mirroredInA2A": "Dicerminkan di A2A", + "actionRepeat": "Ulangi", + "repeatConfirm": "Klik lagi untuk konfirmasi", + "repeatDone": "Proses baru dibuat", + "repeatUnavailable": "Input asli tidak tersedia" }, "cliproxyProviderExposure": { "title": "Paparan Penyedia", diff --git a/src/i18n/messages/it.json b/src/i18n/messages/it.json index 4121406bc5..dc01def708 100644 --- a/src/i18n/messages/it.json +++ b/src/i18n/messages/it.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Cronologia", "drawerMetrics": "Metriche", "drawerResult": "Risultato", + "drawerMemory": "Memoria utilizzata", "drawerActions": "Azioni", "drawerClose": "Chiudi", "copyTrace": "Copia trace JSON", @@ -13997,7 +13998,11 @@ "actionDone": "Azione applicata", "actionFailed": "Azione non riuscita: {error}", "detailFailed": "Impossibile caricare i dettagli: {error}", - "mirroredInA2A": "Rispecchiato in A2A" + "mirroredInA2A": "Rispecchiato in A2A", + "actionRepeat": "Ripeti", + "repeatConfirm": "Clicca di nuovo per confermare", + "repeatDone": "Nuova esecuzione creata", + "repeatUnavailable": "Input originale non disponibile" }, "cliproxyProviderExposure": { "title": "Esposizione del Fornitore", diff --git a/src/i18n/messages/ja.json b/src/i18n/messages/ja.json index 9dad5adbaf..ba8c064462 100644 --- a/src/i18n/messages/ja.json +++ b/src/i18n/messages/ja.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "タイムライン", "drawerMetrics": "メトリクス", "drawerResult": "結果", + "drawerMemory": "使用したメモリ", "drawerActions": "アクション", "drawerClose": "閉じる", "copyTrace": "トレースJSONをコピー", @@ -13997,7 +13998,11 @@ "actionDone": "アクションを適用しました", "actionFailed": "アクションが失敗しました: {error}", "detailFailed": "詳細の読み込みに失敗しました: {error}", - "mirroredInA2A": "A2Aにミラーリング" + "mirroredInA2A": "A2Aにミラーリング", + "actionRepeat": "再実行", + "repeatConfirm": "確認するにはもう一度クリック", + "repeatDone": "新しい実行を作成しました", + "repeatUnavailable": "元の入力を利用できません" }, "cliproxyProviderExposure": { "title": "プロバイダーの露出", diff --git a/src/i18n/messages/ko.json b/src/i18n/messages/ko.json index 2791c958cd..ab4b4a67f7 100644 --- a/src/i18n/messages/ko.json +++ b/src/i18n/messages/ko.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "타임라인", "drawerMetrics": "지표", "drawerResult": "결과", + "drawerMemory": "사용된 메모리", "drawerActions": "작업", "drawerClose": "닫기", "copyTrace": "추적 JSON 복사", @@ -13997,7 +13998,11 @@ "actionDone": "작업이 적용됨", "actionFailed": "작업 실패: {error}", "detailFailed": "세부정보를 불러오지 못했습니다: {error}", - "mirroredInA2A": "A2A에 미러링됨" + "mirroredInA2A": "A2A에 미러링됨", + "actionRepeat": "반복", + "repeatConfirm": "확인하려면 다시 클릭하세요", + "repeatDone": "새 실행이 생성되었습니다", + "repeatUnavailable": "원본 입력을 사용할 수 없습니다" }, "cliproxyProviderExposure": { "title": "제공자 노출", diff --git a/src/i18n/messages/mr.json b/src/i18n/messages/mr.json index 48cd0e4010..53af82d57a 100644 --- a/src/i18n/messages/mr.json +++ b/src/i18n/messages/mr.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "टाइमलाइन", "drawerMetrics": "मेट्रिक्स", "drawerResult": "निकाल", + "drawerMemory": "वापरलेली मेमरी", "drawerActions": "क्रिया", "drawerClose": "बंद करा", "copyTrace": "ट्रेस JSON कॉपी करा", @@ -13997,7 +13998,11 @@ "actionDone": "क्रिया लागू केली", "actionFailed": "क्रिया अयशस्वी: {error}", "detailFailed": "तपशील लोड करण्यात अयशस्वी: {error}", - "mirroredInA2A": "A2A मध्ये प्रतिबिंबित" + "mirroredInA2A": "A2A मध्ये प्रतिबिंबित", + "actionRepeat": "पुनरावृत्ती करा", + "repeatConfirm": "पुष्टीसाठी पुन्हा क्लिक करा", + "repeatDone": "नवीन रन तयार केला", + "repeatUnavailable": "मूळ इनपुट उपलब्ध नाही" }, "cliproxyProviderExposure": { "title": "प्रदाता प्रदर्शन", diff --git a/src/i18n/messages/ms.json b/src/i18n/messages/ms.json index 2bfa46ba2d..05b504ffde 100644 --- a/src/i18n/messages/ms.json +++ b/src/i18n/messages/ms.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Garis Masa", "drawerMetrics": "Metrik", "drawerResult": "Keputusan", + "drawerMemory": "Memori digunakan", "drawerActions": "Tindakan", "drawerClose": "Tutup", "copyTrace": "Salin JSON jejak", @@ -13997,7 +13998,11 @@ "actionDone": "Tindakan digunakan", "actionFailed": "Tindakan gagal: {error}", "detailFailed": "Gagal memuatkan butiran: {error}", - "mirroredInA2A": "Dicerminkan dalam A2A" + "mirroredInA2A": "Dicerminkan dalam A2A", + "actionRepeat": "Ulang", + "repeatConfirm": "Klik sekali lagi untuk sahkan", + "repeatDone": "Larian baharu dicipta", + "repeatUnavailable": "Input asal tidak tersedia" }, "cliproxyProviderExposure": { "title": "Pendedahan Penyedia", diff --git a/src/i18n/messages/nl.json b/src/i18n/messages/nl.json index 3cbf5e726f..c50a7beebc 100644 --- a/src/i18n/messages/nl.json +++ b/src/i18n/messages/nl.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Tijdlijn", "drawerMetrics": "Statistieken", "drawerResult": "Resultaat", + "drawerMemory": "Gebruikt geheugen", "drawerActions": "Acties", "drawerClose": "Sluiten", "copyTrace": "Trace-JSON kopiëren", @@ -13997,7 +13998,11 @@ "actionDone": "Actie toegepast", "actionFailed": "Actie mislukt: {error}", "detailFailed": "Details laden mislukt: {error}", - "mirroredInA2A": "Weergegeven in A2A" + "mirroredInA2A": "Weergegeven in A2A", + "actionRepeat": "Herhalen", + "repeatConfirm": "Klik nogmaals om te bevestigen", + "repeatDone": "Nieuwe run aangemaakt", + "repeatUnavailable": "Oorspronkelijke invoer niet beschikbaar" }, "cliproxyProviderExposure": { "title": "Provider Blootstelling", diff --git a/src/i18n/messages/no.json b/src/i18n/messages/no.json index 786339d422..5f7a1c64b0 100644 --- a/src/i18n/messages/no.json +++ b/src/i18n/messages/no.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Tidslinje", "drawerMetrics": "Målinger", "drawerResult": "Resultat", + "drawerMemory": "Brukt minne", "drawerActions": "Handlinger", "drawerClose": "Lukk", "copyTrace": "Kopiér spor (JSON)", @@ -13997,7 +13998,11 @@ "actionDone": "Handling utført", "actionFailed": "Handling mislyktes: {error}", "detailFailed": "Kunne ikke laste inn detaljer: {error}", - "mirroredInA2A": "Speilet i A2A" + "mirroredInA2A": "Speilet i A2A", + "actionRepeat": "Gjenta", + "repeatConfirm": "Klikk igjen for å bekrefte", + "repeatDone": "Ny kjøring opprettet", + "repeatUnavailable": "Opprinnelig inndata ikke tilgjengelig" }, "cliproxyProviderExposure": { "title": "Leverandør Eksponering", diff --git a/src/i18n/messages/phi.json b/src/i18n/messages/phi.json index dcb12cb233..3a3080eded 100644 --- a/src/i18n/messages/phi.json +++ b/src/i18n/messages/phi.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Timeline", "drawerMetrics": "Mga Sukatan", "drawerResult": "Resulta", + "drawerMemory": "Ginamit na memory", "drawerActions": "Mga Aksyon", "drawerClose": "Isara", "copyTrace": "Kopyahin ang trace JSON", @@ -13997,7 +13998,11 @@ "actionDone": "Naisagawa ang aksyon", "actionFailed": "Nabigo ang aksyon: {error}", "detailFailed": "Nabigong i-load ang mga detalye: {error}", - "mirroredInA2A": "Naka-mirror sa A2A" + "mirroredInA2A": "Naka-mirror sa A2A", + "actionRepeat": "Ulitin", + "repeatConfirm": "I-click ulit para kumpirmahin", + "repeatDone": "Ginawa ang bagong run", + "repeatUnavailable": "Hindi available ang orihinal na input" }, "cliproxyProviderExposure": { "title": "Provider Exposure", diff --git a/src/i18n/messages/pl.json b/src/i18n/messages/pl.json index 4e23cacf45..bd07b8321e 100644 --- a/src/i18n/messages/pl.json +++ b/src/i18n/messages/pl.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Oś czasu", "drawerMetrics": "Metryki", "drawerResult": "Wynik", + "drawerMemory": "Wykorzystana pamięć", "drawerActions": "Akcje", "drawerClose": "Zamknij", "copyTrace": "Kopiuj JSON śladu", @@ -13997,7 +13998,11 @@ "actionDone": "Akcja wykonana", "actionFailed": "Akcja nie powiodła się: {error}", "detailFailed": "Nie udało się wczytać szczegółów: {error}", - "mirroredInA2A": "Odzwierciedlone w A2A" + "mirroredInA2A": "Odzwierciedlone w A2A", + "actionRepeat": "Powtórz", + "repeatConfirm": "Kliknij ponownie, aby potwierdzić", + "repeatDone": "Utworzono nowe uruchomienie", + "repeatUnavailable": "Oryginalne dane wejściowe niedostępne" }, "cliproxyProviderExposure": { "title": "Ekspozycja Dostawcy", diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index 4ce0321975..7c4d1eb12c 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -13996,6 +13996,7 @@ "drawerTimeline": "Linha do tempo", "drawerMetrics": "Métricas", "drawerResult": "Resultado", + "drawerMemory": "Memória usada", "drawerActions": "Ações", "drawerClose": "Fechar", "copyTrace": "Copiar trace em JSON", @@ -14005,7 +14006,11 @@ "actionDone": "Ação aplicada", "actionFailed": "Ação falhou: {error}", "detailFailed": "Falha ao carregar detalhes: {error}", - "mirroredInA2A": "Espelhado no A2A" + "mirroredInA2A": "Espelhado no A2A", + "actionRepeat": "Repetir", + "repeatConfirm": "Clique novamente para confirmar", + "repeatDone": "Nova execução criada", + "repeatUnavailable": "Entrada original não disponível" }, "cliproxyProviderExposure": { "title": "Exposição do Provedor", diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index dfc77dc34a..3533f6370a 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Linha Cronológica", "drawerMetrics": "Métricas", "drawerResult": "Resultado", + "drawerMemory": "Memória utilizada", "drawerActions": "Ações", "drawerClose": "Fechar", "copyTrace": "Copiar trace em JSON", @@ -13997,7 +13998,11 @@ "actionDone": "Ação aplicada", "actionFailed": "Falha na ação: {error}", "detailFailed": "Falha ao carregar detalhes: {error}", - "mirroredInA2A": "Espelhado no A2A" + "mirroredInA2A": "Espelhado no A2A", + "actionRepeat": "Repetir", + "repeatConfirm": "Clique novamente para confirmar", + "repeatDone": "Nova execução criada", + "repeatUnavailable": "Entrada original não disponível" }, "cliproxyProviderExposure": { "title": "Exposição do Fornecedor", diff --git a/src/i18n/messages/ro.json b/src/i18n/messages/ro.json index 657c1c4f49..afe466f408 100644 --- a/src/i18n/messages/ro.json +++ b/src/i18n/messages/ro.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Cronologie", "drawerMetrics": "Metrici", "drawerResult": "Rezultat", + "drawerMemory": "Memorie utilizată", "drawerActions": "Acțiuni", "drawerClose": "Închide", "copyTrace": "Copiază JSON-ul de urmărire", @@ -13997,7 +13998,11 @@ "actionDone": "Acțiune aplicată", "actionFailed": "Acțiune eșuată: {error}", "detailFailed": "Încărcarea detaliilor a eșuat: {error}", - "mirroredInA2A": "Reflectat în A2A" + "mirroredInA2A": "Reflectat în A2A", + "actionRepeat": "Repetă", + "repeatConfirm": "Apasă din nou pentru a confirma", + "repeatDone": "Rulare nouă creată", + "repeatUnavailable": "Intrarea originală nu este disponibilă" }, "cliproxyProviderExposure": { "title": "Expunerea Furnizorului", diff --git a/src/i18n/messages/ru.json b/src/i18n/messages/ru.json index 5249870f77..dbd845f383 100644 --- a/src/i18n/messages/ru.json +++ b/src/i18n/messages/ru.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Хронология", "drawerMetrics": "Метрики", "drawerResult": "Результат", + "drawerMemory": "Использованная память", "drawerActions": "Действия", "drawerClose": "Закрыть", "copyTrace": "Скопировать трассировку JSON", @@ -13997,7 +13998,11 @@ "actionDone": "Действие применено", "actionFailed": "Действие не удалось: {error}", "detailFailed": "Не удалось загрузить детали: {error}", - "mirroredInA2A": "Отражено в A2A" + "mirroredInA2A": "Отражено в A2A", + "actionRepeat": "Повторить", + "repeatConfirm": "Нажмите ещё раз для подтверждения", + "repeatDone": "Создан новый запуск", + "repeatUnavailable": "Исходные данные недоступны" }, "cliproxyProviderExposure": { "title": "Экспозиция Провайдера", diff --git a/src/i18n/messages/sk.json b/src/i18n/messages/sk.json index bdca1f08d8..40f2003b77 100644 --- a/src/i18n/messages/sk.json +++ b/src/i18n/messages/sk.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Časová os", "drawerMetrics": "Metriky", "drawerResult": "Výsledok", + "drawerMemory": "Použitá pamäť", "drawerActions": "Akcie", "drawerClose": "Zavrieť", "copyTrace": "Kopírovať trasovanie JSON", @@ -13997,7 +13998,11 @@ "actionDone": "Akcia použitá", "actionFailed": "Akcia zlyhala: {error}", "detailFailed": "Nepodarilo sa načítať podrobnosti: {error}", - "mirroredInA2A": "Zrkadlené v A2A" + "mirroredInA2A": "Zrkadlené v A2A", + "actionRepeat": "Opakovať", + "repeatConfirm": "Kliknutím znova potvrďte", + "repeatDone": "Vytvorené nové spustenie", + "repeatUnavailable": "Pôvodný vstup nie je k dispozícii" }, "cliproxyProviderExposure": { "title": "Expozícia poskytovateľa", diff --git a/src/i18n/messages/sv.json b/src/i18n/messages/sv.json index 99a0669db9..175c8e1f97 100644 --- a/src/i18n/messages/sv.json +++ b/src/i18n/messages/sv.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Tidslinje", "drawerMetrics": "Mätvärden", "drawerResult": "Resultat", + "drawerMemory": "Använt minne", "drawerActions": "Åtgärder", "drawerClose": "Stäng", "copyTrace": "Kopiera spårnings-JSON", @@ -13997,7 +13998,11 @@ "actionDone": "Åtgärd tillämpad", "actionFailed": "Åtgärden misslyckades: {error}", "detailFailed": "Det gick inte att läsa in detaljer: {error}", - "mirroredInA2A": "Speglad i A2A" + "mirroredInA2A": "Speglad i A2A", + "actionRepeat": "Upprepa", + "repeatConfirm": "Klicka igen för att bekräfta", + "repeatDone": "Ny körning skapad", + "repeatUnavailable": "Ursprunglig indata ej tillgänglig" }, "cliproxyProviderExposure": { "title": "Leverantörsexponering", diff --git a/src/i18n/messages/sw.json b/src/i18n/messages/sw.json index be5407c630..b8288d3863 100644 --- a/src/i18n/messages/sw.json +++ b/src/i18n/messages/sw.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Ratiba ya matukio", "drawerMetrics": "Vipimo", "drawerResult": "Matokeo", + "drawerMemory": "Kumbukumbu iliyotumika", "drawerActions": "Vitendo", "drawerClose": "Funga", "copyTrace": "Nakili JSON ya ufuatiliaji", @@ -13997,7 +13998,11 @@ "actionDone": "Kitendo kimetumika", "actionFailed": "Kitendo kimeshindwa: {error}", "detailFailed": "Imeshindwa kupakia maelezo: {error}", - "mirroredInA2A": "Kimeakisiwa katika A2A" + "mirroredInA2A": "Kimeakisiwa katika A2A", + "actionRepeat": "Rudia", + "repeatConfirm": "Bofya tena kuthibitisha", + "repeatDone": "Uendeshaji mpya umeundwa", + "repeatUnavailable": "Ingizo asili halipatikani" }, "cliproxyProviderExposure": { "title": "Ufunuo wa Mtoa Huduma", diff --git a/src/i18n/messages/ta.json b/src/i18n/messages/ta.json index b492d8a373..e0614361fc 100644 --- a/src/i18n/messages/ta.json +++ b/src/i18n/messages/ta.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "காலவரிசை", "drawerMetrics": "அளவீடுகள்", "drawerResult": "முடிவு", + "drawerMemory": "பயன்படுத்தப்பட்ட நினைவகம்", "drawerActions": "செயல்கள்", "drawerClose": "மூடு", "copyTrace": "ட்ரேஸ் JSON-ஐ நகலெடு", @@ -13997,7 +13998,11 @@ "actionDone": "செயல் பயன்படுத்தப்பட்டது", "actionFailed": "செயல் தோல்வியடைந்தது: {error}", "detailFailed": "விவரங்களை ஏற்ற முடியவில்லை: {error}", - "mirroredInA2A": "A2A இல் பிரதிபலிக்கப்பட்டது" + "mirroredInA2A": "A2A இல் பிரதிபலிக்கப்பட்டது", + "actionRepeat": "மீண்டும் செய்", + "repeatConfirm": "உறுதிப்படுத்த மீண்டும் கிளிக் செய்யவும்", + "repeatDone": "புதிய இயக்கம் உருவாக்கப்பட்டது", + "repeatUnavailable": "அசல் உள்ளீடு கிடைக்கவில்லை" }, "cliproxyProviderExposure": { "title": "சேவையாளர் வெளிப்பாடு", diff --git a/src/i18n/messages/te.json b/src/i18n/messages/te.json index fa123c3386..6dfa5290b7 100644 --- a/src/i18n/messages/te.json +++ b/src/i18n/messages/te.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "కాలక్రమం", "drawerMetrics": "మెట్రిక్స్", "drawerResult": "ఫలితం", + "drawerMemory": "ఉపయోగించిన మెమరీ", "drawerActions": "చర్యలు", "drawerClose": "మూసివేయి", "copyTrace": "ట్రేస్ JSON కాపీ చేయి", @@ -13997,7 +13998,11 @@ "actionDone": "చర్య వర్తింపజేయబడింది", "actionFailed": "చర్య విఫలమైంది: {error}", "detailFailed": "వివరాలను లోడ్ చేయడంలో విఫలమైంది: {error}", - "mirroredInA2A": "A2Aలో ప్రతిబింబించబడింది" + "mirroredInA2A": "A2Aలో ప్రతిబింబించబడింది", + "actionRepeat": "పునరావృతం చేయండి", + "repeatConfirm": "నిర్ధారించడానికి మళ్లీ క్లిక్ చేయండి", + "repeatDone": "కొత్త రన్ సృష్టించబడింది", + "repeatUnavailable": "అసలు ఇన్‌పుట్ అందుబాటులో లేదు" }, "cliproxyProviderExposure": { "title": "ప్రొవైడర్ ఎక్స్‌పోజర్", diff --git a/src/i18n/messages/th.json b/src/i18n/messages/th.json index 6b0e800747..126df99e1b 100644 --- a/src/i18n/messages/th.json +++ b/src/i18n/messages/th.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "ไทม์ไลน์", "drawerMetrics": "เมตริก", "drawerResult": "ผลลัพธ์", + "drawerMemory": "หน่วยความจำที่ใช้", "drawerActions": "การดำเนินการ", "drawerClose": "ปิด", "copyTrace": "คัดลอก JSON การติดตาม", @@ -13997,7 +13998,11 @@ "actionDone": "ดำเนินการเรียบร้อยแล้ว", "actionFailed": "การดำเนินการล้มเหลว: {error}", "detailFailed": "โหลดรายละเอียดไม่สำเร็จ: {error}", - "mirroredInA2A": "สะท้อนใน A2A" + "mirroredInA2A": "สะท้อนใน A2A", + "actionRepeat": "ทำซ้ำ", + "repeatConfirm": "คลิกอีกครั้งเพื่อยืนยัน", + "repeatDone": "สร้างการรันใหม่แล้ว", + "repeatUnavailable": "ไม่มีข้อมูลนำเข้าต้นฉบับ" }, "cliproxyProviderExposure": { "title": "การเปิดเผยผู้ให้บริการ", diff --git a/src/i18n/messages/tr.json b/src/i18n/messages/tr.json index 84cc51f338..5ee309e90d 100644 --- a/src/i18n/messages/tr.json +++ b/src/i18n/messages/tr.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Zaman Çizelgesi", "drawerMetrics": "Metrikler", "drawerResult": "Sonuç", + "drawerMemory": "Kullanılan bellek", "drawerActions": "Eylemler", "drawerClose": "Kapat", "copyTrace": "İzleme JSON'ını kopyala", @@ -13997,7 +13998,11 @@ "actionDone": "İşlem uygulandı", "actionFailed": "İşlem başarısız: {error}", "detailFailed": "Ayrıntılar yüklenemedi: {error}", - "mirroredInA2A": "A2A'da yansıtıldı" + "mirroredInA2A": "A2A'da yansıtıldı", + "actionRepeat": "Tekrarla", + "repeatConfirm": "Onaylamak için tekrar tıklayın", + "repeatDone": "Yeni çalıştırma oluşturuldu", + "repeatUnavailable": "Orijinal girdi kullanılamıyor" }, "cliproxyProviderExposure": { "title": "Sağlayıcı Maruziyeti", diff --git a/src/i18n/messages/uk-UA.json b/src/i18n/messages/uk-UA.json index 566d659277..e1203b1d1b 100644 --- a/src/i18n/messages/uk-UA.json +++ b/src/i18n/messages/uk-UA.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "Хронологія", "drawerMetrics": "Метрики", "drawerResult": "Результат", + "drawerMemory": "Використана пам'ять", "drawerActions": "Дії", "drawerClose": "Закрити", "copyTrace": "Копіювати трасування JSON", @@ -13997,7 +13998,11 @@ "actionDone": "Дію виконано", "actionFailed": "Дія не виконана: {error}", "detailFailed": "Не вдалося завантажити деталі: {error}", - "mirroredInA2A": "Відображено в A2A" + "mirroredInA2A": "Відображено в A2A", + "actionRepeat": "Повторити", + "repeatConfirm": "Натисніть ще раз, щоб підтвердити", + "repeatDone": "Створено новий запуск", + "repeatUnavailable": "Початкові вхідні дані недоступні" }, "cliproxyProviderExposure": { "title": "Виток Постачальника", diff --git a/src/i18n/messages/ur.json b/src/i18n/messages/ur.json index 4ff890871d..fdc5391976 100644 --- a/src/i18n/messages/ur.json +++ b/src/i18n/messages/ur.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "ٹائم لائن", "drawerMetrics": "میٹرکس", "drawerResult": "نتیجہ", + "drawerMemory": "استعمال شدہ میموری", "drawerActions": "اقدامات", "drawerClose": "بند کریں", "copyTrace": "ٹریس JSON کاپی کریں", @@ -13997,7 +13998,11 @@ "actionDone": "اقدام لاگو ہو گیا", "actionFailed": "اقدام ناکام: {error}", "detailFailed": "تفصیلات لوڈ کرنے میں ناکامی: {error}", - "mirroredInA2A": "A2A میں مطابقت شدہ" + "mirroredInA2A": "A2A میں مطابقت شدہ", + "actionRepeat": "دہرائیں", + "repeatConfirm": "تصدیق کے لیے دوبارہ کلک کریں", + "repeatDone": "نئی رن بنائی گئی", + "repeatUnavailable": "اصل ان پٹ دستیاب نہیں ہے" }, "cliproxyProviderExposure": { "title": "پرووائیڈر ایکسپوژر", diff --git a/src/i18n/messages/vi.json b/src/i18n/messages/vi.json index 1200687d7f..61287b0446 100644 --- a/src/i18n/messages/vi.json +++ b/src/i18n/messages/vi.json @@ -13996,6 +13996,7 @@ "drawerTimeline": "Dòng thời gian", "drawerMetrics": "Chỉ số", "drawerResult": "Kết quả", + "drawerMemory": "Bộ nhớ đã sử dụng", "drawerActions": "Hành động", "drawerClose": "Đóng", "copyTrace": "Sao chép JSON theo dõi", @@ -14005,7 +14006,11 @@ "actionDone": "Đã áp dụng hành động", "actionFailed": "Hành động thất bại: {error}", "detailFailed": "Không tải được chi tiết: {error}", - "mirroredInA2A": "Được phản chiếu trong A2A" + "mirroredInA2A": "Được phản chiếu trong A2A", + "actionRepeat": "Lặp lại", + "repeatConfirm": "Nhấp lại để xác nhận", + "repeatDone": "Đã tạo lượt chạy mới", + "repeatUnavailable": "Không có dữ liệu đầu vào gốc" }, "cliproxyProviderExposure": { "title": "Tiếp Xúc Nhà Cung Cấp", diff --git a/src/i18n/messages/zh-CN.json b/src/i18n/messages/zh-CN.json index 7d42b0a86e..4306b98ac2 100644 --- a/src/i18n/messages/zh-CN.json +++ b/src/i18n/messages/zh-CN.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "时间线", "drawerMetrics": "指标", "drawerResult": "结果", + "drawerMemory": "已使用的记忆", "drawerActions": "操作", "drawerClose": "关闭", "copyTrace": "复制追踪 JSON", @@ -13997,7 +13998,11 @@ "actionDone": "操作已应用", "actionFailed": "操作失败:{error}", "detailFailed": "加载详情失败:{error}", - "mirroredInA2A": "已在 A2A 中镜像" + "mirroredInA2A": "已在 A2A 中镜像", + "actionRepeat": "重复", + "repeatConfirm": "再次点击以确认", + "repeatDone": "已创建新的运行", + "repeatUnavailable": "原始输入不可用" }, "cliproxyProviderExposure": { "title": "提供者暴露", diff --git a/src/i18n/messages/zh-TW.json b/src/i18n/messages/zh-TW.json index be9aac3293..1a4db6102b 100644 --- a/src/i18n/messages/zh-TW.json +++ b/src/i18n/messages/zh-TW.json @@ -13988,6 +13988,7 @@ "drawerTimeline": "時間軸", "drawerMetrics": "指標", "drawerResult": "結果", + "drawerMemory": "已使用的記憶", "drawerActions": "動作", "drawerClose": "關閉", "copyTrace": "複製追蹤 JSON", @@ -13997,7 +13998,11 @@ "actionDone": "動作已套用", "actionFailed": "動作失敗:{error}", "detailFailed": "載入詳細資料失敗:{error}", - "mirroredInA2A": "已在 A2A 中鏡射" + "mirroredInA2A": "已在 A2A 中鏡射", + "actionRepeat": "重複", + "repeatConfirm": "再次點擊以確認", + "repeatDone": "已建立新的執行", + "repeatUnavailable": "原始輸入無法使用" }, "cliproxyProviderExposure": { "title": "提供者暴露", diff --git a/src/lib/a2a/taskExecution.ts b/src/lib/a2a/taskExecution.ts index 421a36bb95..68c8ff0937 100644 --- a/src/lib/a2a/taskExecution.ts +++ b/src/lib/a2a/taskExecution.ts @@ -1,4 +1,6 @@ import type { A2ATask, TaskArtifact } from "./taskManager"; +import { appendA2ATaskEvent } from "@/lib/db/a2aTasks"; +import { memoryManager } from "@/lib/memory/manager"; type TaskManagerLike = { updateTask: ( @@ -14,6 +16,90 @@ type StreamTaskResult = { metadata: Record; }; +/** + * Task D2 (Orchestration Canvas Fase 2, PR-C): a memory hit recorded for OBSERVABILITY ONLY. + * The retrieved memory is never injected into a skill's prompt or behavior — it is only + * mirrored into `task.metadata.memoryHits` and a `memory_hits` history event so the dashboard + * can show which memories were consulted for a given A2A task. + * + * Note on drift from the original spec: `Memory` (`src/lib/memory/types.ts`) does not expose a + * `score` field, so hits carry `key`/`type` instead of a relevance score. + */ +export interface MemoryHit { + id: string; + key: string; + type: string; + /** `content` truncated to 200 chars — never the full memory body. */ + snippet: string; +} + +/** DI seam for `collectMemoryHits` — tests need neither a real memory backend nor a database. */ +export interface MemoryHitsDeps { + search?: (cfg: { + query: string; + apiKeyId: string; + limit?: number; + }) => Promise>; + appendEvent?: (taskId: string, eventType: string, dataJson?: string) => void; +} + +/** + * Collect the memories consulted for a task's last user message, as pure observability. + * + * - Kill-switch: `OMNIROUTE_A2A_MEMORY_HITS=0` returns `[]` without querying anything. + * - Query = the content of the LAST message with `role === "user"`; empty/absent ⇒ `[]`. + * - Owner id = `task.owner ?? "mcp"` — the same keyless fallback the MCP memory tools use + * (`open-sse/mcp-server/tools/memoryTools.ts::resolveMemoryOwnerId`). + * - Any failure in the recall path ⇒ `[]` — this must never fail the caller's task. + * + * KNOWN LIMITATION — recall only resolves under the KEYLESS posture. `task.owner` is a + * SHA-256 PREFIX of the raw API key (`src/lib/a2a/authenticate.ts::resolveA2AOwner`), while + * memory rows are keyed by the DB api-key **id** (`String(apiKeyInfo.id)`, the value + * `getApiKeyMetadata()` returns — see `open-sse/mcp-server/mcpCallerIdentity.ts`). The two + * live in different namespaces, so for a keyed caller the search below matches nothing and + * the hits list is always empty; only the keyless case (`owner === undefined` → `"mcp"`) + * lines up with the MCP-tool owner id. Bridging them needs a hash→api-key-id lookup that + * does NOT exist today: `src/lib/db/apiKeys.ts` only ever looks a key up by its RAW value + * (`WHERE key = ? OR key_hash = ?`, with the FULL sha256 hex), and the raw key is long gone + * by the time a task executes. Deliberately NOT worked around here — inventing a + * prefix-scan lookup over `api_keys` would be a new auth-adjacent surface. Follow-up: + * either persist the DB api-key id on the task alongside the hash, or add an explicit + * `getApiKeyIdByKeyHashPrefix()` in the db layer. + */ +export async function collectMemoryHits( + task: A2ATask, + deps?: MemoryHitsDeps +): Promise { + if (process.env.OMNIROUTE_A2A_MEMORY_HITS === "0") return []; + + const messages = task.input?.messages ?? []; + let query: string | undefined; + for (let i = messages.length - 1; i >= 0; i--) { + if (messages[i].role === "user") { + query = messages[i].content; + break; + } + } + if (!query || query.trim() === "") return []; + + try { + const search = + deps?.search ?? + (async (cfg: { query: string; apiKeyId: string; limit?: number }) => + memoryManager.getPrimaryBackend().search(cfg)); + const apiKeyId = task.owner ?? "mcp"; + const results = await search({ query, apiKeyId, limit: 5 }); + return results.map((m) => ({ + id: m.id, + key: m.key, + type: m.type, + snippet: m.content.slice(0, 200), + })); + } catch { + return []; + } +} + export type A2ASkillHandler = (task: A2ATask) => Promise; export const A2A_SKILL_HANDLERS: Record = { @@ -46,9 +132,20 @@ export const A2A_SKILL_HANDLERS: Record = { export async function executeA2ATaskWithState( tm: TaskManagerLike, task: A2ATask, - handler: (task: A2ATask) => Promise + handler: (task: A2ATask) => Promise, + deps?: MemoryHitsDeps ) { try { + const hits = await collectMemoryHits(task, deps); + if (hits.length) { + task.metadata.memoryHits = hits; + try { + (deps?.appendEvent ?? appendA2ATaskEvent)(task.id, "memory_hits", JSON.stringify(hits)); + } catch { + // best-effort — never break the task's write path + } + } + const result = await handler(task); tm.updateTask(task.id, "completed", result.artifacts); return result; diff --git a/src/lib/a2a/taskManager.ts b/src/lib/a2a/taskManager.ts index 7b6d18b254..0c9f280176 100644 --- a/src/lib/a2a/taskManager.ts +++ b/src/lib/a2a/taskManager.ts @@ -14,11 +14,7 @@ import { randomUUID } from "crypto"; import { emit } from "@/lib/events/eventBus"; -import { - upsertA2ATask, - appendA2ATaskEvent, - purgeA2AHistory, -} from "@/lib/db/a2aTasks"; +import { upsertA2ATask, appendA2ATaskEvent, purgeA2AHistory } from "@/lib/db/a2aTasks"; import { logger } from "@omniroute/open-sse/utils/logger"; const log = logger("A2A_TASKS"); @@ -204,7 +200,14 @@ export class A2ATaskManager { input, artifacts: [], events: [{ timestamp: now.toISOString(), state: "submitted" }], - metadata: input.metadata || {}, + // COPY, never the caller's object: `metadata` is the task's own mutable + // runtime bag (`taskExecution.ts` writes `memoryHits` into it), while + // `input.metadata` is the immutable record of what the caller sent. Sharing + // one reference made every runtime write leak back into `input` — and from + // there into the persisted `a2a_tasks.input_json` and into the drawer's + // "Repeat" body, so a repeated task was born carrying the previous run's + // memory snippets even with `OMNIROUTE_A2A_MEMORY_HITS=0`. + metadata: { ...(input.metadata ?? {}) }, createdAt: now.toISOString(), updatedAt: now.toISOString(), expiresAt: new Date(now.getTime() + this.ttlMs).toISOString(), diff --git a/tests/unit/a2a-memory-hits.test.ts b/tests/unit/a2a-memory-hits.test.ts new file mode 100644 index 0000000000..9d6a42486f --- /dev/null +++ b/tests/unit/a2a-memory-hits.test.ts @@ -0,0 +1,331 @@ +/** + * Task D2 (Orchestration Canvas Fase 2, PR-C): `collectMemoryHits` records WHICH memories were + * consulted for an A2A task, as pure observability — the hits are never injected into the + * skill's prompt or behavior, only mirrored into `task.metadata.memoryHits` and a + * `memory_hits` history event. + * + * Uses FAKE `MemoryHitsDeps` throughout (no real memory backend, no SQLite) — the DI seam + * exists precisely so this suite needs neither. + */ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { + collectMemoryHits, + executeA2ATaskWithState, + type MemoryHit, + type MemoryHitsDeps, +} from "../../src/lib/a2a/taskExecution.ts"; +import { + A2ATaskManager, + type A2APersistence, + type A2ATask, +} from "../../src/lib/a2a/taskManager.ts"; + +function makeTask(overrides: Partial = {}): A2ATask { + return { + id: "task-1", + skill: "smart-routing", + state: "working", + input: { + skill: "smart-routing", + messages: [{ role: "user", content: "what is the cheapest gpt-4 provider?" }], + }, + artifacts: [], + events: [], + metadata: {}, + createdAt: new Date().toISOString(), + updatedAt: new Date().toISOString(), + expiresAt: new Date(Date.now() + 60_000).toISOString(), + ...overrides, + }; +} + +const ENV_KEY = "OMNIROUTE_A2A_MEMORY_HITS"; + +function withEnv(value: string | undefined, fn: () => Promise) { + const original = process.env[ENV_KEY]; + if (value === undefined) delete process.env[ENV_KEY]; + else process.env[ENV_KEY] = value; + return fn().finally(() => { + if (original === undefined) delete process.env[ENV_KEY]; + else process.env[ENV_KEY] = original; + }); +} + +test("collectMemoryHits maps search results and truncates snippet to 200 chars", async () => { + const longContent = "x".repeat(250); + const searchCalls: Array<{ query: string; apiKeyId: string; limit?: number }> = []; + const deps: MemoryHitsDeps = { + search: async (cfg) => { + searchCalls.push(cfg); + return [ + { id: "m1", key: "k1", type: "factual", content: longContent }, + { id: "m2", key: "k2", type: "episodic", content: "short" }, + ]; + }, + }; + + const task = makeTask(); + const hits = await collectMemoryHits(task, deps); + + assert.equal(searchCalls.length, 1); + assert.equal(searchCalls[0].query, "what is the cheapest gpt-4 provider?"); + assert.equal(searchCalls[0].apiKeyId, "mcp"); + + assert.deepEqual(hits, [ + { id: "m1", key: "k1", type: "factual", snippet: longContent.slice(0, 200) }, + { id: "m2", key: "k2", type: "episodic", snippet: "short" }, + ] satisfies MemoryHit[]); + assert.equal(hits[0].snippet.length, 200); +}); + +test("collectMemoryHits uses task.owner as apiKeyId when present", async () => { + let seenApiKeyId: string | undefined; + const deps: MemoryHitsDeps = { + search: async (cfg) => { + seenApiKeyId = cfg.apiKeyId; + return []; + }, + }; + + const task = makeTask({ owner: "owner-123" }); + await collectMemoryHits(task, deps); + + assert.equal(seenApiKeyId, "owner-123"); +}); + +test("collectMemoryHits uses the LAST user message as the query", async () => { + let seenQuery: string | undefined; + const deps: MemoryHitsDeps = { + search: async (cfg) => { + seenQuery = cfg.query; + return []; + }, + }; + + const task = makeTask({ + input: { + skill: "smart-routing", + messages: [ + { role: "user", content: "first question" }, + { role: "assistant", content: "an answer" }, + { role: "user", content: "second question" }, + ], + }, + }); + await collectMemoryHits(task, deps); + + assert.equal(seenQuery, "second question"); +}); + +test("collectMemoryHits returns [] and never calls search when there is no user message", async () => { + let called = false; + const deps: MemoryHitsDeps = { + search: async () => { + called = true; + return []; + }, + }; + + const task = makeTask({ + input: { skill: "smart-routing", messages: [{ role: "assistant", content: "hi" }] }, + }); + const hits = await collectMemoryHits(task, deps); + + assert.deepEqual(hits, []); + assert.equal(called, false); +}); + +test("collectMemoryHits returns [] when search throws — never fails the caller", async () => { + const deps: MemoryHitsDeps = { + search: async () => { + throw new Error("boom"); + }, + }; + + const task = makeTask(); + const hits = await collectMemoryHits(task, deps); + + assert.deepEqual(hits, []); +}); + +test("collectMemoryHits kill-switch (OMNIROUTE_A2A_MEMORY_HITS=0) returns [] without calling search", async () => { + await withEnv("0", async () => { + let called = false; + const deps: MemoryHitsDeps = { + search: async () => { + called = true; + return []; + }, + }; + + const task = makeTask(); + const hits = await collectMemoryHits(task, deps); + + assert.deepEqual(hits, []); + assert.equal(called, false); + }); +}); + +test("executeA2ATaskWithState sets task.metadata.memoryHits and appends a memory_hits event when there are hits", async () => { + const appendEventCalls: Array<{ taskId: string; eventType: string; dataJson?: string }> = []; + const deps: MemoryHitsDeps = { + search: async () => [{ id: "m1", key: "k1", type: "factual", content: "hello" }], + appendEvent: (taskId, eventType, dataJson) => { + appendEventCalls.push({ taskId, eventType, dataJson }); + }, + }; + + const updateTaskCalls: unknown[] = []; + const tm = { + updateTask: (...args: unknown[]) => { + updateTaskCalls.push(args); + }, + }; + + const task = makeTask(); + const result = await executeA2ATaskWithState( + tm, + task, + async () => ({ artifacts: [], metadata: {} }), + deps + ); + + assert.deepEqual(result.artifacts, []); + assert.deepEqual(task.metadata.memoryHits, [ + { id: "m1", key: "k1", type: "factual", snippet: "hello" }, + ]); + assert.equal(appendEventCalls.length, 1); + assert.equal(appendEventCalls[0].taskId, "task-1"); + assert.equal(appendEventCalls[0].eventType, "memory_hits"); + assert.deepEqual(JSON.parse(appendEventCalls[0].dataJson ?? "[]"), [ + { id: "m1", key: "k1", type: "factual", snippet: "hello" }, + ]); + assert.equal(updateTaskCalls.length, 1); +}); + +test("executeA2ATaskWithState does not set metadata.memoryHits or append an event when there are no hits", async () => { + const appendEventCalls: unknown[] = []; + const deps: MemoryHitsDeps = { + search: async () => [], + appendEvent: (...args: unknown[]) => { + appendEventCalls.push(args); + }, + }; + + const tm = { updateTask: () => {} }; + const task = makeTask(); + await executeA2ATaskWithState(tm, task, async () => ({ artifacts: [], metadata: {} }), deps); + + assert.equal("memoryHits" in task.metadata, false); + assert.equal(appendEventCalls.length, 0); +}); + +test("executeA2ATaskWithState completes the task normally even when memory recall throws", async () => { + const deps: MemoryHitsDeps = { + search: async () => { + throw new Error("recall backend down"); + }, + }; + + let completedState: string | undefined; + const tm = { + updateTask: (_taskId: string, state: string) => { + completedState = state; + }, + }; + + const task = makeTask(); + const result = await executeA2ATaskWithState( + tm, + task, + async () => ({ artifacts: [{ type: "text", content: "ok" }], metadata: {} }), + deps + ); + + assert.equal(completedState, "completed"); + assert.deepEqual(result.artifacts, [{ type: "text", content: "ok" }]); + assert.equal("memoryHits" in task.metadata, false); +}); + +test("executeA2ATaskWithState swallows a throwing appendEvent (best-effort) and still completes", async () => { + const deps: MemoryHitsDeps = { + search: async () => [{ id: "m1", key: "k1", type: "factual", content: "hello" }], + appendEvent: () => { + throw new Error("db unavailable"); + }, + }; + + let completedState: string | undefined; + const tm = { + updateTask: (_taskId: string, state: string) => { + completedState = state; + }, + }; + + const task = makeTask(); + await executeA2ATaskWithState(tm, task, async () => ({ artifacts: [], metadata: {} }), deps); + + assert.equal(completedState, "completed"); + assert.deepEqual(task.metadata.memoryHits, [ + { id: "m1", key: "k1", type: "factual", snippet: "hello" }, + ]); +}); + +/** + * Regression (whole-branch review, Important 1): `createTask` used to store the CALLER's + * `input.metadata` object as the task's own `metadata`, so the `memoryHits` written above + * landed inside `task.input.metadata` too — from where it was serialized into + * `a2a_tasks.input_json` and echoed back by the drawer's "Repeat" body, making the repeated + * task be born carrying the previous run's memory snippets (visible even with the + * `OMNIROUTE_A2A_MEMORY_HITS=0` kill-switch on). `metadata` must be a COPY. + */ +test("executeA2ATaskWithState never leaks memoryHits into task.input.metadata or the persisted input", async () => { + const upsertCalls: Array<{ inputJson: string | null }> = []; + const persistence: A2APersistence = { + upsert: ((row: { inputJson: string | null }) => { + upsertCalls.push(row); + }) as A2APersistence["upsert"], + appendEvent: (() => {}) as A2APersistence["appendEvent"], + purge: ((): number => 0) as A2APersistence["purge"], + }; + const tm = new A2ATaskManager(5, persistence); + try { + const callerMetadata = { role: "general" }; + const task = tm.createTask({ + skill: "smart-routing", + messages: [{ role: "user", content: "route this please" }], + metadata: callerMetadata, + }); + + await executeA2ATaskWithState( + { updateTask: () => {} }, + task, + async () => ({ artifacts: [], metadata: {} }), + { + search: async () => [{ id: "m1", key: "k1", type: "factual", content: "hello" }], + appendEvent: () => {}, + } + ); + + // The hits ARE recorded on the task's runtime metadata … + assert.deepEqual(task.metadata.memoryHits, [ + { id: "m1", key: "k1", type: "factual", snippet: "hello" }, + ]); + // … but never on the immutable record of what the caller sent. + assert.equal("memoryHits" in (task.input.metadata ?? {}), false); + assert.deepEqual(task.input.metadata, { role: "general" }); + // … nor on the caller's own object (no aliasing in either direction). + assert.deepEqual(callerMetadata, { role: "general" }); + + // A persist AFTER the hits were recorded must still write a clean input_json. + tm.updateTask(task.id, "working"); + assert.ok(upsertCalls.length >= 2); + for (const row of upsertCalls) { + assert.ok(!String(row.inputJson).includes("memoryHits"), "input_json carries no memoryHits"); + } + } finally { + tm.destroy(); + } +}); diff --git a/tests/unit/conductor-create-route.test.ts b/tests/unit/conductor-create-route.test.ts new file mode 100644 index 0000000000..d16d47b1d3 --- /dev/null +++ b/tests/unit/conductor-create-route.test.ts @@ -0,0 +1,155 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { createServer, type Server } from "node:http"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-conductor-create-route-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const createRoute = await import("../../src/app/api/conductor/tasks/route.ts"); + +const servers: Server[] = []; + +function fakeHub(routes: Record): Promise { + const server = createServer((req, res) => { + const hit = Object.entries(routes).find(([p]) => (req.url ?? "").startsWith(p)); + res.writeHead(hit ? hit[1].status : 404, { "content-type": "application/json" }); + res.end( + JSON.stringify(hit ? hit[1].body : { error: "hub: segredo interno que NÃO pode vazar" }) + ); + }); + servers.push(server); + return new Promise((resolve) => { + server.listen(0, "127.0.0.1", () => { + const addr = server.address(); + resolve(`http://127.0.0.1:${typeof addr === "object" && addr ? addr.port : 0}`); + }); + }); +} + +function postJson(body: unknown): Request { + return new Request("http://localhost/api/conductor/tasks", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify(body), + }); +} + +test.beforeEach(() => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); + delete process.env.CONDUCTOR_HUB_URL; + delete process.env.CONDUCTOR_HUB_TOKEN; +}); + +test.after(async () => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + delete process.env.CONDUCTOR_HUB_URL; + delete process.env.CONDUCTOR_HUB_TOKEN; + while (servers.length > 0) { + const s = servers.pop(); + await new Promise((resolve) => s?.close(resolve)); + } +}); + +test("route: requireManagementAuth antes de criar a task no hub", () => { + const src = fs.readFileSync( + path.join(process.cwd(), "src/app/api/conductor/tasks/route.ts"), + "utf8" + ); + const authAt = src.indexOf("requireManagementAuth("); + assert.ok(authAt > 0, "handler chama requireManagementAuth"); + assert.match(src, /if \(authError\) return authError;/, "curto-circuito no erro de auth"); + const proxyAt = src.indexOf("createConductorTask("); + assert.ok(proxyAt > authAt, "proxy ao hub só depois do gate de auth"); + assert.ok( + !src.includes("CONDUCTOR_HUB_TOKEN"), + "token nunca manuseado na rota (vive no hubProxy)" + ); +}); + +test("POST /api/conductor/tasks: body válido + hub ok → 201 {task_id}", async () => { + process.env.CONDUCTOR_HUB_URL = await fakeHub({ + "/v1/tasks": { status: 201, body: { id: "t_repeat_1" } }, + }); + process.env.CONDUCTOR_HUB_TOKEN = "tok"; + + const res = await createRoute.POST( + postJson({ repoUrl: "https://git.x/repo", prompt: "refaça isso" }) + ); + assert.equal(res.status, 201); + assert.deepEqual(await res.json(), { task_id: "t_repeat_1" }); +}); + +test("POST /api/conductor/tasks: hub recusa (502) → status espelhado, sem corpo upstream", async () => { + process.env.CONDUCTOR_HUB_URL = await fakeHub({ + "/v1/tasks": { status: 502, body: { error: "segredo interno que NÃO pode vazar" } }, + }); + process.env.CONDUCTOR_HUB_TOKEN = "tok"; + + const res = await createRoute.POST( + postJson({ repoUrl: "https://git.x/repo", prompt: "refaça isso" }) + ); + assert.equal(res.status, 502); + const text = await res.text(); + assert.ok(!text.includes("segredo interno"), "corpo do hub NUNCA repassado (HR#12)"); +}); + +test("POST /api/conductor/tasks: body inválido (sem prompt) → 400", async () => { + const res = await createRoute.POST(postJson({ repoUrl: "https://git.x/repo" })); + assert.equal(res.status, 400); +}); + +test("POST /api/conductor/tasks: body inválido (sem repoUrl) → 400", async () => { + const res = await createRoute.POST(postJson({ prompt: "refaça isso" })); + assert.equal(res.status, 400); +}); + +test("POST /api/conductor/tasks: JSON malformado → 400", async () => { + const res = await createRoute.POST( + new Request("http://localhost/api/conductor/tasks", { + method: "POST", + headers: { "content-type": "application/json" }, + body: "{not json", + }) + ); + assert.equal(res.status, 400); +}); + +/** + * Review finding (Minor A): the hub's status was forwarded verbatim as OUR response status. + * `Response.json()` throws a `RangeError` for anything outside 200-599, and a 3xx/2xx is + * meaningless as an error status anyway — so anything outside 400-599 must become a 502 + * instead of an unhandled throw. A 302 with no `Location` is returned as-is by fetch (there + * is nothing to follow), which reproduces the out-of-band status without a fake fetch impl. + */ +test("POST /api/conductor/tasks: status fora de 400-599 vindo do hub é clampado para 502", async () => { + process.env.CONDUCTOR_HUB_URL = await fakeHub({ + "/v1/tasks": { status: 302, body: { error: "segredo interno que NÃO pode vazar" } }, + }); + process.env.CONDUCTOR_HUB_TOKEN = "tok"; + + const res = await createRoute.POST( + postJson({ repoUrl: "https://git.x/repo", prompt: "refaça isso" }) + ); + assert.equal(res.status, 502); + const text = await res.text(); + assert.ok(!text.includes("segredo interno"), "corpo do hub NUNCA repassado (HR#12)"); +}); + +test("POST /api/conductor/tasks: status 4xx/5xx legítimo do hub continua espelhado", async () => { + process.env.CONDUCTOR_HUB_URL = await fakeHub({ + "/v1/tasks": { status: 429, body: { error: "rate limited" } }, + }); + process.env.CONDUCTOR_HUB_TOKEN = "tok"; + + const res = await createRoute.POST( + postJson({ repoUrl: "https://git.x/repo", prompt: "refaça isso" }) + ); + assert.equal(res.status, 429); +}); diff --git a/tests/unit/ui/orchestrationDrawerRepeat.test.tsx b/tests/unit/ui/orchestrationDrawerRepeat.test.tsx new file mode 100644 index 0000000000..761b0d922f --- /dev/null +++ b/tests/unit/ui/orchestrationDrawerRepeat.test.tsx @@ -0,0 +1,786 @@ +// @vitest-environment jsdom +import React, { act } from "react"; +import { createRoot } from "react-dom/client"; +import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; + +vi.mock("next-intl", () => ({ + useTranslations: () => (k: string, v?: Record) => + v ? `${k}:${JSON.stringify(v)}` : k, +})); + +import { OrchestrationDrawer } from "@/app/(dashboard)/dashboard/orchestration/drawer/OrchestrationDrawer"; +import { repeatReqFor } from "@/app/(dashboard)/dashboard/orchestration/drawer/useDrawerDetail"; + +function render(el: React.ReactElement) { + const c = document.createElement("div"); + document.body.appendChild(c); + const root = createRoot(c); + act(() => root.render(el)); + return { + c, + cleanup: () => { + act(() => root.unmount()); + c.remove(); + }, + }; +} +afterEach(() => { + document.body.innerHTML = ""; +}); + +describe("OrchestrationDrawer memory section", () => { + it("renders the memory-used section (type/key/snippet) when an a2a task carries metadata.memoryHits", async () => { + const a2aTask = { + id: "1", + skill: "smart-routing", + state: "working", + input: { skill: "smart-routing", messages: [{ role: "user", content: "route this please" }] }, + artifacts: [], + events: [], + metadata: { + memoryHits: [ + { id: "m1", key: "user-pref-model", type: "preference", snippet: "prefers claude" }, + ], + }, + createdAt: "x", + updatedAt: "y", + expiresAt: "z", + }; + vi.stubGlobal( + "fetch", + vi.fn(() => Promise.resolve({ ok: true, json: () => Promise.resolve({ task: a2aTask }) })) + ); + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "running", label: "x" }; + const { c, cleanup } = render( + {}} onActionDone={() => {}} /> + ); + await act(async () => { + await Promise.resolve(); + }); + expect(c.textContent).toContain("drawerMemory"); + expect(c.textContent).toContain("preference"); + expect(c.textContent).toContain("user-pref-model"); + expect(c.textContent).toContain("prefers claude"); + cleanup(); + }); + + it("omits the memory-used section when an a2a task has no memoryHits", async () => { + const a2aTask = { + id: "1", + skill: "smart-routing", + state: "working", + input: { skill: "smart-routing", messages: [{ role: "user", content: "route this please" }] }, + artifacts: [], + events: [], + metadata: {}, + createdAt: "x", + updatedAt: "y", + expiresAt: "z", + }; + vi.stubGlobal( + "fetch", + vi.fn(() => Promise.resolve({ ok: true, json: () => Promise.resolve({ task: a2aTask }) })) + ); + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "running", label: "x" }; + const { c, cleanup } = render( + {}} onActionDone={() => {}} /> + ); + await act(async () => { + await Promise.resolve(); + }); + expect(c.textContent).not.toContain("drawerMemory"); + cleanup(); + }); + + it("renders nothing and does not throw when metadata.memoryHits is a malformed, non-array shape (a string, not an array of hits)", async () => { + const a2aTask = { + id: "1", + skill: "smart-routing", + state: "working", + input: { skill: "smart-routing", messages: [{ role: "user", content: "route this please" }] }, + artifacts: [], + events: [], + metadata: { memoryHits: "boom" }, + createdAt: "x", + updatedAt: "y", + expiresAt: "z", + }; + vi.stubGlobal( + "fetch", + vi.fn(() => Promise.resolve({ ok: true, json: () => Promise.resolve({ task: a2aTask }) })) + ); + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "running", label: "x" }; + const { c, cleanup } = render( + {}} onActionDone={() => {}} /> + ); + await expect( + act(async () => { + await Promise.resolve(); + }) + ).resolves.not.toThrow(); + expect(c.textContent).not.toContain("drawerMemory"); + cleanup(); + }); + + it("filters out malformed entries in metadata.memoryHits (an array of junk) without throwing", async () => { + const a2aTask = { + id: "1", + skill: "smart-routing", + state: "working", + input: { skill: "smart-routing", messages: [{ role: "user", content: "route this please" }] }, + artifacts: [], + events: [], + // `{ id: "x", key: {...} }` is the dangerous shape: a VALID string id next to an + // object field that the section renders as a React child — a guard that only checks + // `id` lets it through and React throws "Objects are not valid as a React child", + // taking the whole drawer down. Every one of the four rendered fields must be a string. + metadata: { + memoryHits: [ + { notId: "x" }, + "nope", + 123, + null, + { id: "x", key: { a: 1 }, type: "factual", snippet: "s" }, + { id: "y", key: "k", type: ["nope"], snippet: "s" }, + { id: "z", key: "k", type: "factual", snippet: { toString: "boom" } }, + { id: "w", key: "k", type: "factual" }, + ], + }, + createdAt: "x", + updatedAt: "y", + expiresAt: "z", + }; + vi.stubGlobal( + "fetch", + vi.fn(() => Promise.resolve({ ok: true, json: () => Promise.resolve({ task: a2aTask }) })) + ); + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "running", label: "x" }; + const { c, cleanup } = render( + {}} onActionDone={() => {}} /> + ); + await expect( + act(async () => { + await Promise.resolve(); + }) + ).resolves.not.toThrow(); + expect(c.textContent).not.toContain("drawerMemory"); + cleanup(); + }); + + it("never shows the memory-used section for non-a2a sources, even with attacker-shaped raw data", async () => { + const detail = { + data: { + id: "t1", + providerId: "devin", + status: "succeeded", + prompt: "x", + source: { repoName: "r", repoUrl: "https://x" }, + options: {}, + activities: [], + metadata: { + memoryHits: [{ id: "m1", key: "k", type: "t", snippet: "s" }], + }, + createdAt: "x", + updatedAt: "y", + }, + }; + vi.stubGlobal( + "fetch", + vi.fn(() => Promise.resolve({ ok: true, json: () => Promise.resolve(detail) })) + ); + const node = { + id: "cloud-agent:t1", + kind: "work", + source: "cloud-agent", + state: "succeeded", + label: "x", + }; + const { c, cleanup } = render( + {}} onActionDone={() => {}} /> + ); + await act(async () => { + await Promise.resolve(); + }); + expect(c.textContent).not.toContain("drawerMemory"); + cleanup(); + }); +}); + +describe("repeatReqFor", () => { + it("builds the cloud-agent repeat request from the loaded detail (CreateCloudAgentTaskSchema shape)", () => { + const node = { + id: "cloud-agent:t1", + kind: "work", + source: "cloud-agent", + state: "succeeded", + label: "x", + }; + const detail = { + id: "t1", + providerId: "devin", + prompt: "do the thing", + source: { repoName: "r", repoUrl: "https://x" }, + options: { autoCreatePr: true }, + activities: [], + }; + const req = repeatReqFor(node as never, detail); + expect(req?.url).toBe("/api/v1/agents/tasks"); + expect(req?.init.method).toBe("POST"); + expect(JSON.parse(String(req?.init.body))).toEqual({ + providerId: "devin", + prompt: "do the thing", + source: { repoName: "r", repoUrl: "https://x" }, + options: { autoCreatePr: true }, + }); + }); + + it("returns null for cloud-agent when neither providerId nor prompt is recoverable", () => { + const node = { + id: "cloud-agent:t1", + kind: "work", + source: "cloud-agent", + state: "succeeded", + label: "x", + }; + expect(repeatReqFor(node as never, { source: {}, options: {}, activities: [] })).toBeNull(); + }); + + it("returns null for cloud-agent when providerId is the only missing field", () => { + const node = { + id: "cloud-agent:t1", + kind: "work", + source: "cloud-agent", + state: "succeeded", + label: "x", + }; + const detail = { + prompt: "do the thing", + source: { repoName: "r", repoUrl: "https://x" }, + options: {}, + activities: [], + }; + expect(repeatReqFor(node as never, detail)).toBeNull(); + }); + + it("returns null for cloud-agent when prompt is the only missing field", () => { + const node = { + id: "cloud-agent:t1", + kind: "work", + source: "cloud-agent", + state: "succeeded", + label: "x", + }; + const detail = { + providerId: "devin", + source: { repoName: "r", repoUrl: "https://x" }, + options: {}, + activities: [], + }; + expect(repeatReqFor(node as never, detail)).toBeNull(); + }); + + it("returns null for cloud-agent when source is the only missing field (CreateCloudAgentTaskSchema also requires it — the field this fix started checking)", () => { + const node = { + id: "cloud-agent:t1", + kind: "work", + source: "cloud-agent", + state: "succeeded", + label: "x", + }; + const detail = { + providerId: "devin", + prompt: "do the thing", + options: {}, + activities: [], + }; + expect(repeatReqFor(node as never, detail)).toBeNull(); + }); + + it("builds the a2a repeat request as a message/send JSON-RPC call from detail.input", () => { + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "succeeded", label: "x" }; + const detail = { + input: { + skill: "smart-routing", + messages: [{ role: "user", content: "route this please" }], + metadata: { role: "general" }, + }, + }; + const req = repeatReqFor(node as never, detail); + expect(req?.url).toBe("/a2a"); + expect(JSON.parse(String(req?.init.body))).toEqual({ + jsonrpc: "2.0", + id: "a2a:1", + method: "message/send", + params: { + skill: "smart-routing", + messages: [{ role: "user", content: "route this please" }], + metadata: { role: "general" }, + }, + }); + }); + + it("strips memoryHits from the a2a repeat metadata (never re-sends the previous run's memory)", () => { + // `memoryHits` is observability written by the PREVIOUS run, never caller input. Tasks + // persisted before the createTask copy-fix still carry it inside `input.metadata`, so the + // repeat path has to drop it — otherwise the new task is born with the old run's snippets + // and shows them in the drawer even with `OMNIROUTE_A2A_MEMORY_HITS=0`. + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "succeeded", label: "x" }; + const detail = { + input: { + skill: "smart-routing", + messages: [{ role: "user", content: "route this please" }], + metadata: { + role: "general", + memoryHits: [{ id: "m1", key: "k1", type: "factual", snippet: "leaked" }], + }, + }, + }; + const body = JSON.parse(String(repeatReqFor(node as never, detail)?.init.body)); + expect(body.params.metadata).toEqual({ role: "general" }); + expect(JSON.stringify(body)).not.toContain("memoryHits"); + }); + + it("omits metadata entirely when the a2a detail carries none (or a non-object one)", () => { + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "succeeded", label: "x" }; + const messages = [{ role: "user", content: "route this please" }]; + const bare = JSON.parse( + String(repeatReqFor(node as never, { input: { skill: "s", messages } })?.init.body) + ); + expect("metadata" in bare.params).toBe(false); + const junk = JSON.parse( + String( + repeatReqFor(node as never, { input: { skill: "s", messages, metadata: "boom" } })?.init + .body + ) + ); + expect("metadata" in junk.params).toBe(false); + }); + + it("returns null for a2a when input.messages is empty or missing", () => { + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "succeeded", label: "x" }; + expect(repeatReqFor(node as never, { input: { skill: "s", messages: [] } })).toBeNull(); + expect(repeatReqFor(node as never, {})).toBeNull(); + }); + + it("builds the conductor repeat request against the D1 task-creation route", () => { + const node = { + id: "conductor:task:1", + kind: "work", + source: "conductor", + state: "succeeded", + label: "x", + }; + const detail = { + repo: "https://github.com/x/y", + prompt: "fix the bug", + base_ref: "main", + mode: "auto", + }; + const req = repeatReqFor(node as never, detail); + expect(req?.url).toBe("/api/conductor/tasks"); + expect(JSON.parse(String(req?.init.body))).toEqual({ + repoUrl: "https://github.com/x/y", + prompt: "fix the bug", + baseRef: "main", + mode: "auto", + }); + }); + + it("returns null for conductor when neither repo nor prompt is recoverable", () => { + const node = { + id: "conductor:task:1", + kind: "work", + source: "conductor", + state: "succeeded", + label: "x", + }; + expect(repeatReqFor(node as never, { mode: "auto" })).toBeNull(); + }); + + it("returns null for conductor when prompt is the only missing field (a hub task with repo but no spec.prompt must not POST prompt:null — HTTP 400)", () => { + const node = { + id: "conductor:task:1", + kind: "work", + source: "conductor", + state: "succeeded", + label: "x", + }; + const detail = { repo: "https://github.com/x/y", prompt: null, base_ref: "main", mode: "auto" }; + expect(repeatReqFor(node as never, detail)).toBeNull(); + }); + + it("returns null for conductor when repo is the only missing field", () => { + const node = { + id: "conductor:task:1", + kind: "work", + source: "conductor", + state: "succeeded", + label: "x", + }; + const detail = { repo: null, prompt: "fix the bug", base_ref: "main", mode: "auto" }; + expect(repeatReqFor(node as never, detail)).toBeNull(); + }); + + it("returns null for a source with no known repeat contract (runner/overflow)", () => { + const node = { id: "overflow:1", kind: "overflow", state: "succeeded", label: "x" }; + expect(repeatReqFor(node as never, {})).toBeNull(); + }); +}); + +describe("OrchestrationDrawer repeat action (two-click confirm)", () => { + const A2A_TASK = { + id: "1", + skill: "smart-routing", + state: "working", + input: { + skill: "smart-routing", + messages: [{ role: "user", content: "route this please" }], + }, + artifacts: [], + events: [], + metadata: {}, + createdAt: "x", + updatedAt: "y", + expiresAt: "z", + }; + + beforeEach(() => { + vi.useFakeTimers(); + }); + afterEach(() => { + vi.useRealTimers(); + }); + + function findRepeatButton(c: HTMLElement): HTMLButtonElement { + return Array.from(c.querySelectorAll("button")).find( + (b) => b.textContent?.includes("actionRepeat") || b.textContent?.includes("repeatConfirm") + ) as HTMLButtonElement; + } + + it("disables the repeat button with the repeatUnavailable tooltip when the input cannot be recovered", async () => { + const unrecoverable = { ...A2A_TASK, input: { skill: "smart-routing", messages: [] } }; + vi.stubGlobal( + "fetch", + vi.fn(() => + Promise.resolve({ ok: true, json: () => Promise.resolve({ task: unrecoverable }) }) + ) + ); + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "running", label: "x" }; + const { c, cleanup } = render( + {}} onActionDone={() => {}} /> + ); + await act(async () => { + await Promise.resolve(); + await Promise.resolve(); + }); + const btn = findRepeatButton(c); + expect(btn.disabled).toBe(true); + expect(btn.getAttribute("title")).toBe("repeatUnavailable"); + cleanup(); + }); + + it("first click arms the confirm label without posting; second click within the window posts and reports success", async () => { + const fetchMock = vi.fn((_url: string, init?: RequestInit) => { + if (init?.method === "POST") { + return Promise.resolve({ ok: true, json: () => Promise.resolve({}) }); + } + return Promise.resolve({ ok: true, json: () => Promise.resolve({ task: A2A_TASK }) }); + }); + vi.stubGlobal("fetch", fetchMock); + let done = false; + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "running", label: "x" }; + const { c, cleanup } = render( + {}} + onActionDone={() => { + done = true; + }} + /> + ); + await act(async () => { + await Promise.resolve(); + await Promise.resolve(); + }); + + await act(async () => { + findRepeatButton(c).click(); + }); + expect(fetchMock.mock.calls.some(([, init]) => (init as RequestInit)?.method === "POST")).toBe( + false + ); + expect(findRepeatButton(c).textContent).toContain("repeatConfirm"); + + await act(async () => { + findRepeatButton(c).click(); + await Promise.resolve(); + await Promise.resolve(); + }); + const post = fetchMock.mock.calls.find(([, init]) => (init as RequestInit)?.method === "POST"); + expect(post).toBeTruthy(); + expect(post![0]).toBe("/a2a"); + expect(JSON.parse(String((post![1] as RequestInit).body))).toEqual({ + jsonrpc: "2.0", + id: "a2a:1", + method: "message/send", + params: { + skill: "smart-routing", + messages: [{ role: "user", content: "route this please" }], + }, + }); + expect(done).toBe(true); + expect(c.textContent).toContain("repeatDone"); + cleanup(); + }); + + it("resets the confirm label back to actionRepeat after 3s with no second click", async () => { + vi.stubGlobal( + "fetch", + vi.fn(() => Promise.resolve({ ok: true, json: () => Promise.resolve({ task: A2A_TASK }) })) + ); + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "running", label: "x" }; + const { c, cleanup } = render( + {}} onActionDone={() => {}} /> + ); + await act(async () => { + await Promise.resolve(); + await Promise.resolve(); + }); + + await act(async () => { + findRepeatButton(c).click(); + }); + expect(findRepeatButton(c).textContent).toContain("repeatConfirm"); + + await act(async () => { + vi.advanceTimersByTime(3000); + }); + expect(findRepeatButton(c).textContent).toContain("actionRepeat"); + cleanup(); + }); + + it("a click after the 3s window expired re-arms the confirm instead of posting (it is a fresh first click, not a stale second click)", async () => { + const fetchMock = vi.fn((_url: string, init?: RequestInit) => { + if (init?.method === "POST") { + return Promise.resolve({ ok: true, json: () => Promise.resolve({}) }); + } + return Promise.resolve({ ok: true, json: () => Promise.resolve({ task: A2A_TASK }) }); + }); + vi.stubGlobal("fetch", fetchMock); + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "running", label: "x" }; + const { c, cleanup } = render( + {}} onActionDone={() => {}} /> + ); + await act(async () => { + await Promise.resolve(); + await Promise.resolve(); + }); + + await act(async () => { + findRepeatButton(c).click(); + }); + expect(findRepeatButton(c).textContent).toContain("repeatConfirm"); + + await act(async () => { + vi.advanceTimersByTime(3000); + }); + expect(findRepeatButton(c).textContent).toContain("actionRepeat"); + + // The window has expired — this click must be treated as a fresh first click + // (arm + wait), never as the stale second click that would fire the POST. + await act(async () => { + findRepeatButton(c).click(); + }); + expect(fetchMock.mock.calls.some(([, init]) => (init as RequestInit)?.method === "POST")).toBe( + false + ); + expect(findRepeatButton(c).textContent).toContain("repeatConfirm"); + cleanup(); + }); + + it("clears the pending 3s confirm timer on unmount so it can never fire after teardown", async () => { + vi.stubGlobal( + "fetch", + vi.fn(() => Promise.resolve({ ok: true, json: () => Promise.resolve({ task: A2A_TASK }) })) + ); + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "running", label: "x" }; + const { c, cleanup } = render( + {}} onActionDone={() => {}} /> + ); + await act(async () => { + await Promise.resolve(); + await Promise.resolve(); + }); + + await act(async () => { + findRepeatButton(c).click(); + }); + expect(vi.getTimerCount()).toBeGreaterThan(0); + + cleanup(); + expect(vi.getTimerCount()).toBe(0); + }); + + it("shows actionFailed when the repeat POST fails, without touching onActionDone", async () => { + const fetchMock = vi.fn((_url: string, init?: RequestInit) => { + if (init?.method === "POST") { + return Promise.resolve({ ok: false, status: 500, json: () => Promise.resolve({}) }); + } + return Promise.resolve({ ok: true, json: () => Promise.resolve({ task: A2A_TASK }) }); + }); + vi.stubGlobal("fetch", fetchMock); + let done = false; + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "running", label: "x" }; + const { c, cleanup } = render( + {}} + onActionDone={() => { + done = true; + }} + /> + ); + await act(async () => { + await Promise.resolve(); + await Promise.resolve(); + }); + + await act(async () => { + findRepeatButton(c).click(); + }); + await act(async () => { + findRepeatButton(c).click(); + await Promise.resolve(); + await Promise.resolve(); + }); + expect(c.textContent).toContain("actionFailed"); + expect(done).toBe(false); + cleanup(); + }); + it("does not send the previous run's memoryHits in the repeat POST body", async () => { + const withHits = { + ...A2A_TASK, + input: { + ...A2A_TASK.input, + metadata: { + role: "general", + memoryHits: [{ id: "m1", key: "k1", type: "factual", snippet: "leaked" }], + }, + }, + }; + const fetchMock = vi.fn((_url: string, init?: RequestInit) => { + if (init?.method === "POST") { + return Promise.resolve({ ok: true, json: () => Promise.resolve({}) }); + } + return Promise.resolve({ ok: true, json: () => Promise.resolve({ task: withHits }) }); + }); + vi.stubGlobal("fetch", fetchMock); + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "running", label: "x" }; + const { c, cleanup } = render( + {}} onActionDone={() => {}} /> + ); + await act(async () => { + await Promise.resolve(); + await Promise.resolve(); + }); + + await act(async () => { + findRepeatButton(c).click(); + }); + await act(async () => { + findRepeatButton(c).click(); + await Promise.resolve(); + await Promise.resolve(); + }); + const post = fetchMock.mock.calls.find(([, init]) => (init as RequestInit)?.method === "POST"); + expect(post).toBeTruthy(); + const body = JSON.parse(String((post![1] as RequestInit).body)); + expect(body.params.metadata).toEqual({ role: "general" }); + expect(String((post![1] as RequestInit).body)).not.toContain("memoryHits"); + cleanup(); + }); + + it("treats a JSON-RPC error answered with HTTP 200 as a failure, never as a success toast", async () => { + // `/a2a` maps most JSON-RPC error codes to `status: 200` (src/app/a2a/route.ts), so + // `res.ok` alone would report a run that never happened as done. + const fetchMock = vi.fn((_url: string, init?: RequestInit) => { + if (init?.method === "POST") { + return Promise.resolve({ + ok: true, + status: 200, + json: () => + Promise.resolve({ + jsonrpc: "2.0", + id: "a2a:1", + error: { code: -32602, message: "segredo interno que NAO pode vazar" }, + }), + }); + } + return Promise.resolve({ ok: true, json: () => Promise.resolve({ task: A2A_TASK }) }); + }); + vi.stubGlobal("fetch", fetchMock); + let done = false; + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "running", label: "x" }; + const { c, cleanup } = render( + {}} + onActionDone={() => { + done = true; + }} + /> + ); + await act(async () => { + await Promise.resolve(); + await Promise.resolve(); + }); + + await act(async () => { + findRepeatButton(c).click(); + }); + await act(async () => { + findRepeatButton(c).click(); + await Promise.resolve(); + await Promise.resolve(); + await Promise.resolve(); + }); + expect(done).toBe(false); + expect(c.textContent).not.toContain("repeatDone"); + expect(c.textContent).toContain("actionFailed"); + expect(c.textContent).toContain("RPC -32602"); + expect(c.textContent).not.toContain("segredo interno"); + cleanup(); + }); + + it("surfaces the sanitized HTTP status when a secured deployment rejects the a2a repeat (HTTP 400)", async () => { + // With REQUIRE_API_KEY / OMNIROUTE_API_KEY set, `/a2a` answers -32600 => HTTP 400 to a + // dashboard-session caller. The drawer must say so instead of pretending success. + const fetchMock = vi.fn((_url: string, init?: RequestInit) => { + if (init?.method === "POST") { + return Promise.resolve({ ok: false, status: 400, json: () => Promise.resolve({}) }); + } + return Promise.resolve({ ok: true, json: () => Promise.resolve({ task: A2A_TASK }) }); + }); + vi.stubGlobal("fetch", fetchMock); + const node = { id: "a2a:1", kind: "work", source: "a2a", state: "running", label: "x" }; + const { c, cleanup } = render( + {}} onActionDone={() => {}} /> + ); + await act(async () => { + await Promise.resolve(); + await Promise.resolve(); + }); + await act(async () => { + findRepeatButton(c).click(); + }); + await act(async () => { + findRepeatButton(c).click(); + await Promise.resolve(); + await Promise.resolve(); + }); + expect(c.textContent).toContain("actionFailed"); + expect(c.textContent).toContain("HTTP 400"); + expect(c.textContent).not.toContain("repeatDone"); + cleanup(); + }); +}); diff --git a/tests/unit/ui/orchestrationHistoryTab.test.tsx b/tests/unit/ui/orchestrationHistoryTab.test.tsx index c27f4a57c4..ceee2897ca 100644 --- a/tests/unit/ui/orchestrationHistoryTab.test.tsx +++ b/tests/unit/ui/orchestrationHistoryTab.test.tsx @@ -250,6 +250,36 @@ describe("HistoryTab", () => { cleanup(); }); + it("keeps the drawer open on onActionDone so its success toast is visible, and refetches", async () => { + // Review finding (Minor B): this tab used to pass `onActionDone={() => setSelected(null)}`, + // which unmounted the drawer BEFORE it rendered the repeat/cancel confirmation — the + // operator saw the action silently do nothing. The callback must keep the drawer mounted + // (and re-sample the range so the new run shows up). + const fetchMock = mockFetch({ a2aTasks: [a2aTask()] }); + vi.stubGlobal("fetch", fetchMock); + const { c, cleanup } = render(); + await flush(); + + const cell = c.querySelector('button[aria-label*="smart-routing"]') as HTMLButtonElement; + act(() => { + cell.click(); + }); + expect((drawerCalls.at(-1) as { node: unknown }).node).toBeTruthy(); + const callsBefore = fetchMock.mock.calls.length; + + await act(async () => { + (drawerCalls.at(-1) as { onActionDone: () => void }).onActionDone(); + }); + await flush(); + + // Still open on the same node … + const last = drawerCalls.at(-1) as { node: { id: string } | null }; + expect(last.node?.id).toBe("a2a:t1"); + // … and the history was refetched. + expect(fetchMock.mock.calls.length).toBeGreaterThan(callsBefore); + cleanup(); + }); + it("shows a source-failed warning for A2A while Cloud Agent rows still render", async () => { vi.stubGlobal("fetch", mockFetch({ a2aFail: true, cloudAgentTasks: [cloudAgentTask()] })); const { c, cleanup } = render(); From 1a0375fba308c9b189d49b6ac40cf56ff2c354e6 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Wed, 2 Sep 2026 21:20:40 -0300 Subject: [PATCH 007/143] chore(quality): tighten the CodeQL ratchet baseline from 11 to 6 (#12530) Co-authored-by: Markus Hartung --- config/quality/quality-baseline.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/config/quality/quality-baseline.json b/config/quality/quality-baseline.json index ba75b42ac5..f5f9d56bf2 100644 --- a/config/quality/quality-baseline.json +++ b/config/quality/quality-baseline.json @@ -157,7 +157,7 @@ "dedicatedGate": true }, "codeqlAlerts": { - "value": 11, + "value": 6, "direction": "down", "dedicatedGate": true, "_rebaseline_2026_08_06_base_grew": "Base branch file-size drift: translator-openai-to-gemini.test.ts grew 1619->1622 (test assertions for Gemini translator compatibility). CodeQL alert (js/insufficient-password-hash in raycast.ts) is pre-existing base-red; incremented baseline to match.", From fcddea78989d31f763ae55bc9c04bce971b8d2da Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:18:59 -0300 Subject: [PATCH 008/143] deps: bump @humanfs/node from 0.16.7 to 0.16.8 (#12515) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 7 bumps da raiz sobre o tip de release/v3.8.51: npm install reconciliou o lock sem diff (11/11 pacotes na versão alvo), typecheck:core e typecheck:noimplicit:core limpos, lint exit 0, e 149/149 testes nos 11 arquivos que exercitam zod diretamente. As falhas de CI do PR foram discriminadas como estado da base, não do bump. Obrigado, Dependabot. --- package-lock.json | 28 +++++++++++++++++++++------- 1 file changed, 21 insertions(+), 7 deletions(-) diff --git a/package-lock.json b/package-lock.json index 270593d514..4b4739e580 100644 --- a/package-lock.json +++ b/package-lock.json @@ -4578,29 +4578,43 @@ } }, "node_modules/@humanfs/core": { - "version": "0.19.1", - "resolved": "https://registry.npmjs.org/@humanfs/core/-/core-0.19.1.tgz", - "integrity": "sha512-5DyQ4+1JEUzejeK1JGICcideyfUbGixgS9jNgex5nqkW+cY7WZhxBigmieN5Qnw9ZosSNVC9KQKyb+GUaGyKUA==", + "version": "0.19.2", + "resolved": "https://registry.npmjs.org/@humanfs/core/-/core-0.19.2.tgz", + "integrity": "sha512-UhXNm+CFMWcbChXywFwkmhqjs3PRCmcSa/hfBgLIb7oQ5HNb1wS0icWsGtSAUNgefHeI+eBrA8I1fxmbHsGdvA==", "dev": true, "license": "Apache-2.0", + "dependencies": { + "@humanfs/types": "^0.15.0" + }, "engines": { "node": ">=18.18.0" } }, "node_modules/@humanfs/node": { - "version": "0.16.7", - "resolved": "https://registry.npmjs.org/@humanfs/node/-/node-0.16.7.tgz", - "integrity": "sha512-/zUx+yOsIrG4Y43Eh2peDeKCxlRt/gET6aHfaKpuq267qXdYDFViVHfMaLyygZOnl0kGWxFIgsBy8QFuTLUXEQ==", + "version": "0.16.8", + "resolved": "https://registry.npmjs.org/@humanfs/node/-/node-0.16.8.tgz", + "integrity": "sha512-gE1eQNZ3R++kTzFUpdGlpmy8kDZD/MLyHqDwqjkVQI0JMdI1D51sy1H958PNXYkM2rAac7e5/CnIKZrHtPh3BQ==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@humanfs/core": "^0.19.1", + "@humanfs/core": "^0.19.2", + "@humanfs/types": "^0.15.0", "@humanwhocodes/retry": "^0.4.0" }, "engines": { "node": ">=18.18.0" } }, + "node_modules/@humanfs/types": { + "version": "0.15.0", + "resolved": "https://registry.npmjs.org/@humanfs/types/-/types-0.15.0.tgz", + "integrity": "sha512-ZZ1w0aoQkwuUuC7Yf+7sdeaNfqQiiLcSRbfI08oAxqLtpXQr9AIVX7Ay7HLDuiLYAaFPu8oBYNq/QIi9URHJ3Q==", + "dev": true, + "license": "Apache-2.0", + "engines": { + "node": ">=18.18.0" + } + }, "node_modules/@humanwhocodes/module-importer": { "version": "1.0.1", "resolved": "https://registry.npmjs.org/@humanwhocodes/module-importer/-/module-importer-1.0.1.tgz", From d6770bda0c02780d423555b62513c566ddf5bf42 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:19:31 -0300 Subject: [PATCH 009/143] deps: bump @xmldom/xmldom from 0.9.10 to 0.9.12 (#12513) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 7 bumps da raiz sobre o tip de release/v3.8.51: npm install reconciliou o lock sem diff (11/11 pacotes na versão alvo), typecheck:core e typecheck:noimplicit:core limpos, lint exit 0, e 149/149 testes nos 11 arquivos que exercitam zod diretamente. As falhas de CI do PR foram discriminadas como estado da base, não do bump. Obrigado, Dependabot. --- package-lock.json | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/package-lock.json b/package-lock.json index 4b4739e580..6c29cc0847 100644 --- a/package-lock.json +++ b/package-lock.json @@ -15123,9 +15123,9 @@ } }, "node_modules/@xmldom/xmldom": { - "version": "0.9.10", - "resolved": "https://registry.npmjs.org/@xmldom/xmldom/-/xmldom-0.9.10.tgz", - "integrity": "sha512-A9gOqLdi6cV4ibazAjcQufGj0B1y/vDqYrcuP6d/6x8P27gRS8643Dj9o1dEKtB6O7fwxb2FgBmJS2mX7gpvdw==", + "version": "0.9.12", + "resolved": "https://registry.npmjs.org/@xmldom/xmldom/-/xmldom-0.9.12.tgz", + "integrity": "sha512-5AXjrcMClTryPe9LgZrygpB1lj7s0S9E0+W+AHaVKAVyHanafK86iPSvG5xHVSp/jC+VH1UXu0TAEmY279xH7A==", "dev": true, "license": "MIT", "engines": { From a986ef2e2bd436e7e73a612bf13772eea918eff8 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:19:49 -0300 Subject: [PATCH 010/143] deps: bump browserslist from 4.28.2 to 4.28.8 (#12396) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 7 bumps da raiz sobre o tip de release/v3.8.51: npm install reconciliou o lock sem diff (11/11 pacotes na versão alvo), typecheck:core e typecheck:noimplicit:core limpos, lint exit 0, e 149/149 testes nos 11 arquivos que exercitam zod diretamente. As falhas de CI do PR foram discriminadas como estado da base, não do bump. Obrigado, Dependabot. --- package-lock.json | 46 +++++++++++++++++++++++----------------------- 1 file changed, 23 insertions(+), 23 deletions(-) diff --git a/package-lock.json b/package-lock.json index 6c29cc0847..308eb97730 100644 --- a/package-lock.json +++ b/package-lock.json @@ -16266,9 +16266,9 @@ } }, "node_modules/baseline-browser-mapping": { - "version": "2.10.13", - "resolved": "https://registry.npmjs.org/baseline-browser-mapping/-/baseline-browser-mapping-2.10.13.tgz", - "integrity": "sha512-BL2sTuHOdy0YT1lYieUxTw/QMtPBC3pmlJC6xk8BBYVv6vcw3SGdKemQ+Xsx9ik2F/lYDO9tqsFQH1r9PFuHKw==", + "version": "2.11.20", + "resolved": "https://registry.npmjs.org/baseline-browser-mapping/-/baseline-browser-mapping-2.11.20.tgz", + "integrity": "sha512-H0ulySigv6icDJ1F7SjtdCD6PrhTpdYCmP0CactWy1+ekh0AFd0o1Wn5T8b+hnTmdBx19u9yhL6wvCylXMY7zw==", "license": "Apache-2.0", "bin": { "baseline-browser-mapping": "dist/cli.cjs" @@ -16626,9 +16626,9 @@ } }, "node_modules/browserslist": { - "version": "4.28.2", - "resolved": "https://registry.npmjs.org/browserslist/-/browserslist-4.28.2.tgz", - "integrity": "sha512-48xSriZYYg+8qXna9kwqjIVzuQxi+KYWp2+5nCYnYKPTr0LvD89Jqk2Or5ogxz0NUMfIjhh2lIUX/LyX9B4oIg==", + "version": "4.28.8", + "resolved": "https://registry.npmjs.org/browserslist/-/browserslist-4.28.8.tgz", + "integrity": "sha512-V2NpofLblG64mfOtSgDhOJESZEGogzDMBv/q+W6oc4LXWP/q75eOXoOaaOu1EOadB9U4Bwx/e0yzbvwKH8zalA==", "dev": true, "funding": [ { @@ -16646,11 +16646,11 @@ ], "license": "MIT", "dependencies": { - "baseline-browser-mapping": "^2.10.12", - "caniuse-lite": "^1.0.30001782", - "electron-to-chromium": "^1.5.328", - "node-releases": "^2.0.36", - "update-browserslist-db": "^1.2.3" + "baseline-browser-mapping": "^2.11.12", + "caniuse-lite": "^1.0.30001809", + "electron-to-chromium": "^1.5.402", + "node-releases": "^2.0.53", + "update-browserslist-db": "^1.3.0" }, "bin": { "browserslist": "cli.js" @@ -17100,9 +17100,9 @@ } }, "node_modules/caniuse-lite": { - "version": "1.0.30001784", - "resolved": "https://registry.npmjs.org/caniuse-lite/-/caniuse-lite-1.0.30001784.tgz", - "integrity": "sha512-WU346nBTklUV9YfUl60fqRbU5ZqyXlqvo1SgigE1OAXK5bFL8LL9q1K7aap3N739l4BvNqnkm3YrGHiY9sfUQw==", + "version": "1.0.30001810", + "resolved": "https://registry.npmjs.org/caniuse-lite/-/caniuse-lite-1.0.30001810.tgz", + "integrity": "sha512-TITQPUkaz+aVk5GL6NhOdwk1aEaNTSDPsGFWrTuhKGtjTF70jL/Oht2W4c6rXUe5fu7Ie19VIahAXHIIiWWNeg==", "funding": [ { "type": "opencollective", @@ -20018,9 +20018,9 @@ "license": "MIT" }, "node_modules/electron-to-chromium": { - "version": "1.5.375", - "resolved": "https://registry.npmjs.org/electron-to-chromium/-/electron-to-chromium-1.5.375.tgz", - "integrity": "sha512-ZWP5eB4BVPW/ZYo9252hQZHZ5XavtsTgpbhcmMmRwymavC5AsLWQWBPaKMeNd2LW0KGby5HPXvj7+sr4ta5j/Q==", + "version": "1.5.420", + "resolved": "https://registry.npmjs.org/electron-to-chromium/-/electron-to-chromium-1.5.420.tgz", + "integrity": "sha512-2yD6XreGusOfNV+dUcvipJEXc3n/n7fgr7996aszTG+YY5E4mqM4tOq/3uhP129cazL9YHbVWSpc79ePotWtPA==", "dev": true, "license": "ISC" }, @@ -30776,9 +30776,9 @@ "license": "MIT" }, "node_modules/node-releases": { - "version": "2.0.47", - "resolved": "https://registry.npmjs.org/node-releases/-/node-releases-2.0.47.tgz", - "integrity": "sha512-Uzmd6LXpouKo8EUK68IjH4+E01w/hXyV3R3g/geCJo+rXLNfh1xucB+LOzYEOQPSiUK3h/xZf0cQGcSsmyL2Og==", + "version": "2.0.54", + "resolved": "https://registry.npmjs.org/node-releases/-/node-releases-2.0.54.tgz", + "integrity": "sha512-YHs7BmmcsdAI5Ozuf8JZo6PT0mv2GIWC9vMfvUC3dp65M8hn7Ux8CPL+2oBI7juNuj9d0ndhTcznq2ODBps9cQ==", "dev": true, "license": "MIT", "engines": { @@ -38352,9 +38352,9 @@ } }, "node_modules/update-browserslist-db": { - "version": "1.2.3", - "resolved": "https://registry.npmjs.org/update-browserslist-db/-/update-browserslist-db-1.2.3.tgz", - "integrity": "sha512-Js0m9cx+qOgDxo0eMiFGEueWztz+d4+M3rGlmKPT+T4IS/jP4ylw3Nwpu6cpTTP8R1MAC1kF4VbdLt3ARf209w==", + "version": "1.3.2", + "resolved": "https://registry.npmjs.org/update-browserslist-db/-/update-browserslist-db-1.3.2.tgz", + "integrity": "sha512-UQ+MSxlhRm1bzjhU+DcuXfjFO1FzNtqhK5+9Yvlp90ItDLk5vT932A0rFu619nf7RVS+Y/VeaUW1jaRDqZ8VJw==", "dev": true, "funding": [ { From df97d46f48f0bad4fd1f70845ca6989ed19b2539 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:20:07 -0300 Subject: [PATCH 011/143] deps: bump fast-uri from 3.1.5 to 3.1.7 (#12514) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 7 bumps da raiz sobre o tip de release/v3.8.51: npm install reconciliou o lock sem diff (11/11 pacotes na versão alvo), typecheck:core e typecheck:noimplicit:core limpos, lint exit 0, e 149/149 testes nos 11 arquivos que exercitam zod diretamente. As falhas de CI do PR foram discriminadas como estado da base, não do bump. Obrigado, Dependabot. --- package-lock.json | 6 +++--- package.json | 2 +- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/package-lock.json b/package-lock.json index 308eb97730..ce1d92be0a 100644 --- a/package-lock.json +++ b/package-lock.json @@ -21700,9 +21700,9 @@ } }, "node_modules/fast-uri": { - "version": "3.1.5", - "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.5.tgz", - "integrity": "sha512-gHwA1O9LDIcKunMKhObS/HimwtehO1nPUECKAu5TpKgaO19fcWEl4bliWe1jWxVFvIXztJjjQ4L8XQ1EU9f7Jw==", + "version": "3.1.7", + "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.7.tgz", + "integrity": "sha512-dOvZVzjdZdz7phd9v6jCbwxrBW3fK6n8Rc0CtdmM4bumzMnxywBYhuph6J819RRw/ku+rLbelwfMunktuzVVHg==", "funding": [ { "type": "github", diff --git a/package.json b/package.json index f019c45e90..da0d54b4da 100644 --- a/package.json +++ b/package.json @@ -475,7 +475,7 @@ "@babel/core": "^7.29.6", "hono": "^4.12.34", "@hono/node-server": "^2.0.5", - "fast-uri": "^3.1.5", + "fast-uri": "^3.1.7", "body-parser": "^2.3.0", "@yarnpkg/parsers": { "js-yaml": "^4.3.1" From fa64266e32692c3fb01d11e8d023149c354269b0 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:20:24 -0300 Subject: [PATCH 012/143] deps: bump qs from 6.15.2 to 6.16.0 (#12512) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 7 bumps da raiz sobre o tip de release/v3.8.51: npm install reconciliou o lock sem diff (11/11 pacotes na versão alvo), typecheck:core e typecheck:noimplicit:core limpos, lint exit 0, e 149/149 testes nos 11 arquivos que exercitam zod diretamente. As falhas de CI do PR foram discriminadas como estado da base, não do bump. Obrigado, Dependabot. --- package-lock.json | 27 ++++++++++++++------------- package.json | 2 +- 2 files changed, 15 insertions(+), 14 deletions(-) diff --git a/package-lock.json b/package-lock.json index ce1d92be0a..6059d042ed 100644 --- a/package-lock.json +++ b/package-lock.json @@ -34034,12 +34034,13 @@ } }, "node_modules/qs": { - "version": "6.15.2", - "resolved": "https://registry.npmjs.org/qs/-/qs-6.15.2.tgz", - "integrity": "sha512-Rzq0KEyX/w/tEybncDgdkZrJgVUsUMk3xjh3t5bv3S1HTAtg+uOYt72+ZfwiQwKdysThkTBdL/rTi6HDmX9Ddw==", + "version": "6.16.0", + "resolved": "https://registry.npmjs.org/qs/-/qs-6.16.0.tgz", + "integrity": "sha512-h6fhOIaRrID2CbEY2fqs+7t+UXZo+MLAnU5gRIq85uFtdiUPCdsApMlHhXogKVM4HM2DVbIjGNTTYH2OcmP1vA==", "license": "BSD-3-Clause", "dependencies": { - "side-channel": "^1.1.0" + "es-define-property": "^1.0.1", + "side-channel": "^1.1.1" }, "engines": { "node": ">=0.6" @@ -35739,14 +35740,14 @@ } }, "node_modules/side-channel": { - "version": "1.1.0", - "resolved": "https://registry.npmjs.org/side-channel/-/side-channel-1.1.0.tgz", - "integrity": "sha512-ZX99e6tRweoUXqR+VBrslhda51Nh5MTQwou5tnUDgbtyM0dBgmhEDtWGP/xbKn6hqfPRHujUNwz5fy/wbbhnpw==", + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/side-channel/-/side-channel-1.1.1.tgz", + "integrity": "sha512-6x6dK6zJdpTzF4sQeNYxwtvBzf6Eg4GtlesS94HOvTudUeyK2WXAaIfmDgsyslYrRBeFIlsi54AYsFGUuhmvrQ==", "license": "MIT", "dependencies": { "es-errors": "^1.3.0", - "object-inspect": "^1.13.3", - "side-channel-list": "^1.0.0", + "object-inspect": "^1.13.4", + "side-channel-list": "^1.0.1", "side-channel-map": "^1.0.1", "side-channel-weakmap": "^1.0.2" }, @@ -35758,13 +35759,13 @@ } }, "node_modules/side-channel-list": { - "version": "1.0.0", - "resolved": "https://registry.npmjs.org/side-channel-list/-/side-channel-list-1.0.0.tgz", - "integrity": "sha512-FCLHtRD/gnpCiCHEiJLOwdmFP+wzCmDEkc9y7NsYxeF4u7Btsn1ZuwgwJGxImImHicJArLP4R0yX4c2KCrMrTA==", + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/side-channel-list/-/side-channel-list-1.0.1.tgz", + "integrity": "sha512-mjn/0bi/oUURjc5Xl7IaWi/OJJJumuoJFQJfDDyO46+hBWsfaVM65TBHq2eoZBhzl9EchxOijpkbRC8SVBQU0w==", "license": "MIT", "dependencies": { "es-errors": "^1.3.0", - "object-inspect": "^1.13.3" + "object-inspect": "^1.13.4" }, "engines": { "node": ">= 0.4" diff --git a/package.json b/package.json index da0d54b4da..af8d67e748 100644 --- a/package.json +++ b/package.json @@ -467,7 +467,7 @@ "sharp": "^0.35.4", "postcss": "^8.5.18", "ip-address": "^10.3.1", - "qs": "^6.15.2", + "qs": "^6.16.0", "uuid": "^14.0.2", "form-data": "^4.0.6", "vite": "^8.0.16", From 60ea5f8858a3740b11e9d844f8b7f133bad7b757 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:20:42 -0300 Subject: [PATCH 013/143] deps: bump the development group across 1 directory with 2 updates (#12347) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 7 bumps da raiz sobre o tip de release/v3.8.51: npm install reconciliou o lock sem diff (11/11 pacotes na versão alvo), typecheck:core e typecheck:noimplicit:core limpos, lint exit 0, e 149/149 testes nos 11 arquivos que exercitam zod diretamente. As falhas de CI do PR foram discriminadas como estado da base, não do bump. Obrigado, Dependabot. --- package-lock.json | 110 +++++++++++++++++++++++----------------------- package.json | 2 +- 2 files changed, 56 insertions(+), 56 deletions(-) diff --git a/package-lock.json b/package-lock.json index 6059d042ed..41a4c03dba 100644 --- a/package-lock.json +++ b/package-lock.json @@ -148,7 +148,7 @@ "lint-staged": "^17.4.1", "lockfile-lint": "^5.0.1", "node-loader": "^2.1.0", - "opencode-ai": "1.18.23", + "opencode-ai": "1.18.25", "playwright-ctrf-json-reporter": "^0.0.29", "prettier": "^3.9.6", "promptfoo": "^0.122.1", @@ -14817,9 +14817,9 @@ } }, "node_modules/@vitejs/plugin-react": { - "version": "6.1.0", - "resolved": "https://registry.npmjs.org/@vitejs/plugin-react/-/plugin-react-6.1.0.tgz", - "integrity": "sha512-qd2BzUBehkov86WFhg0JkEFEYyCLG9uPCe6qWTY/kRlss9OvJrOF2UbIWT7p+8IzZHkEu0DNGHc4HSv+JdDLsw==", + "version": "6.1.1", + "resolved": "https://registry.npmjs.org/@vitejs/plugin-react/-/plugin-react-6.1.1.tgz", + "integrity": "sha512-yxLaQV9gkhS8ezJqCM6+ndU7mDY6gqAg75NQ+0IjwEI8IYOmQCgkRwHKVSfWXW076DsqMo0Dk+0FK1U+M5RgFw==", "dev": true, "license": "MIT", "dependencies": { @@ -31451,9 +31451,9 @@ } }, "node_modules/opencode-ai": { - "version": "1.18.23", - "resolved": "https://registry.npmjs.org/opencode-ai/-/opencode-ai-1.18.23.tgz", - "integrity": "sha512-3NkT0XINL7d0HYkTyGV1SPChHXhvRgKqNaTgKRTGb0TXUWszXA7MW/y3zMZw29y1AQuUDAzRvVYmQ9KGRQhroA==", + "version": "1.18.25", + "resolved": "https://registry.npmjs.org/opencode-ai/-/opencode-ai-1.18.25.tgz", + "integrity": "sha512-pS4RKJ9eKwU7Dp5G5pdj1rhMnpG5APixXzfTKNoFqv9aFVI36Rnza2jESvKifxyPZlsA65MQB03WCArY0EK6mg==", "cpu": [ "arm64", "x64" @@ -31470,24 +31470,24 @@ "opencode": "bin/opencode.exe" }, "optionalDependencies": { - "opencode-darwin-arm64": "1.18.23", - "opencode-darwin-x64": "1.18.23", - "opencode-darwin-x64-baseline": "1.18.23", - "opencode-linux-arm64": "1.18.23", - "opencode-linux-arm64-musl": "1.18.23", - "opencode-linux-x64": "1.18.23", - "opencode-linux-x64-baseline": "1.18.23", - "opencode-linux-x64-baseline-musl": "1.18.23", - "opencode-linux-x64-musl": "1.18.23", - "opencode-windows-arm64": "1.18.23", - "opencode-windows-x64": "1.18.23", - "opencode-windows-x64-baseline": "1.18.23" + "opencode-darwin-arm64": "1.18.25", + "opencode-darwin-x64": "1.18.25", + "opencode-darwin-x64-baseline": "1.18.25", + "opencode-linux-arm64": "1.18.25", + "opencode-linux-arm64-musl": "1.18.25", + "opencode-linux-x64": "1.18.25", + "opencode-linux-x64-baseline": "1.18.25", + "opencode-linux-x64-baseline-musl": "1.18.25", + "opencode-linux-x64-musl": "1.18.25", + "opencode-windows-arm64": "1.18.25", + "opencode-windows-x64": "1.18.25", + "opencode-windows-x64-baseline": "1.18.25" } }, "node_modules/opencode-darwin-arm64": { - "version": "1.18.23", - "resolved": "https://registry.npmjs.org/opencode-darwin-arm64/-/opencode-darwin-arm64-1.18.23.tgz", - "integrity": "sha512-QP9PjwpHtZoLVXw2WvUmPZecz7mWbQkT4t3K36B//fCaDG+zWa+SsztIeaW5azujNwwtUemLA5icE/zINng48Q==", + "version": "1.18.25", + "resolved": "https://registry.npmjs.org/opencode-darwin-arm64/-/opencode-darwin-arm64-1.18.25.tgz", + "integrity": "sha512-W4dyMFtHBglWZ1SEooh3Ke9v1M9lv945Y58atb8e1yKII8YykJ8LknOFyKipYC028oPDO4IZc3GYGKbg9PCg2w==", "cpu": [ "arm64" ], @@ -31498,9 +31498,9 @@ ] }, "node_modules/opencode-darwin-x64": { - "version": "1.18.23", - "resolved": "https://registry.npmjs.org/opencode-darwin-x64/-/opencode-darwin-x64-1.18.23.tgz", - "integrity": "sha512-R9nWP3edz/0FnEfwmuxtiWBB7bS4NtZCyCffJyiMlrbwdDC+bIXYrWxWXVrzaP1mJujs6g2MAwTCUSx/qpBhDw==", + "version": "1.18.25", + "resolved": "https://registry.npmjs.org/opencode-darwin-x64/-/opencode-darwin-x64-1.18.25.tgz", + "integrity": "sha512-YYKrfeUSJhD7hZl+yNmayS51sDwxiE9o5XwrfgYSSie6sOyHFc9Ei13VBkVU6T+IJHhFhTahOFAwSDxggrAnGA==", "cpu": [ "x64" ], @@ -31511,9 +31511,9 @@ ] }, "node_modules/opencode-darwin-x64-baseline": { - "version": "1.18.23", - "resolved": "https://registry.npmjs.org/opencode-darwin-x64-baseline/-/opencode-darwin-x64-baseline-1.18.23.tgz", - "integrity": "sha512-QGx6I/nFYur7qJ/Nx2L3fC4XYQt44cyDsm7p8twNA+cdjGX3ndnPbMdAl5ikdZyAfSMUGYK8VWY2JMxv0rmfjw==", + "version": "1.18.25", + "resolved": "https://registry.npmjs.org/opencode-darwin-x64-baseline/-/opencode-darwin-x64-baseline-1.18.25.tgz", + "integrity": "sha512-rRgTaoTeIN2diL1e1HGZ48Zh4ynMDEB1jYjD76LaFUVzMwakEY1i7NvG8e/rbjRMDkgXIr6TwzCtIMcMOpLQMA==", "cpu": [ "x64" ], @@ -31524,9 +31524,9 @@ ] }, "node_modules/opencode-linux-arm64": { - "version": "1.18.23", - "resolved": "https://registry.npmjs.org/opencode-linux-arm64/-/opencode-linux-arm64-1.18.23.tgz", - "integrity": "sha512-g1zDFhuE9FOYwjSGderlu69wfd4GQzS0xsDiIY11QUciuBmM6DrHqLvmhlLFsUVHYVnCPY1YeN1pq5ewE3x72Q==", + "version": "1.18.25", + "resolved": "https://registry.npmjs.org/opencode-linux-arm64/-/opencode-linux-arm64-1.18.25.tgz", + "integrity": "sha512-PMvcpFpha3yAhaVCC0QbegHPxsEZ0FuQf+52PXvqQut1r3w1l1Pilor9tUA7TyCRa4UokACI90nTmKmtMnQBag==", "cpu": [ "arm64" ], @@ -31537,9 +31537,9 @@ ] }, "node_modules/opencode-linux-arm64-musl": { - "version": "1.18.23", - "resolved": "https://registry.npmjs.org/opencode-linux-arm64-musl/-/opencode-linux-arm64-musl-1.18.23.tgz", - "integrity": "sha512-VyDkzUJfJgkx9h9RhazTW9xeTgSXBmVFPblsbPGBW9tR612f6gjQxfOfu4cpHnQHK1qjuW9ClzLKHWqv3EcJTA==", + "version": "1.18.25", + "resolved": "https://registry.npmjs.org/opencode-linux-arm64-musl/-/opencode-linux-arm64-musl-1.18.25.tgz", + "integrity": "sha512-IwIPKmNwIjLshlSgjoRLKFwxxiLpZ5Y0zjv6r456RQtJKK62IbYGXkCm3AiSZ/lqGsu3XF+xn/Xza29ivgpgcg==", "cpu": [ "arm64" ], @@ -31553,9 +31553,9 @@ ] }, "node_modules/opencode-linux-x64": { - "version": "1.18.23", - "resolved": "https://registry.npmjs.org/opencode-linux-x64/-/opencode-linux-x64-1.18.23.tgz", - "integrity": "sha512-5x9d1Cm/YtqzR6lAlNbgVprTQ3R3hx7qGTWCzm5l5u6lBNkhYTrhy2s8k25dxKKuxqZ9Kngqz9JYWvSVHy2Lmw==", + "version": "1.18.25", + "resolved": "https://registry.npmjs.org/opencode-linux-x64/-/opencode-linux-x64-1.18.25.tgz", + "integrity": "sha512-bdRSJ6gbK/EnLNWxROOQYXFXiUeqeFxGz8DIO8LCqnii99A2OWFAyZ3Da5gpvfT1Yrp9/lYL55n/tM3ale5smg==", "cpu": [ "x64" ], @@ -31566,9 +31566,9 @@ ] }, "node_modules/opencode-linux-x64-baseline": { - "version": "1.18.23", - "resolved": "https://registry.npmjs.org/opencode-linux-x64-baseline/-/opencode-linux-x64-baseline-1.18.23.tgz", - "integrity": "sha512-yUhBOXfTQour2JCdAkwD3DDqSnyxB0grefwdPqEhYmJHIkYxfJIIzyy6V//pyouvkE0XMouFtiuZXw8S6Wo0iQ==", + "version": "1.18.25", + "resolved": "https://registry.npmjs.org/opencode-linux-x64-baseline/-/opencode-linux-x64-baseline-1.18.25.tgz", + "integrity": "sha512-+b0w7XyHx0XPQWHBk2JymXbXnyZQ2PjIPuu4a4QJgSUqGuGz1L2flA3wgpZVAWFUhrEIr9DFhBk3AkKKNgMuRw==", "cpu": [ "x64" ], @@ -31579,9 +31579,9 @@ ] }, "node_modules/opencode-linux-x64-baseline-musl": { - "version": "1.18.23", - "resolved": "https://registry.npmjs.org/opencode-linux-x64-baseline-musl/-/opencode-linux-x64-baseline-musl-1.18.23.tgz", - "integrity": "sha512-c1DPxauhzAurlIBhJBr/rokDpc65l084T4qTl36gDDT9Xzc/Nk5Q5yMDaPm1DDI3WeHKDt11MDlxT5AjQW5gtw==", + "version": "1.18.25", + "resolved": "https://registry.npmjs.org/opencode-linux-x64-baseline-musl/-/opencode-linux-x64-baseline-musl-1.18.25.tgz", + "integrity": "sha512-E2JUeOOSXPbG1cNOzxnqjqkd0a3+oFmwkbJe6bZ308CFgLWBFfVh0fF42HTCEqfK+yYbidpEkQuEkUgxq/11IA==", "cpu": [ "x64" ], @@ -31595,9 +31595,9 @@ ] }, "node_modules/opencode-linux-x64-musl": { - "version": "1.18.23", - "resolved": "https://registry.npmjs.org/opencode-linux-x64-musl/-/opencode-linux-x64-musl-1.18.23.tgz", - "integrity": "sha512-t/5mlnTBZKdZpqKHwdwxlWqGakntauvMSmXtyJc17M7XJRmZaaGHtNSaSefbbYFIL4agoQCXTvIkhhyxOvr7zQ==", + "version": "1.18.25", + "resolved": "https://registry.npmjs.org/opencode-linux-x64-musl/-/opencode-linux-x64-musl-1.18.25.tgz", + "integrity": "sha512-W15qTNDz1fsTzs1SkE6bB/gpIDBF3rwDbewUKdbyXD3dVs6umyugOql1T4u9n/gqWa/Z/VDURbn39VsejeSdbQ==", "cpu": [ "x64" ], @@ -31611,9 +31611,9 @@ ] }, "node_modules/opencode-windows-arm64": { - "version": "1.18.23", - "resolved": "https://registry.npmjs.org/opencode-windows-arm64/-/opencode-windows-arm64-1.18.23.tgz", - "integrity": "sha512-QtJQcLU0yPz6on3jjks3f/EHgZuIDFw7FvAKu3wsHhL09NYDh7GczfRXDPRHa3NgqnU8dkB8p9mhqgaPRPogoQ==", + "version": "1.18.25", + "resolved": "https://registry.npmjs.org/opencode-windows-arm64/-/opencode-windows-arm64-1.18.25.tgz", + "integrity": "sha512-GFp74pProoPwqktHMf+9wQ8fza1RvFt0RG0iRtTQnJ4VWVY62qEeVuJkH6ki9QXS270HXyqtxvF8AuHQzuVZlA==", "cpu": [ "arm64" ], @@ -31624,9 +31624,9 @@ ] }, "node_modules/opencode-windows-x64": { - "version": "1.18.23", - "resolved": "https://registry.npmjs.org/opencode-windows-x64/-/opencode-windows-x64-1.18.23.tgz", - "integrity": "sha512-mMaIITuXzkNfjdcYL8uZaZuMDjulFyH/UCq9bxblam2mUZf9uWisoi5J6CXFsS/mkN7CZfTAt6PttSp4n3PH4g==", + "version": "1.18.25", + "resolved": "https://registry.npmjs.org/opencode-windows-x64/-/opencode-windows-x64-1.18.25.tgz", + "integrity": "sha512-xW5wtSxWYbI7DcmQWMlNWIiDBdMJON1vDiEmVWo88R9tT/PaahOhWKgp7FoWDqJKf89jS3ZIzkqnkU3F2dio7A==", "cpu": [ "x64" ], @@ -31637,9 +31637,9 @@ ] }, "node_modules/opencode-windows-x64-baseline": { - "version": "1.18.23", - "resolved": "https://registry.npmjs.org/opencode-windows-x64-baseline/-/opencode-windows-x64-baseline-1.18.23.tgz", - "integrity": "sha512-AqXsTKaPcDx3rrid5bLUwJbQ/3vr9rJ6fvOStIznTzwrbOgP8wy5G4jCoIzu6KB/WxGx/d1MrV4cGaJ73qnjBA==", + "version": "1.18.25", + "resolved": "https://registry.npmjs.org/opencode-windows-x64-baseline/-/opencode-windows-x64-baseline-1.18.25.tgz", + "integrity": "sha512-/28bGRQwT+2JdGbtGaNr95tstgysiULEXtcvgNg7yLDxitqmSVgd8V8XRGS0UWDdfiWWMND9A9T5EAsbF1/xDQ==", "cpu": [ "x64" ], diff --git a/package.json b/package.json index af8d67e748..924b7b6eeb 100644 --- a/package.json +++ b/package.json @@ -419,7 +419,7 @@ "lint-staged": "^17.4.1", "lockfile-lint": "^5.0.1", "node-loader": "^2.1.0", - "opencode-ai": "1.18.23", + "opencode-ai": "1.18.25", "playwright-ctrf-json-reporter": "^0.0.29", "prettier": "^3.9.6", "promptfoo": "^0.122.1", From fdee0ec2083c63fd9f4d187fdee8676ce89ed630 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:21:00 -0300 Subject: [PATCH 014/143] deps: bump the production group across 1 directory with 4 updates (#12399) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 7 bumps da raiz sobre o tip de release/v3.8.51: npm install reconciliou o lock sem diff (11/11 pacotes na versão alvo), typecheck:core e typecheck:noimplicit:core limpos, lint exit 0, e 149/149 testes nos 11 arquivos que exercitam zod diretamente. As falhas de CI do PR foram discriminadas como estado da base, não do bump. Obrigado, Dependabot. --- package-lock.json | 60 ++++++++++++++++++++++++++++++----------------- package.json | 8 +++---- 2 files changed, 43 insertions(+), 25 deletions(-) diff --git a/package-lock.json b/package-lock.json index 41a4c03dba..6f720491df 100644 --- a/package-lock.json +++ b/package-lock.json @@ -54,7 +54,7 @@ "lucide-react": "^1.33.0", "marked": "^18.0.11", "marked-terminal": "^7.3.0", - "material-symbols": "^0.46.0", + "material-symbols": "^0.47.0", "mermaid": "^11.17.2", "monaco-editor": "^0.56.0", "next": "16.3.3", @@ -62,7 +62,7 @@ "next-themes": "^0.4.6", "node-machine-id": "^1.1.12", "omniglyph": "^1.4.0", - "open": "^11.0.1", + "open": "^11.0.2", "ora": "^9.4.1", "parse5": "^8.0.1", "pino": "^10.3.1", @@ -85,7 +85,7 @@ "sql.js": "^1.14.2", "tailwind-merge": "^3.6.0", "tiktoken": "^1.0.22", - "tsx": "^4.23.12", + "tsx": "^4.23.13", "turndown": "7.2.4", "turndown-plugin-gfm": "1.0.2", "undici": "^8.10.0", @@ -94,7 +94,7 @@ "ws": "^8.21.3", "xxhash-wasm": "^1.1.0", "yazl": "^3.3.1", - "zod": "^4.4.3", + "zod": "^4.5.4", "zustand": "^5.0.15" }, "bin": { @@ -19427,7 +19427,9 @@ "version": "5.5.0", "resolved": "https://registry.npmjs.org/default-browser/-/default-browser-5.5.0.tgz", "integrity": "sha512-H9LMLr5zwIbSxrmvikGuI/5KGhZ8E2zH3stkMgM5LpOWDutGM2JZaj460Udnf1a+946zc7YBgrqEWwbk7zHvGw==", + "dev": true, "license": "MIT", + "optional": true, "dependencies": { "bundle-name": "^4.1.0", "default-browser-id": "^5.0.0" @@ -28350,9 +28352,9 @@ } }, "node_modules/material-symbols": { - "version": "0.46.0", - "resolved": "https://registry.npmjs.org/material-symbols/-/material-symbols-0.46.0.tgz", - "integrity": "sha512-YxmTXwOhLOI6EupAwFfxFERbaDe61dG/tveOSy2HecndGKqvJ74WqXrrXLNWpIGDkk6TDpieuQPDS+hA7+z3Ig==", + "version": "0.47.0", + "resolved": "https://registry.npmjs.org/material-symbols/-/material-symbols-0.47.0.tgz", + "integrity": "sha512-/Wt7QSv5Hih8EFj9ySHnnF+NAGajTMRLkfajj0j4MNCY3FB2QhGRg5xtnBs5bUE63mzVo7J5XbbIDr0HeITpHg==", "license": "Apache-2.0" }, "node_modules/math-intrinsics": { @@ -31399,16 +31401,16 @@ "optional": true }, "node_modules/open": { - "version": "11.0.1", - "resolved": "https://registry.npmjs.org/open/-/open-11.0.1.tgz", - "integrity": "sha512-NzwMUB6C1D0+Kd+9iMS/H4k+Ck3cTX6Ckyfr/gAGlmvSE1LUQZnEZvWBi4PYmMwH/S5SMeTXnE+9uAz8uF+pWw==", + "version": "11.0.2", + "resolved": "https://registry.npmjs.org/open/-/open-11.0.2.tgz", + "integrity": "sha512-RWqF+pBSkqecEvCKOn8QYhaNdRMJDZRIrlS/7rTDdLHaPcfXGCZ/h8zb413NfvdeAV0MR7T1yJcA34/q+CSm1Q==", "license": "MIT", "dependencies": { - "default-browser": "^5.4.0", + "default-browser": "^5.5.1", "define-lazy-prop": "^3.0.0", "is-in-ssh": "^1.0.0", "is-inside-container": "^1.0.0", - "powershell-utils": "^0.2.0", + "powershell-utils": "^0.2.1", "wsl-utils": "^1.0.0" }, "engines": { @@ -31418,6 +31420,22 @@ "url": "https://github.com/sponsors/sindresorhus" } }, + "node_modules/open/node_modules/default-browser": { + "version": "5.5.1", + "resolved": "https://registry.npmjs.org/default-browser/-/default-browser-5.5.1.tgz", + "integrity": "sha512-m1pAzaJgZ/gssEqlOhJkPJp8Xly7QyW6xcrkUa2KKcDeDSEMP7X8xipU3snUcfisTQx0w1AGae+9UtJSfVnXGw==", + "license": "MIT", + "dependencies": { + "bundle-name": "^4.1.0", + "default-browser-id": "^5.0.0" + }, + "engines": { + "node": ">=18" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, "node_modules/openai": { "version": "6.46.0", "resolved": "https://registry.npmjs.org/openai/-/openai-6.46.0.tgz", @@ -32980,9 +32998,9 @@ } }, "node_modules/powershell-utils": { - "version": "0.2.0", - "resolved": "https://registry.npmjs.org/powershell-utils/-/powershell-utils-0.2.0.tgz", - "integrity": "sha512-ZlsFlG7MtSFCoc5xreOvBAozCJ6Pf06opgJjh9ONEv418xpZSAzNjstD36C6+JwOnfSqOW/9uDkqKjezTdxZhw==", + "version": "0.2.1", + "resolved": "https://registry.npmjs.org/powershell-utils/-/powershell-utils-0.2.1.tgz", + "integrity": "sha512-C+y9x90UElAddDZmV4qOx9W53B61PO7cIqWz2dQsWlwswuq4mr8NEwytdGKboYbQlGZ3awrkTeNvcZiZNHnQ8A==", "license": "MIT", "engines": { "node": ">=20" @@ -37644,9 +37662,9 @@ "license": "0BSD" }, "node_modules/tsx": { - "version": "4.23.12", - "resolved": "https://registry.npmjs.org/tsx/-/tsx-4.23.12.tgz", - "integrity": "sha512-FDf4L4sYzKtzWYhU/Xm0AQFdTjdIxNo9ElTf2mxXM6k8YMHXzYUe4yODVaXP4V9uMFbVg8c0qyBccK2OOxb45Q==", + "version": "4.23.13", + "resolved": "https://registry.npmjs.org/tsx/-/tsx-4.23.13.tgz", + "integrity": "sha512-BL5MGkRln6aDYhb0xbQlEAGw743BaZYWdbWtdJOBriYJboKgUUYCadFp2/FpBBZquBC/ezNBn7wMMPx7FDZUDw==", "license": "MIT", "dependencies": { "esbuild": "~0.28.0" @@ -39917,9 +39935,9 @@ } }, "node_modules/zod": { - "version": "4.4.3", - "resolved": "https://registry.npmjs.org/zod/-/zod-4.4.3.tgz", - "integrity": "sha512-ytENFjIJFl2UwYglde2jchW2Hwm4GJFLDiSXWdTrJQBIN9Fcyp7n4DhxJEiWNAJMV1/BqWfW/kkg71UDcHJyTQ==", + "version": "4.5.4", + "resolved": "https://registry.npmjs.org/zod/-/zod-4.5.4.tgz", + "integrity": "sha512-sC95tT5iHHH9gtpj6A81kh+NEaRAUFN+qlUPDUbRfOMvNf5QCBqsb3WgvnpVtK5Y+4UfA6KqufotuTvMGiTlsA==", "license": "MIT", "funding": { "url": "https://github.com/sponsors/colinhacks" diff --git a/package.json b/package.json index 924b7b6eeb..b3b33ae2f3 100644 --- a/package.json +++ b/package.json @@ -319,7 +319,7 @@ "lucide-react": "^1.33.0", "marked": "^18.0.11", "marked-terminal": "^7.3.0", - "material-symbols": "^0.46.0", + "material-symbols": "^0.47.0", "mermaid": "^11.17.2", "monaco-editor": "^0.56.0", "next": "16.3.3", @@ -327,7 +327,7 @@ "next-themes": "^0.4.6", "node-machine-id": "^1.1.12", "omniglyph": "^1.4.0", - "open": "^11.0.1", + "open": "^11.0.2", "ora": "^9.4.1", "parse5": "^8.0.1", "pino": "^10.3.1", @@ -350,7 +350,7 @@ "sql.js": "^1.14.2", "tailwind-merge": "^3.6.0", "tiktoken": "^1.0.22", - "tsx": "^4.23.12", + "tsx": "^4.23.13", "turndown": "7.2.4", "turndown-plugin-gfm": "1.0.2", "undici": "^8.10.0", @@ -359,7 +359,7 @@ "ws": "^8.21.3", "xxhash-wasm": "^1.1.0", "yazl": "^3.3.1", - "zod": "^4.4.3", + "zod": "^4.5.4", "zustand": "^5.0.15" }, "optionalDependencies": { From 2a6d45abea0a70768402c896dfe69422e731144a Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:21:44 -0300 Subject: [PATCH 015/143] deps: bump electron from 43.4.1 to 44.0.0 in /electron (#12217) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bump MAJOR electron 43.4.1 → 44.0.0 (Chromium 152, Node 24.18.1, V8 15.2), aprovado pelo operador após análise dos breaking changes contra o código real: - **Remoção dos builds 32-bit (Windows ia32, Linux armv7l)** — sem impacto: `electron/package.json` só declara alvos x64 e arm64. - **libEGL/libGLESv2 deixam de ser distribuídos (ANGLE estático)** — sem impacto: nenhuma referência em `electron/`, nos scripts de build ou no afterPack. - **`clipboard` deixa de ser exposto ao renderer** — sem impacto: sem uso no projeto. - **`net.request` passa a rejeitar `Sec-Fetch-Dest` document/frame/iframe/fencedframe sem `Sec-Fetch-Mode: navigate`** — sem impacto: `net.request`/`net.fetch` não são usados. - **`openAsHidden` e `wasOpenedAsHidden` removidos de `app.set/getLoginItemSettings()`** — usados em `electron/main.js:1116` e `electron/lib/windowLifecycle.js:7`, mas sem regressão em plataforma suportada: o caminho vivo do autostart oculto é `args: ["--hidden"]` combinado com a checagem `argv.includes("--hidden")`, que `shouldStartHidden()` avalia primeiro; Linux nem chega nessa API (usa `enableLinuxDesktopAutostart`). `openAsHidden` só funcionava em macOS ≤ 12, que esta própria versão deixa de suportar. - **macOS 12 (Monterey) sai do suporte** — é o único efeito voltado ao usuário. Registrado no CHANGELOG em PR de acompanhamento. O smoke de empacotamento (`electron-package-smoke`) não roda em PR para branch de release — `ci.yml` dispara em `main` — então a validação de empacotamento acontece no merge → main, que é o modelo do repositório. Um PR de acompanhamento remove o código morto de `openAsHidden`/`wasOpenedAsHidden`. Obrigado, Dependabot. --- electron/package-lock.json | 8 ++++---- electron/package.json | 2 +- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/electron/package-lock.json b/electron/package-lock.json index 1aeabc7edc..7dbc37a13c 100644 --- a/electron/package-lock.json +++ b/electron/package-lock.json @@ -12,7 +12,7 @@ "electron-updater": "^6.8.9" }, "devDependencies": { - "electron": "^43.4.1", + "electron": "^44.0.0", "electron-builder": "^26.15.3" }, "engines": { @@ -1367,9 +1367,9 @@ } }, "node_modules/electron": { - "version": "43.4.1", - "resolved": "https://registry.npmjs.org/electron/-/electron-43.4.1.tgz", - "integrity": "sha512-5b+EuiwkgG5iRcsEL34rimgRpkYp15SsfZOa0pC5kXs0Tb82TH4n95rpQzTZa7yRCbA7tm0WoEbuBL6NaAhAcA==", + "version": "44.0.0", + "resolved": "https://registry.npmjs.org/electron/-/electron-44.0.0.tgz", + "integrity": "sha512-FkTqPrFPZYljdPI5b7KORGsJTd6FgUQDefl5MrU3Xz9R87pAj9JLreIjDqcRN8hJIkFHIou0o8kKzvcpT9qiRQ==", "dev": true, "license": "MIT", "dependencies": { diff --git a/electron/package.json b/electron/package.json index d4fd2bd1b2..1ed7410a4d 100644 --- a/electron/package.json +++ b/electron/package.json @@ -28,7 +28,7 @@ "electron-updater": "^6.8.9" }, "devDependencies": { - "electron": "^43.4.1", + "electron": "^44.0.0", "electron-builder": "^26.15.3" }, "overrides": { From 032adb0809246fcbc70376a4e2d3ad96c4ff21e7 Mon Sep 17 00:00:00 2001 From: backryun Date: Thu, 3 Sep 2026 19:49:42 +0900 Subject: [PATCH 016/143] feat(providers): refresh Z.ai Web models and browser transport (#12524) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com #12524, #12538, #12277 e #12367 sobre o tip de release/v3.8.51: os quatro boardaram sem conflito (áreas disjuntas — zai-web, nvidia, clova, cursor/devin/fable). typecheck:core limpo, check:provider-consistency OK (272 entradas REGISTRY, 355 providers canônicos), check:known-symbols OK, e 305/305 nos testes tocados pelos quatro PRs. Os IDs de modelo adicionados foram conferidos individualmente. Obrigado, @backryun. --- .../providers/registry/zai-web/index.ts | 35 ++-- open-sse/executors/zai-web.ts | 9 +- .../executors/zai-web/browserAutomation.ts | 11 +- open-sse/executors/zai-web/protocol.ts | 106 ++++++---- open-sse/services/browserBackedChat.ts | 18 ++ open-sse/services/browserBackedChat/types.ts | 4 + tests/unit/executor-zai-web.test.ts | 197 ++++++++++-------- tests/unit/model-test-runner.test.ts | 2 +- .../zai-web-chat-endpoint-8014-probe.test.ts | 5 +- .../zai-web-models-discovery-7678.test.ts | 36 ++-- 10 files changed, 237 insertions(+), 186 deletions(-) diff --git a/open-sse/config/providers/registry/zai-web/index.ts b/open-sse/config/providers/registry/zai-web/index.ts index 98901daab9..59b2b61c31 100644 --- a/open-sse/config/providers/registry/zai-web/index.ts +++ b/open-sse/config/providers/registry/zai-web/index.ts @@ -14,30 +14,27 @@ export const zai_webProvider: RegistryEntry = { // Z.ai's visible "Tools" switch enables its internal VLM/MCP tools. It does // not accept caller-supplied OpenAI `tools`, which remains disabled here. models: [ + { + id: "glm-5.3-flash", + name: "GLM-5.3-Flash", + toolCalling: false, + supportsReasoning: true, + supportedThinkingEfforts: ["low", "high", "max"], + supportsVision: true, + }, + { + id: "glm-5.3", + name: "GLM-5.3", + toolCalling: false, + supportsReasoning: true, + supportedThinkingEfforts: ["low", "high", "max"], + }, { id: "glm-5.2", name: "GLM-5.2", toolCalling: false, supportsReasoning: true, - }, - { - id: "GLM-5.1", - name: "GLM-5.1", - toolCalling: false, - supportsReasoning: true, - }, - { - id: "GLM-5-Turbo", - name: "GLM-5-Turbo", - toolCalling: false, - supportsReasoning: true, - }, - { - id: "GLM-5v-Turbo", - name: "GLM-5V-Turbo", - toolCalling: false, - supportsReasoning: true, - supportsVision: true, + supportedThinkingEfforts: ["high", "max"], }, ], }; diff --git a/open-sse/executors/zai-web.ts b/open-sse/executors/zai-web.ts index 6e01aee98d..fb09e4ea82 100644 --- a/open-sse/executors/zai-web.ts +++ b/open-sse/executors/zai-web.ts @@ -5,8 +5,9 @@ * browser-issued CAPTCHA proof for chat completions. The browser transport is * the default; callers with a short-lived proof can use the direct HTTP path. * - * Completions go to /api/v2/chat/completions; the older unversioned - * /api/chat/completions path is stale and 404s model-independently (#8014). + * Completions go to /api/v2/chat/completions. Z.ai's CAPTCHA rejects true + * headless Chromium with F001, so the browser transport uses off-screen headed + * Chromium while retaining the shared browser pool. */ import { createHash, randomUUID } from "node:crypto"; import { BaseExecutor, type ExecuteInput } from "./base.ts"; @@ -67,6 +68,7 @@ export { parseZaiFrontendVersion, resolveZaiThinkingConfig, resolveZaiVlmConfig, + zaiUpstreamModelId, } from "./zai-web/protocol.ts"; export type { ZaiModelCapabilities, @@ -174,6 +176,7 @@ function buildZaiBrowserChatOptions(input: { userAgent: ZAI_USER_AGENT, locale: "en-US", timezone: "Asia/Seoul", + headless: false, inputSelector: "#chat-input", submitButtonSelector: '[aria-label="Send Message"] button:not([disabled])', submitButtonMode: "dom", @@ -243,7 +246,7 @@ function resolveZaiRequest( const modelId = (bodyObj.model as string) || model || ZAI_DEFAULT_MODEL; if (imageUrls.length > 0 && !getZaiModelCapabilities(modelId).vision) { return fail( - `Z.ai model ${unprefixedModelId(modelId)} does not accept image input; use GLM-5V-Turbo.` + `Z.ai model ${unprefixedModelId(modelId)} does not accept image input; use GLM-5.3-Flash.` ); } diff --git a/open-sse/executors/zai-web/browserAutomation.ts b/open-sse/executors/zai-web/browserAutomation.ts index 38924ada4d..a8aabccc33 100644 --- a/open-sse/executors/zai-web/browserAutomation.ts +++ b/open-sse/executors/zai-web/browserAutomation.ts @@ -23,14 +23,17 @@ async function runStage(name: string, action: () => Promise): Promise { const selector = page.locator('[aria-label="Select a model"]').first(); await selector.waitFor({ state: "visible", timeout: 10_000 }); - if ((await selector.innerText()).includes(modelName)) return; + if ((await selector.getByText(modelName, { exact: true }).count()) > 0) return; // The landing-page hero animation can remain above the already-visible // selector and make coordinate-based clicks time out. await selector.evaluate((element) => (element as HTMLElement).click()); const menu = page.locator('[role="menu"]').filter({ hasText: modelName }).first(); await menu.waitFor({ state: "visible", timeout: 5_000 }); - const modelButton = menu.locator("button").filter({ hasText: modelName }).first(); + const modelButton = menu + .getByText(modelName, { exact: true }) + .first() + .locator("xpath=ancestor::button[1]"); await modelButton.evaluate((element) => (element as HTMLElement).click()); await page .locator('[aria-label="Select a model"]') @@ -73,13 +76,13 @@ async function setZaiBrowserWebSearch(page: Page, enabled: boolean): Promise, effort: ZaiThinkingConfig["effort"] ): Promise { const effortButton = menu.locator("button").filter({ - hasText: effort === "high" ? "High" : "Max", + hasText: effort === "low" ? "Low" : effort === "high" ? "High" : "Max", }); if ((await effortButton.getAttribute("data-selected")) === "true") return; await runStage(`select ${effort}`, () => diff --git a/open-sse/executors/zai-web/protocol.ts b/open-sse/executors/zai-web/protocol.ts index 562ca85a7f..511e5d3aa9 100644 --- a/open-sse/executors/zai-web/protocol.ts +++ b/open-sse/executors/zai-web/protocol.ts @@ -7,8 +7,8 @@ import { normalizeCookie, sanitizeErrorMessage } from "../../utils/error.ts"; export const ZAI_BASE_URL = "https://chat.z.ai"; export const ZAI_NEW_CHAT_URL = `${ZAI_BASE_URL}/api/v1/chats/new`; export const ZAI_CHAT_URL = `${ZAI_BASE_URL}/api/v2/chat/completions`; -export const ZAI_DEFAULT_MODEL = "GLM-5.1"; -export const ZAI_DEFAULT_FE_VERSION = "prod-fe-1.1.79"; +export const ZAI_DEFAULT_MODEL = "glm-5.3"; +export const ZAI_DEFAULT_FE_VERSION = "prod-fe-1.1.92"; export const ZAI_USER_AGENT = "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/150.0.0.0 Safari/537.36"; export const ZAI_FE_VERSION_CACHE_TTL_MS = 15 * 60 * 1000; @@ -21,7 +21,7 @@ export interface NewChatRequest { userMessageId: string; } -export type ZaiReasoningEffort = "high" | "max"; +export type ZaiReasoningEffort = "low" | "high" | "max"; export interface ZaiThinkingConfig { enabled: boolean; @@ -61,12 +61,23 @@ const NO_ZAI_MODEL_CAPABILITIES: ZaiModelCapabilities = Object.freeze({ }); /** - * Verified against chat.z.ai/api/models (prod-fe-1.1.79). + * Verified against chat.z.ai/api/models (prod-fe-1.1.92). * `returnFc` is the site's internal function-call result capability; it is * distinct from accepting caller-supplied OpenAI `tools`. */ const ZAI_MODEL_CAPABILITIES: Record = { - "glm-5.2": { + "glm-5.3-flash": { + mcp: false, + reasoningEffort: true, + returnFc: true, + thinking: true, + vision: true, + vlmTools: false, + vlmWebSearch: false, + vlmWebsiteMode: false, + webSearch: true, + }, + "glm-5.3": { mcp: true, reasoningEffort: true, returnFc: true, @@ -77,9 +88,9 @@ const ZAI_MODEL_CAPABILITIES: Record = { vlmWebsiteMode: false, webSearch: true, }, - "glm-5.1": { + "glm-5.2": { mcp: true, - reasoningEffort: false, + reasoningEffort: true, returnFc: true, thinking: true, vision: false, @@ -88,28 +99,6 @@ const ZAI_MODEL_CAPABILITIES: Record = { vlmWebsiteMode: false, webSearch: true, }, - "glm-5-turbo": { - mcp: true, - reasoningEffort: false, - returnFc: true, - thinking: true, - vision: false, - vlmTools: false, - vlmWebSearch: false, - vlmWebsiteMode: false, - webSearch: true, - }, - "glm-5v-turbo": { - mcp: false, - reasoningEffort: false, - returnFc: true, - thinking: true, - vision: true, - vlmTools: true, - vlmWebSearch: true, - vlmWebsiteMode: true, - webSearch: true, - }, }; export function asRecord(value: unknown): Record | null { @@ -136,6 +125,7 @@ export function describeZaiBrowserFailure(result: { status: number; body: Buffer; observedPostUrls?: string[]; + observedPostResponses?: Array<{ url: string; status: number }>; timing: { captureResponseMs: number; totalMs: number }; }): string { const status = result.status > 0 ? String(result.status) : "no matching response"; @@ -144,10 +134,16 @@ export function describeZaiBrowserFailure(result: { result.observedPostUrls && result.observedPostUrls.length > 0 ? ` Observed POST targets: ${result.observedPostUrls.join(", ")}.` : ""; + const observedResponses = + result.observedPostResponses && result.observedPostResponses.length > 0 + ? ` Observed POST responses: ${result.observedPostResponses + .map(({ url, status }) => `${url} [${status}]`) + .join(", ")}.` + : ""; const detail = browserFailureDetail(result.body) || (result.status === 0 - ? `The page did not issue the expected authenticated chat completion request.${observed}` + ? `The page did not issue the expected authenticated chat completion request.${observed}${observedResponses}` : "The browser response body was empty."); return `Z.ai browser transport failed (${status}; ${timing}): ${detail}`; } @@ -307,17 +303,27 @@ export function unprefixedModelId(modelId: string): string { return modelId.trim().split("/").at(-1) || modelId.trim(); } -export function browserModelName(modelId: string): string { +/** Map OmniRoute's public Flash id to the opaque id used by chat.z.ai's wire API. */ +export function zaiUpstreamModelId(modelId: string): string { const unprefixed = unprefixedModelId(modelId); - if (unprefixed.toLowerCase() === "glm-5.2") return "GLM-5.2"; - if (unprefixed.toLowerCase() === "glm-5v-turbo") return "GLM-5V-Turbo"; - return unprefixed; + return unprefixed.toLowerCase() === "glm-5.3-flash" ? "x-preview-l" : unprefixed; +} + +function zaiCapabilityModelId(modelId: string): string { + const unprefixed = unprefixedModelId(modelId).toLowerCase(); + return unprefixed === "x-preview-l" ? "glm-5.3-flash" : unprefixed; +} + +export function browserModelName(modelId: string): string { + const normalized = zaiCapabilityModelId(modelId); + if (normalized === "glm-5.3-flash") return "GLM-5.3-Flash"; + if (normalized === "glm-5.3") return "GLM-5.3"; + if (normalized === "glm-5.2") return "GLM-5.2"; + return unprefixedModelId(modelId); } export function getZaiModelCapabilities(modelId: string): ZaiModelCapabilities { - return ( - ZAI_MODEL_CAPABILITIES[unprefixedModelId(modelId).toLowerCase()] ?? NO_ZAI_MODEL_CAPABILITIES - ); + return ZAI_MODEL_CAPABILITIES[zaiCapabilityModelId(modelId)] ?? NO_ZAI_MODEL_CAPABILITIES; } function getFeatureOption(body: Record, key: string): unknown { @@ -325,7 +331,7 @@ function getFeatureOption(body: Record, key: string): unknown { return asRecord(body.features)?.[key]; } -/** Resolve each model's Deep Think control; only GLM-5.2 accepts High/Max effort. */ +/** Resolve each model's Deep Think control using its currently exposed effort vocabulary. */ export function resolveZaiThinkingConfig( modelId: string, body: Record @@ -339,19 +345,26 @@ export function resolveZaiThinkingConfig( : typeof reasoning?.effort === "string" ? reasoning.effort.trim().toLowerCase() : ""; - const disabled = body.enable_thinking === false || rawEffort === "none" || rawEffort === "off"; + const supportsLowEffort = zaiCapabilityModelId(modelId) !== "glm-5.2"; const effort: ZaiReasoningEffort = - rawEffort === "low" || rawEffort === "medium" || rawEffort === "high" ? "high" : "max"; + rawEffort === "low" && supportsLowEffort + ? "low" + : rawEffort === "low" || rawEffort === "medium" || rawEffort === "high" + ? "high" + : "max"; return { supported, - enabled: supported && !disabled, + // The current GLM-5.3/5.2 consumer models expose effort selection but no + // non-thinking mode. Keep Deep Think enabled even when a generic client + // sends an off/none compatibility value. + enabled: supported, effort, effortSupported: capabilities.reasoningEffort, }; } -/** Resolve GLM-5V-Turbo's visible Web Search and Tools controls. */ +/** Resolve the selected model's visible Web Search and Tools controls. */ export function resolveZaiVlmConfig(modelId: string, body: Record): ZaiVlmConfig { const capabilities = getZaiModelCapabilities(modelId); const toolsOption = getFeatureOption(body, "vlm_tools_enable"); @@ -425,7 +438,7 @@ export function buildZaiCompletionUrl(input: { hostname: "chat.z.ai", protocol: "https:", referrer: "", - title: "Z.ai - Advanced AI Chatbot & Agent powered by GLM-5.2", + title: "Z.ai - Advanced AI Chatbot & Agent powered by GLM-5.3", timezone_offset: "0", local_time: now.toISOString(), utc_time: now.toUTCString(), @@ -448,13 +461,14 @@ export function buildZaiNewChatBody( ): NewChatRequest { const prompt = latestUserPrompt(messages); const userMessageId = randomUUID(); + const upstreamModelId = zaiUpstreamModelId(modelId); return { userMessageId, payload: { chat: { id: "", title: "New Chat", - models: [modelId], + models: [upstreamModelId], params: {}, history: { messages: { @@ -465,7 +479,7 @@ export function buildZaiNewChatBody( role: "user", content: prompt, timestamp: Math.floor(Date.now() / 1000), - models: [modelId], + models: [upstreamModelId], }, }, currentId: userMessageId, @@ -530,7 +544,7 @@ export function buildZaiRequestBody(input: { } return { stream: true, - model: input.modelId, + model: zaiUpstreamModelId(input.modelId), messages: foldMessages(input.messages), signature_prompt: input.prompt, params, diff --git a/open-sse/services/browserBackedChat.ts b/open-sse/services/browserBackedChat.ts index 4b3c7078e7..fa51a9c266 100644 --- a/open-sse/services/browserBackedChat.ts +++ b/open-sse/services/browserBackedChat.ts @@ -236,6 +236,7 @@ export async function browserBackedChat( userAgent, locale, timezone, + headless, inputSelector, submitButtonSelector, submitButtonMode = "playwright", @@ -257,11 +258,13 @@ export async function browserBackedChat( userAgent, locale, timezone, + headless, }); const acquireContextMs = Date.now() - tAcquireStart; const page = await openPage(pooled); const observedPostUrls: string[] = []; + const observedPostResponses: Array<{ url: string; status: number }> = []; page.on("request", (request) => { if (request.method() !== "POST") return; try { @@ -273,6 +276,19 @@ export async function browserBackedChat( // Ignore malformed/non-HTTP request URLs. } }); + page.on("response", (response) => { + if (response.request().method() !== "POST") return; + try { + const url = new URL(response.url()); + if (!url.hostname.endsWith(chatUrlMatchDomain)) return; + observedPostResponses.push({ + url: `${url.origin}${url.pathname}`, + status: response.status(), + }); + } catch { + // Ignore malformed/non-HTTP response URLs. + } + }); try { const tNavStart = Date.now(); await withAbort( @@ -379,6 +395,7 @@ export async function browserBackedChat( body, isStealth: pooled.isStealth, observedPostUrls, + observedPostResponses, timing: { acquireContextMs, navigateMs, @@ -404,6 +421,7 @@ export async function browserBackedChat( body, isStealth: pooled.isStealth, observedPostUrls, + observedPostResponses, timing: { acquireContextMs, navigateMs: 0, diff --git a/open-sse/services/browserBackedChat/types.ts b/open-sse/services/browserBackedChat/types.ts index c90fb3b80e..3a197be16c 100644 --- a/open-sse/services/browserBackedChat/types.ts +++ b/open-sse/services/browserBackedChat/types.ts @@ -29,6 +29,8 @@ export interface BrowserBackedChatRequest { locale?: string; /** Browser IANA timezone. Defaults to America/New_York. */ timezone?: string; + /** Launch a headed browser when the provider rejects true headless mode. */ + headless?: boolean; /** Selector for the provider chat input. */ inputSelector: string; /** Optional selector for the provider submit button. */ @@ -64,6 +66,8 @@ export interface BrowserBackedChatResult { isStealth: boolean; /** Sanitized POST targets observed while submitting. */ observedPostUrls?: string[]; + /** Sanitized POST response targets and statuses observed while submitting. */ + observedPostResponses?: Array<{ url: string; status: number }>; timing: { acquireContextMs: number; navigateMs: number; diff --git a/tests/unit/executor-zai-web.test.ts b/tests/unit/executor-zai-web.test.ts index aa26418f9c..ec77c94f5e 100644 --- a/tests/unit/executor-zai-web.test.ts +++ b/tests/unit/executor-zai-web.test.ts @@ -30,7 +30,7 @@ function installZaiFetch( const value = String(url); if (value === ZAI_HOME_URL) { return new Response( - '' + '' ); } if (value === ZAI_NEW_CHAT_URL) { @@ -90,13 +90,18 @@ describe("ZaiWebExecutor", () => { "Z.ai browser transport failed (502; capture 30001ms, total 33412ms): " + "browserBackedChat failed: response.body unavailable" ); - assert.match( + assert.equal( mod.describeZaiBrowserFailure({ status: 0, body: Buffer.alloc(0), + observedPostUrls: ["https://chat.z.ai/api/v1/chats/new"], + observedPostResponses: [{ url: "https://chat.z.ai/api/v1/chats/new", status: 200 }], timing: { captureResponseMs: 30_000, totalMs: 33_000 }, }), - /no matching response.*did not issue the expected authenticated chat completion request/ + "Z.ai browser transport failed (no matching response; capture 30000ms, total 33000ms): " + + "The page did not issue the expected authenticated chat completion request. " + + "Observed POST targets: https://chat.z.ai/api/v1/chats/new. " + + "Observed POST responses: https://chat.z.ai/api/v1/chats/new [200]." ); }); @@ -128,9 +133,9 @@ describe("ZaiWebExecutor", () => { it("parses the deployed frontend version from the homepage asset path", () => { assert.equal( mod.parseZaiFrontendVersion( - "https://z-cdn.chatglm.cn/z-ai/frontend/prod-fe-1.1.79/assets/index.js" + "https://z-cdn.chatglm.cn/z-ai/frontend/prod-fe-1.1.92/assets/index.js" ), - "prod-fe-1.1.79" + "prod-fe-1.1.92" ); assert.equal(mod.parseZaiFrontendVersion(""), null); }); @@ -211,86 +216,87 @@ describe("ZaiWebExecutor", () => { ]); }); - it("enables Deep Think for every public model and limits effort to GLM-5.2", () => { - assert.deepEqual(mod.resolveZaiThinkingConfig("glm-5.2", {}), { + it("maps the three public models to their current Deep Think effort vocabularies", () => { + assert.deepEqual(mod.resolveZaiThinkingConfig("glm-5.3-flash", { reasoning_effort: "low" }), { supported: true, enabled: true, - effort: "max", + effort: "low", effortSupported: true, }); - assert.deepEqual(mod.resolveZaiThinkingConfig("zw/glm-5.2", { reasoning_effort: "medium" }), { + assert.deepEqual(mod.resolveZaiThinkingConfig("zw/glm-5.3", { reasoning_effort: "medium" }), { supported: true, enabled: true, effort: "high", effortSupported: true, }); - assert.deepEqual(mod.resolveZaiThinkingConfig("glm-5.2", { reasoning: { effort: "high" } }), { + assert.deepEqual(mod.resolveZaiThinkingConfig("glm-5.3", { reasoning: { effort: "high" } }), { supported: true, enabled: true, effort: "high", effortSupported: true, }); - assert.deepEqual(mod.resolveZaiThinkingConfig("glm-5.2", { reasoning_effort: "off" }), { - supported: true, - enabled: false, - effort: "max", - effortSupported: true, - }); - assert.deepEqual(mod.resolveZaiThinkingConfig("GLM-5.1", { reasoning_effort: "max" }), { + assert.deepEqual(mod.resolveZaiThinkingConfig("glm-5.3", { reasoning_effort: "off" }), { supported: true, enabled: true, effort: "max", - effortSupported: false, + effortSupported: true, + }); + assert.deepEqual(mod.resolveZaiThinkingConfig("glm-5.3", { enable_thinking: false }), { + supported: true, + enabled: true, + effort: "max", + effortSupported: true, + }); + assert.deepEqual(mod.resolveZaiThinkingConfig("glm-5.2", { reasoning_effort: "low" }), { + supported: true, + enabled: true, + effort: "high", + effortSupported: true, }); }); - it("maps GLM-5V-Turbo vision and internal VLM controls from live capabilities", () => { - assert.deepEqual(mod.getZaiModelCapabilities("zw/GLM-5v-Turbo"), { + it("maps GLM-5.3-Flash vision and web controls from live capabilities", () => { + assert.deepEqual(mod.getZaiModelCapabilities("zw/glm-5.3-flash"), { mcp: false, - reasoningEffort: false, + reasoningEffort: true, returnFc: true, thinking: true, vision: true, - vlmTools: true, - vlmWebSearch: true, - vlmWebsiteMode: true, + vlmTools: false, + vlmWebSearch: false, + vlmWebsiteMode: false, webSearch: true, }); - assert.deepEqual(mod.resolveZaiVlmConfig("GLM-5v-Turbo", {}), { - toolsEnabled: true, - webSearchEnabled: true, - websiteModeEnabled: true, + assert.deepEqual(mod.getZaiModelCapabilities("x-preview-l"), { + mcp: false, + reasoningEffort: true, + returnFc: true, + thinking: true, + vision: true, + vlmTools: false, + vlmWebSearch: false, + vlmWebsiteMode: false, + webSearch: true, }); - assert.deepEqual( - mod.resolveZaiVlmConfig("GLM-5v-Turbo", { - features: { - vlm_tools_enable: false, - vlm_web_search_enable: false, - vlm_website_mode: false, - }, - }), - { - toolsEnabled: false, - webSearchEnabled: false, - websiteModeEnabled: true, - } - ); - assert.deepEqual(mod.resolveZaiVlmConfig("GLM-5.1", {}), { + assert.deepEqual(mod.resolveZaiVlmConfig("glm-5.3-flash", { web_search: true }), { + toolsEnabled: false, + webSearchEnabled: true, + websiteModeEnabled: false, + }); + assert.deepEqual(mod.resolveZaiVlmConfig("glm-5.3", {}), { toolsEnabled: false, webSearchEnabled: false, websiteModeEnabled: false, }); - assert.deepEqual(mod.resolveZaiVlmConfig("GLM-5.1", { web_search: true }), { - toolsEnabled: false, - webSearchEnabled: true, - websiteModeEnabled: false, - }); + assert.equal(mod.zaiUpstreamModelId("zai-web/glm-5.3-flash"), "x-preview-l"); + assert.equal(mod.zaiUpstreamModelId("zai-web/glm-5.3"), "glm-5.3"); + assert.equal(mod.getZaiModelCapabilities("GLM-5.1").thinking, false); }); it("returns a credential error when no session credential is provided", async () => { const executor = new mod.ZaiWebExecutor(); const result = await executor.execute({ - model: "GLM-5.1", + model: "glm-5.3", body: { messages: [{ role: "user", content: "hi" }] }, stream: false, credentials: { apiKey: "" }, @@ -313,7 +319,6 @@ describe("ZaiWebExecutor", () => { try { const executor = new mod.ZaiWebExecutor(); const result = await executor.execute({ - model: "glm-5.2", body: { messages: [{ role: "user", content: "hi" }] }, stream: false, credentials: { apiKey: TEST_TOKEN }, @@ -322,8 +327,10 @@ describe("ZaiWebExecutor", () => { const completion = await result.response.json(); assert.equal(completion.choices[0].message.content, "Browser"); + assert.equal(completion.model, "glm-5.3"); assert.equal(capturedRequest?.localStorage?.token, TEST_TOKEN); assert.equal(capturedRequest?.localStorageOrigin, "https://chat.z.ai"); + assert.equal(capturedRequest?.headless, false); assert.equal(capturedRequest?.inputSelector, "#chat-input"); assert.equal( capturedRequest?.submitButtonSelector, @@ -331,7 +338,7 @@ describe("ZaiWebExecutor", () => { ); assert.equal(capturedRequest?.submitButtonMode, "dom"); assert.equal(capturedRequest?.userMessage, "hi"); - assert.match(capturedRequest?.chatPageUrl ?? "", /model=GLM-5\.2/); + assert.match(capturedRequest?.chatPageUrl ?? "", /model=GLM-5\.3/); assert.equal(typeof capturedRequest?.beforeSubmit, "function"); assert.equal(result.headers["X-OmniRoute-Transport"], "browser"); assert.equal(result.transformedBody.browser_backed, true); @@ -342,7 +349,7 @@ describe("ZaiWebExecutor", () => { } }); - it("configures GLM-5V-Turbo controls on the browser transport", async () => { + it("configures GLM-5.3-Flash on the browser transport", async () => { let capturedRequest: BrowserBackedChatRequest | null = null; browserChat.__setBrowserBackedChatOverrideForTesting(async (request) => { capturedRequest = request; @@ -352,8 +359,8 @@ describe("ZaiWebExecutor", () => { try { const executor = new mod.ZaiWebExecutor(); const result = await executor.execute({ - model: "GLM-5v-Turbo", - body: { messages: [{ role: "user", content: "use the model tools" }] }, + model: "glm-5.3-flash", + body: { messages: [{ role: "user", content: "use flash" }] }, stream: false, credentials: { apiKey: TEST_TOKEN }, signal: null, @@ -361,19 +368,19 @@ describe("ZaiWebExecutor", () => { const completion = await result.response.json(); assert.equal(completion.choices[0].message.content, "VLM"); - assert.match(capturedRequest?.chatPageUrl ?? "", /model=GLM-5V-Turbo/); + assert.match(capturedRequest?.chatPageUrl ?? "", /model=GLM-5\.3-Flash/); assert.equal(typeof capturedRequest?.beforeSubmit, "function"); assert.equal(result.transformedBody.enable_thinking, true); - assert.equal(result.transformedBody.vlm_tools_enable, true); - assert.equal(result.transformedBody.vlm_web_search_enable, true); - assert.equal(result.transformedBody.vlm_website_mode, true); - assert.equal("reasoning_effort" in result.transformedBody, false); + assert.equal(result.transformedBody.reasoning_effort, "max"); + assert.equal(result.transformedBody.vlm_tools_enable, false); + assert.equal(result.transformedBody.vlm_web_search_enable, false); + assert.equal(result.transformedBody.vlm_website_mode, false); } finally { browserChat.__resetBrowserBackedChatOverrideForTesting(); } }); - it("uploads GLM-5V-Turbo image input through the authenticated browser page", async () => { + it("uploads GLM-5.3-Flash image input through the authenticated browser page", async () => { let capturedRequest: BrowserBackedChatRequest | null = null; browserChat.__setBrowserBackedChatOverrideForTesting(async (request) => { capturedRequest = request; @@ -383,7 +390,7 @@ describe("ZaiWebExecutor", () => { try { const executor = new mod.ZaiWebExecutor(); const result = await executor.execute({ - model: "GLM-5v-Turbo", + model: "glm-5.3-flash", body: { messages: [ { @@ -445,7 +452,7 @@ describe("ZaiWebExecutor", () => { assert.equal(result.response.status, 400); const parsed = await result.response.json(); - assert.match(parsed.error.message, /use GLM-5V-Turbo/); + assert.match(parsed.error.message, /use GLM-5\.3-Flash/); }); it("creates a chat, signs the v2 request, and forwards the CAPTCHA proof", async () => { @@ -461,9 +468,9 @@ describe("ZaiWebExecutor", () => { try { const executor = new mod.ZaiWebExecutor(); const result = await executor.execute({ - model: "GLM-5.1", + model: "glm-5.3", body: { - model: "GLM-5.1", + model: "glm-5.3", messages: [{ role: "user", content: "hello" }], temperature: 0.4, web_search: true, @@ -477,7 +484,7 @@ describe("ZaiWebExecutor", () => { const newChatHeaders = capture.newChatInit?.headers as Record; assert.equal(newChatHeaders.Authorization, `Bearer ${TEST_TOKEN}`); const newChatBody = JSON.parse(String(capture.newChatInit?.body)); - assert.deepEqual(newChatBody.chat.models, ["GLM-5.1"]); + assert.deepEqual(newChatBody.chat.models, ["glm-5.3"]); assert.equal(newChatBody.chat.history.currentId.length, 36); assert.equal(newChatBody.chat.enable_thinking, true); assert.equal(newChatBody.chat.auto_web_search, true); @@ -494,11 +501,11 @@ describe("ZaiWebExecutor", () => { const headers = capture.completionInit?.headers as Record; assert.equal(headers.Authorization, `Bearer ${TEST_TOKEN}`); - assert.equal(headers["X-FE-Version"], "prod-fe-1.1.79"); + assert.equal(headers["X-FE-Version"], "prod-fe-1.1.92"); assert.match(headers["X-Signature"], /^[a-f0-9]{64}$/); const parsedBody = JSON.parse(String(capture.completionInit?.body)); - assert.equal(parsedBody.model, "GLM-5.1"); + assert.equal(parsedBody.model, "glm-5.3"); assert.equal(parsedBody.stream, true); assert.deepEqual(parsedBody.messages, [{ role: "user", content: "hello" }]); assert.equal(parsedBody.signature_prompt, "hello"); @@ -508,7 +515,7 @@ describe("ZaiWebExecutor", () => { assert.equal(parsedBody.features.web_search, false); assert.equal(parsedBody.features.auto_web_search, true); assert.equal(parsedBody.features.enable_thinking, true); - assert.equal("reasoning_effort" in parsedBody.features, false); + assert.equal(parsedBody.features.reasoning_effort, "max"); assert.equal(result.headers.Authorization, "Bearer [REDACTED]"); assert.equal(result.transformedBody.captcha_verify_param, "[REDACTED]"); } finally { @@ -516,7 +523,7 @@ describe("ZaiWebExecutor", () => { } }); - it("sends GLM-5.2 Deep Think High through the direct request path", async () => { + it("sends GLM-5.3 Deep Think Low through the direct request path", async () => { const capture: ZaiFetchCapture = {}; const originalFetch = installZaiFetch( () => @@ -529,36 +536,36 @@ describe("ZaiWebExecutor", () => { try { const executor = new mod.ZaiWebExecutor(); await executor.execute({ - model: "glm-5.2", + model: "glm-5.3", body: { - model: "glm-5.2", + model: "glm-5.3", messages: [{ role: "user", content: "think carefully" }], - reasoning_effort: "high", + reasoning_effort: "low", }, stream: false, credentials: { apiKey: TEST_CREDENTIAL }, signal: null, }); - // #8014: completions must target the versioned v2 path. The query string + // The query string // carries the per-request signature payload, so match the endpoint prefix. assert.ok( String(capture.completionUrl).startsWith("https://chat.z.ai/api/v2/chat/completions?"), - `expected the v2 completions endpoint, got ${capture.completionUrl}` + `expected the current completions endpoint, got ${capture.completionUrl}` ); const newChatBody = JSON.parse(String(capture.newChatInit?.body)); assert.equal(newChatBody.chat.enable_thinking, true); - assert.equal(newChatBody.chat.reasoning_effort, "high"); + assert.equal(newChatBody.chat.reasoning_effort, "low"); const completionBody = JSON.parse(String(capture.completionInit?.body)); assert.equal(completionBody.features.enable_thinking, true); - assert.equal(completionBody.features.reasoning_effort, "high"); + assert.equal(completionBody.features.reasoning_effort, "low"); } finally { globalThis.fetch = originalFetch; } }); - it("sends GLM-5V-Turbo VLM tools and web-search flags through the direct path", async () => { + it("maps GLM-5.3-Flash to its opaque wire id on the direct path", async () => { const capture: ZaiFetchCapture = {}; const originalFetch = installZaiFetch( () => @@ -571,10 +578,11 @@ describe("ZaiWebExecutor", () => { try { const executor = new mod.ZaiWebExecutor(); await executor.execute({ - model: "GLM-5v-Turbo", + model: "glm-5.3-flash", body: { - model: "GLM-5v-Turbo", - messages: [{ role: "user", content: "inspect this image" }], + model: "glm-5.3-flash", + messages: [{ role: "user", content: "answer quickly" }], + web_search: true, }, stream: false, credentials: { apiKey: TEST_CREDENTIAL }, @@ -582,19 +590,26 @@ describe("ZaiWebExecutor", () => { }); const newChatBody = JSON.parse(String(capture.newChatInit?.body)); + assert.deepEqual(newChatBody.chat.models, ["x-preview-l"]); + assert.deepEqual( + newChatBody.chat.history.messages[newChatBody.chat.history.currentId].models, + ["x-preview-l"] + ); assert.equal(newChatBody.chat.enable_thinking, true); assert.equal(newChatBody.chat.auto_web_search, true); - assert.equal(newChatBody.chat.extra.vlm_tools_enable, true); - assert.equal(newChatBody.chat.extra.vlm_web_search_enable, true); - assert.equal(newChatBody.chat.extra.vlm_website_mode, true); + assert.equal(newChatBody.chat.reasoning_effort, "max"); + assert.equal(newChatBody.chat.extra.vlm_tools_enable, false); + assert.equal(newChatBody.chat.extra.vlm_web_search_enable, false); + assert.equal(newChatBody.chat.extra.vlm_website_mode, false); const completionBody = JSON.parse(String(capture.completionInit?.body)); + assert.equal(completionBody.model, "x-preview-l"); assert.equal(completionBody.features.enable_thinking, true); - assert.equal(completionBody.features.auto_web_search, false); - assert.equal(completionBody.features.vlm_tools_enable, true); - assert.equal(completionBody.features.vlm_web_search_enable, true); - assert.equal(completionBody.features.vlm_website_mode, true); - assert.equal("reasoning_effort" in completionBody.features, false); + assert.equal(completionBody.features.reasoning_effort, "max"); + assert.equal(completionBody.features.auto_web_search, true); + assert.equal(completionBody.features.vlm_tools_enable, false); + assert.equal(completionBody.features.vlm_web_search_enable, false); + assert.equal(completionBody.features.vlm_website_mode, false); } finally { globalThis.fetch = originalFetch; } @@ -619,7 +634,7 @@ describe("ZaiWebExecutor", () => { try { const executor = new mod.ZaiWebExecutor(); const result = await executor.execute({ - model: "GLM-5.1", + model: "glm-5.3", body: { messages: [{ role: "user", content: "hi" }] }, stream: false, credentials: { apiKey: TEST_CREDENTIAL }, @@ -651,7 +666,7 @@ describe("ZaiWebExecutor", () => { try { const executor = new mod.ZaiWebExecutor(); const result = await executor.execute({ - model: "GLM-5.1", + model: "glm-5.3", body: { messages: [{ role: "user", content: "hi" }] }, stream: true, credentials: { apiKey: TEST_CREDENTIAL }, @@ -673,7 +688,7 @@ describe("ZaiWebExecutor", () => { try { const executor = new mod.ZaiWebExecutor(); const result = await executor.execute({ - model: "GLM-5.1", + model: "glm-5.3", body: { messages: [{ role: "user", content: "hi" }] }, stream: false, credentials: { apiKey: TEST_CREDENTIAL }, diff --git a/tests/unit/model-test-runner.test.ts b/tests/unit/model-test-runner.test.ts index 7a19e67482..c717ea0bb0 100644 --- a/tests/unit/model-test-runner.test.ts +++ b/tests/unit/model-test-runner.test.ts @@ -314,7 +314,7 @@ test("resolveModelTestTimeoutMs defaults ordinary model checks to 30 seconds", ( test("resolveModelTestTimeoutMs gives zai-web checks up to 60 seconds", () => { assert.equal(resolveModelTestTimeoutMs("zai-web", "glm-5.2", 30_000), 60_000); - assert.equal(resolveModelTestTimeoutMs("zai-web", "zai-web/GLM-5V-Turbo", 90_000), 90_000); + assert.equal(resolveModelTestTimeoutMs("zai-web", "zai-web/glm-5.3-flash", 90_000), 90_000); }); // --------------------------------------------------------------------------- diff --git a/tests/unit/zai-web-chat-endpoint-8014-probe.test.ts b/tests/unit/zai-web-chat-endpoint-8014-probe.test.ts index 926697d31e..46bb23266a 100644 --- a/tests/unit/zai-web-chat-endpoint-8014-probe.test.ts +++ b/tests/unit/zai-web-chat-endpoint-8014-probe.test.ts @@ -9,7 +9,7 @@ const TEST_TOKEN = "e30.eyJpZCI6InVzZXItMTIzIn0.sig"; /** * #8014 guard: the executor must target the versioned v2 completions endpoint, * never the stale unversioned `/api/chat/completions` path, which 404s - * model-independently as of 2026-07. + * model-independently. * * Setup notes for this flow (the executor now creates a remote chat first and * signs the completion request): @@ -43,7 +43,7 @@ test("#8014: ZaiWebExecutor must POST to the current chat.z.ai v2 chat-completio try { const executor = new mod.ZaiWebExecutor(); const result = await executor.execute({ - model: "glm-4.6", + model: "glm-5.3", body: { messages: [{ role: "user", content: "hello" }] }, stream: false, credentials: { @@ -55,7 +55,6 @@ test("#8014: ZaiWebExecutor must POST to the current chat.z.ai v2 chat-completio assert.ok(requested.length > 0, "the direct path must actually reach fetch"); assert.ok( - // Exact-URL match (not a substring test): `requested` holds whole URLs. !requested.some((url) => url === STALE_URL), `zai-web executor POSTed to the stale endpoint — matches #8014's model-independent 404 "Not Found"` ); diff --git a/tests/unit/zai-web-models-discovery-7678.test.ts b/tests/unit/zai-web-models-discovery-7678.test.ts index 82837c541f..380c4ba504 100644 --- a/tests/unit/zai-web-models-discovery-7678.test.ts +++ b/tests/unit/zai-web-models-discovery-7678.test.ts @@ -13,7 +13,7 @@ const providersDb = await import("../../src/lib/db/providers.ts"); const modelsRoute = await import("../../src/app/api/providers/[id]/models/route.ts"); const registry = await import("../../open-sse/config/providers/registry/zai-web/index.ts"); -const CURATED_ZAI_WEB_MODEL_IDS = ["glm-5.2", "GLM-5.1", "GLM-5-Turbo", "GLM-5v-Turbo"]; +const CURATED_ZAI_WEB_MODEL_IDS = ["glm-5.3-flash", "glm-5.3", "glm-5.2"]; async function resetStorage() { core.resetDbInstance(); @@ -31,34 +31,32 @@ test("zai-web publishes the live reasoning and vision capabilities", () => { registry.zai_webProvider.models.map((model) => ({ id: model.id, supportsReasoning: model.supportsReasoning === true, + supportedThinkingEfforts: model.supportedThinkingEfforts, supportsVision: model.supportsVision === true, toolCalling: model.toolCalling === true, })), [ + { + id: "glm-5.3-flash", + supportsReasoning: true, + supportedThinkingEfforts: ["low", "high", "max"], + supportsVision: true, + toolCalling: false, + }, + { + id: "glm-5.3", + supportsReasoning: true, + supportedThinkingEfforts: ["low", "high", "max"], + supportsVision: false, + toolCalling: false, + }, { id: "glm-5.2", supportsReasoning: true, + supportedThinkingEfforts: ["high", "max"], supportsVision: false, toolCalling: false, }, - { - id: "GLM-5.1", - supportsReasoning: true, - supportsVision: false, - toolCalling: false, - }, - { - id: "GLM-5-Turbo", - supportsReasoning: true, - supportsVision: false, - toolCalling: false, - }, - { - id: "GLM-5v-Turbo", - supportsReasoning: true, - supportsVision: true, - toolCalling: false, - }, ] ); }); From cdd07df700e2e1fea1fb0f866787b1d451bec762 Mon Sep 17 00:00:00 2001 From: backryun Date: Thu, 3 Sep 2026 19:50:02 +0900 Subject: [PATCH 017/143] feat(providers): refresh NVIDIA hosted models (#12538) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com #12524, #12538, #12277 e #12367 sobre o tip de release/v3.8.51: os quatro boardaram sem conflito (áreas disjuntas — zai-web, nvidia, clova, cursor/devin/fable). typecheck:core limpo, check:provider-consistency OK (272 entradas REGISTRY, 355 providers canônicos), check:known-symbols OK, e 305/305 nos testes tocados pelos quatro PRs. Os IDs de modelo adicionados foram conferidos individualmente. Obrigado, @backryun. --- README.md | 8 +- docs/diagrams/free-tier-budget.svg | 4 +- open-sse/config/freeModelCatalog.data.ts | 11 +- .../config/nvidiaHostedModels.snapshot.json | 24 ++-- .../config/providers/registry/nvidia/index.ts | 127 +++--------------- src/lib/providers/nvidiaValidationModel.ts | 9 +- tests/integration/freeModelBenchmarkShared.ts | 4 +- tests/unit/catalog-updates-v3x.test.ts | 17 ++- tests/unit/clinepass-thinking-budget.test.ts | 4 +- tests/unit/free-models.test.ts | 24 ++-- .../combo-vision-provider-id-12112.test.ts | 20 ++- .../unit/model-capabilities-registry.test.ts | 2 +- tests/unit/nvidia-410-model-scope.test.ts | 2 +- tests/unit/nvidia-eol-catalog.test.ts | 40 ++---- .../nvidia-minimax-m3-removed-3329.test.ts | 14 +- .../nvidia-nim-catalog-expansion-2373.test.ts | 81 ----------- tests/unit/nvidia-nim-registry-6108.test.ts | 50 +++++-- tests/unit/nvidia-nim-validator.test.ts | 2 +- .../nvidia-passthrough-models-6773.test.ts | 14 +- .../unit/nvidia-validation-model-3116.test.ts | 15 ++- .../opencode-go-effort-aliases-8353.test.ts | 9 -- 21 files changed, 156 insertions(+), 325 deletions(-) delete mode 100644 tests/unit/nvidia-nim-catalog-expansion-2373.test.ts diff --git a/README.md b/README.md index 60d5827d99..a026f9c9c3 100644 --- a/README.md +++ b/README.md @@ -17,9 +17,9 @@ -> Stacking free tiers by hand is painful — dozens of SDKs, dozens of rate limits, and no idea how much you actually have. OmniRoute catalogs **446 free-tier entries across 38 recurring pool keys** and computes the token headline from the **20 pools with a published positive monthly budget**, deduplicated by shared pool. The result stays visible on the dashboard (`/dashboard/free-tiers`). +> Stacking free tiers by hand is painful — dozens of SDKs, dozens of rate limits, and no idea how much you actually have. OmniRoute catalogs **437 free-tier entries across 38 recurring pool keys** and computes the token headline from the **20 pools with a published positive monthly budget**, deduplicated by shared pool. The result stays visible on the dashboard (`/dashboard/free-tiers`). -OmniRoute free-tier budget card: ~1.51B free tokens per month steady, up to ~2.13B in the first month with signup credits, from 38 documented recurring pool keys covering 446 cataloged free-tier entries behind one endpoint. Honest pool-deduped math — each shared pool counted once, including 20 recurring pools with a published positive monthly token budget; 13 providers are marked avoid in the terms-risk catalog so you decide. Budget bar includes Mistral 1B, LLM7 150M, Nara 150M, Gemini 60M and smaller pools, plus first-month signup credits and permanently-free no-token-cap providers surfaced separately so they never inflate the headline. Live used/remaining on /dashboard/free-tiers. +OmniRoute free-tier budget card: ~1.51B free tokens per month steady, up to ~2.13B in the first month with signup credits, from 38 documented recurring pool keys covering 437 cataloged free-tier entries behind one endpoint. Honest pool-deduped math — each shared pool counted once, including 20 recurring pools with a published positive monthly token budget; 13 providers are marked avoid in the terms-risk catalog so you decide. Budget bar includes Mistral 1B, LLM7 150M, Nara 150M, Gemini 60M and smaller pools, plus first-month signup credits and permanently-free no-token-cap providers surfaced separately so they never inflate the headline. Live used/remaining on /dashboard/free-tiers. > Animated summary of the live `/dashboard/free-tiers` page. Full methodology (pool dedupe, credit tiers, provider terms): **[docs/reference/FREE_TIERS.md](docs/reference/FREE_TIERS.md)**. > @@ -648,7 +648,7 @@ of your shell history. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md) -> **352 registered providers** across the canonical chat, media, search, local, cloud-agent and system collections, including **152 carrying `hasFree: true` discovery metadata**. The chat model registry covers **229 providers / 2,554 distinct provider-model pairs / 1,283 raw model IDs**; the separate free-budget catalog has **446 per-model rows**, **38 recurring pools** and **53 recurring/keyless free-forever providers**. These are different denominators by design; definitions and pool-deduped calculations live in the [Provider Reference](docs/reference/PROVIDER_REFERENCE.md) and [Free Tiers](docs/reference/FREE_TIERS.md). +> **352 registered providers** across the canonical chat, media, search, local, cloud-agent and system collections, including **152 carrying `hasFree: true` discovery metadata**. The chat model registry covers **229 providers / 2,554 distinct provider-model pairs / 1,283 raw model IDs**; the separate free-budget catalog has **437 per-model rows**, **38 recurring pools** and **53 recurring/keyless free-forever providers**. These are different denominators by design; definitions and pool-deduped calculations live in the [Provider Reference](docs/reference/PROVIDER_REFERENCE.md) and [Free Tiers](docs/reference/FREE_TIERS.md).
@@ -1270,7 +1270,7 @@ Métricas canônicas em 2026-08-24: **1.029 vídeos únicos** · **11.132.922 vi Resilience GuideCircuit breakers, cooldowns, queue, anti-thundering herd, TLS spoofing Auto-Combo Engine16-factor scoring, mode packs, self-healing Proxy Guide3-level proxy system, 1proxy marketplace, registry CRUD - Free TiersConsolidated directory: 38 documented recurring pools / 446 cataloged free-tier entries + Free TiersConsolidated directory: 38 documented recurring pools / 437 cataloged free-tier entries Features GalleryVisual dashboard tour with screenshots Codebase DocumentationBeginner-friendly codebase walkthrough diff --git a/docs/diagrams/free-tier-budget.svg b/docs/diagrams/free-tier-budget.svg index 51267b3ae2..72d1219cb2 100644 --- a/docs/diagrams/free-tier-budget.svg +++ b/docs/diagrams/free-tier-budget.svg @@ -1,4 +1,4 @@ - + Pool-deduplicated chart of the 20 recurring free-token pools with positive published budgets, plus signup credits and uncapped providers shown separately. @@ -64,7 +64,7 @@ ~1.51B FREE TOKENS / MONTH · STEADY up to ~2.13B in your first month — signup credits - documented free tiers · 38 recurring pools · 446 catalog entries · one endpoint + documented free tiers · 38 recurring pools · 437 catalog entries · one endpoint diff --git a/open-sse/config/freeModelCatalog.data.ts b/open-sse/config/freeModelCatalog.data.ts index 2d82b69382..f107d7fef4 100644 --- a/open-sse/config/freeModelCatalog.data.ts +++ b/open-sse/config/freeModelCatalog.data.ts @@ -277,18 +277,9 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "nscale", modelId: "openai/gpt-oss-20b", displayName: "openai/gpt-oss-20b", monthlyTokens: 0, creditTokens: 5000000, freeType: "one-time-initial", poolKey: "nscale", tos: "caution" }, { provider: "nscale", modelId: "meta-llama/Llama-4-Scout-17B-16E-Instruct", displayName: "meta-llama/Llama-4-Scout-17B-16E-Instruct", monthlyTokens: 0, creditTokens: 5000000, freeType: "one-time-initial", poolKey: "nscale", tos: "caution" }, { provider: "nscale", modelId: "meta-llama/Llama-3.3-70B-Instruct", displayName: "meta-llama/Llama-3.3-70B-Instruct", monthlyTokens: 0, creditTokens: 5000000, freeType: "one-time-initial", poolKey: "nscale", tos: "caution" }, - { provider: "nvidia", modelId: "z-ai/glm-5.2", displayName: "GLM 5.2", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, - { provider: "nvidia", modelId: "minimaxai/minimax-m2.7", displayName: "MiniMax M2.7", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, { provider: "nvidia", modelId: "google/gemma-4-31b-it", displayName: "Gemma 4 31B", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, - { provider: "nvidia", modelId: "mistralai/mistral-small-4-119b-2603", displayName: "Mistral Small 4 2603", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, - { provider: "nvidia", modelId: "mistralai/mistral-large-3-675b-instruct-2512", displayName: "Mistral Large 3 675B", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, - { provider: "nvidia", modelId: "mistralai/devstral-2-123b-instruct-2512", displayName: "Devstral 2 123B", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, - { provider: "nvidia", modelId: "qwen/qwen3.5-397b-a17b", displayName: "Qwen3.5-397B-A17B", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, - { provider: "nvidia", modelId: "qwen/qwen3.5-122b-a10b", displayName: "Qwen3.5-122B-A10B", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, - { provider: "nvidia", modelId: "stepfun-ai/step-3.5-flash", displayName: "Step 3.5 Flash", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, - { provider: "nvidia", modelId: "openai/gpt-oss-120b", displayName: "GPT OSS 120B", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, - { provider: "nvidia", modelId: "openai/gpt-oss-20b", displayName: "GPT OSS 20B", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, { provider: "nvidia", modelId: "nvidia/nemotron-3-super-120b-a12b", displayName: "Nemotron 3 Super 120B A12B", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, + { provider: "nvidia", modelId: "openai/gpt-oss-120b", displayName: "GPT OSS 120B", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, { provider: "ollama-cloud", modelId: "deepseek-v4-pro", displayName: "DeepSeek V4 Pro", monthlyTokens: 20000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "ollama-cloud", tos: "ambiguous" }, { provider: "ollama-cloud", modelId: "deepseek-v4-flash", displayName: "DeepSeek V4 Flash", monthlyTokens: 20000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "ollama-cloud", tos: "ambiguous" }, { provider: "ollama-cloud", modelId: "kimi-k2.6", displayName: "Kimi K2.6", monthlyTokens: 20000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "ollama-cloud", tos: "ambiguous" }, diff --git a/open-sse/config/nvidiaHostedModels.snapshot.json b/open-sse/config/nvidiaHostedModels.snapshot.json index 60d2f2e76d..fa94a6a4a7 100644 --- a/open-sse/config/nvidiaHostedModels.snapshot.json +++ b/open-sse/config/nvidiaHostedModels.snapshot.json @@ -1,16 +1,14 @@ [ - "google/gemma-4-31b-it", - "minimaxai/minimax-m2.7", - "mistralai/devstral-2-123b-instruct-2512", - "mistralai/mistral-large-3-675b-instruct-2512", - "mistralai/mistral-small-4-119b-2603", - "nvidia/nemotron-3-super-120b-a12b", - "openai/gpt-oss-120b", - "openai/gpt-oss-20b", + "moonshotai/kimi-k3", + "deepseek-ai/deepseek-v4-pro-0813", + "deepseek-ai/deepseek-v4-flash-0731", + "meta/muse-glimmer-30b", "poolside/laguna-xs-2.1", - "qwen/qwen3.5-122b-a10b", - "qwen/qwen3.5-397b-a17b", - "stepfun-ai/step-3.5-flash", - "thinkingmachines/inkling", - "z-ai/glm-5.2" + "google/gemma-4-31b-it", + "google/diffusiongemma-26b-a4b-it", + "nvidia/nemotron-3-ultra-550b-a55b", + "nvidia/nemotron-3-super-120b-a12b", + "nvidia/nemotron-3.5-lightning-30b-a3b", + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", + "openai/gpt-oss-120b" ] diff --git a/open-sse/config/providers/registry/nvidia/index.ts b/open-sse/config/providers/registry/nvidia/index.ts index 0966fa5941..dab45294d3 100644 --- a/open-sse/config/providers/registry/nvidia/index.ts +++ b/open-sse/config/providers/registry/nvidia/index.ts @@ -9,104 +9,34 @@ export const nvidiaProvider: RegistryEntry = { authType: "apikey", authHeader: "bearer", toolNameMaxLength: 64, - // #6773: nvidia multiplexes 17 models from 9 different upstream vendors - // (z-ai/, minimaxai/, deepseek-ai/, qwen/, mistralai/, stepfun-ai/, - // moonshotai/, openai/, nvidia/) behind ONE connection — mark it passthrough + // #6773: NVIDIA multiplexes models from multiple upstream vendors + // (moonshotai/, deepseek-ai/, nvidia/, meta/, poolside/, google/, openai/) + // behind ONE connection — mark it passthrough // so a single stale/renamed model's 404 locks out only that model instead // of cooling down the whole connection (see accountFallback.ts // hasPerModelQuota doc comment; matches modelscope/synthetic/kilo-gateway). passthroughModels: true, models: [ - // #6108: z-ai/glm-5.1 EOL'd 2026-07-02 (direct probe returns 410) — dropped. - // #10788: NVIDIA's hosted GLM-5.2 exposes a BINARY thinking switch - // (chat_template_kwargs.enable_thinking), not effort tiers — see - // mapNvidiaGlm52ReasoningParams. Declaring an empty tier list keeps the - // catalog from synthesizing unresolvable -low/-high/-max variant ids while - // still marking the model reasoning-capable. + { id: "moonshotai/kimi-k3", name: "Kimi K3" }, { - id: "z-ai/glm-5.2", - name: "GLM 5.2", + id: "deepseek-ai/deepseek-v4-pro-0813", + name: "DeepSeek V4 Pro 0813", supportsReasoning: true, - supportedThinkingEfforts: [], }, - // #3329/#6108: minimaxai/minimax-m3 stays excluded from the nvidia tier — it - // still 404s here for most callers; the single 200 probe in #6108 was not - // reproducible enough to override the #3329 guard. Re-add only once NVIDIA - // reliably serves it (and flip nvidia-minimax-m3-removed-3329.test.ts then). - { id: "minimaxai/minimax-m2.7", name: "MiniMax M2.7" }, + { + id: "deepseek-ai/deepseek-v4-flash-0731", + name: "DeepSeek V4 Flash 0731", + supportsReasoning: true, + }, + { id: "meta/muse-glimmer-30b", name: "Muse Glimmer 30B" }, + { id: "poolside/laguna-xs-2.1", name: "Laguna XS 2.1" }, { id: "google/gemma-4-31b-it", name: "Gemma 4 31B" }, - { id: "mistralai/mistral-small-4-119b-2603", name: "Mistral Small 4 2603" }, - { id: "mistralai/mistral-large-3-675b-instruct-2512", name: "Mistral Large 3 675B" }, - { id: "mistralai/devstral-2-123b-instruct-2512", name: "Devstral 2 123B" }, - { id: "qwen/qwen3.5-397b-a17b", name: "Qwen3.5-397B-A17B" }, - { id: "qwen/qwen3.5-122b-a10b", name: "Qwen3.5-122B-A10B" }, - { id: "stepfun-ai/step-3.5-flash", name: "Step 3.5 Flash" }, - { id: "stepfun-ai/step-3.7-flash", name: "Step 3.7 Flash" }, - // Sweep 2026-06-19: verified present in the live NVIDIA NIM /v1/models catalog. - { id: "moonshotai/kimi-k2.6", name: "Kimi K2.6" }, - { id: "openai/gpt-oss-120b", name: "GPT OSS 120B", toolCalling: false }, - { id: "openai/gpt-oss-20b", name: "GPT OSS 20B", toolCalling: false }, + { id: "google/diffusiongemma-26b-a4b-it", name: "DiffusionGemma 26B A4B IT" }, + { id: "nvidia/nemotron-3-ultra-550b-a55b", name: "Nemotron 3 Ultra 550B A55B" }, { id: "nvidia/nemotron-3-super-120b-a12b", name: "Nemotron 3 Super 120B A12B" }, - { id: "nvidia/nemotron-3-ultra-550b-a55b", name: "Nemotron 3 Ultra 550B" }, - // Port of decolua/9router#2373 ("fix(nvidia): expand NIM chat model catalog"): - // additional live-catalog models observed to serve /v1/chat/completions. - // `minimaxai/minimax-m3` from that PR is intentionally NOT re-added — it stays - // excluded per the #3329 guard (nvidia-minimax-m3-removed-3329.test.ts). - // Non-chat entries from the same PR (nvidia/gliner-pii — NER tagger, not a chat - // model; google/diffusiongemma-26b-a4b-it — diffusion model) are dropped for the - // same reason: this registry only models the /v1/chat/completions surface. - { id: "abacusai/dracarys-llama-3.1-70b-instruct", name: "Dracarys Llama 3.1 70B Instruct" }, - { id: "google/gemma-2-2b-it", name: "Gemma 2 2B IT" }, - { id: "google/gemma-3n-e2b-it", name: "Gemma 3n E2B IT" }, - { id: "meta/llama-3.1-8b-instruct", name: "Llama 3.1 8B Instruct", toolCalling: false }, { - id: "meta/llama-3.2-11b-vision-instruct", - name: "Llama 3.2 11B Vision Instruct", - supportsVision: true, - }, - { id: "meta/llama-3.2-1b-instruct", name: "Llama 3.2 1B Instruct" }, - { id: "meta/llama-3.2-3b-instruct", name: "Llama 3.2 3B Instruct", toolCalling: false }, - { - id: "meta/llama-3.2-90b-vision-instruct", - name: "Llama 3.2 90B Vision Instruct", - supportsVision: true, - }, - { id: "meta/llama-4-maverick-17b-128e-instruct", name: "Llama 4 Maverick 17B 128E Instruct" }, - { id: "meta/llama-guard-4-12b", name: "Llama Guard 4 12B", toolCalling: false }, - { id: "mistralai/ministral-14b-instruct-2512", name: "Ministral 14B Instruct 2512" }, - { id: "mistralai/mistral-medium-3.5-128b", name: "Mistral Medium 3.5 128B" }, - { id: "mistralai/mistral-nemotron", name: "Mistral Nemotron" }, - { id: "mistralai/mixtral-8x7b-instruct-v0.1", name: "Mixtral 8x7B Instruct v0.1" }, - { - id: "nvidia/ising-calibration-1-35b-a3b", - name: "Ising Calibration 1 35B A3B", - supportsReasoning: true, - }, - { - id: "nvidia/llama-3.1-nemoguard-8b-content-safety", - name: "Llama 3.1 Nemoguard 8B Content Safety", - }, - { - id: "nvidia/llama-3.1-nemoguard-8b-topic-control", - name: "Llama 3.1 Nemoguard 8B Topic Control", - }, - { id: "nvidia/llama-3.1-nemotron-nano-8b-v1", name: "Llama 3.1 Nemotron Nano 8B v1" }, - { - id: "nvidia/llama-3.1-nemotron-nano-vl-8b-v1", - name: "Llama 3.1 Nemotron Nano VL 8B v1", - supportsVision: true, - }, - { - id: "nvidia/llama-3.1-nemotron-safety-guard-8b-v3", - name: "Llama 3.1 Nemotron Safety Guard 8B v3", - }, - { id: "nvidia/llama-3.3-nemotron-super-49b-v1", name: "Llama 3.3 Nemotron Super 49B v1" }, - { id: "nvidia/llama-3.3-nemotron-super-49b-v1.5", name: "Llama 3.3 Nemotron Super 49B v1.5" }, - { id: "nvidia/nemotron-3-content-safety", name: "Nemotron 3 Content Safety" }, - { - id: "nvidia/nemotron-3-nano-30b-a3b", - name: "Nemotron 3 Nano 30B A3B", - supportsReasoning: true, + id: "nvidia/nemotron-3.5-lightning-30b-a3b", + name: "Nemotron 3.5 Lightning 30B A3B", }, { id: "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", @@ -114,27 +44,6 @@ export const nvidiaProvider: RegistryEntry = { supportsReasoning: true, supportsVision: true, }, - { id: "nvidia/nemotron-3.5-content-safety", name: "Nemotron 3.5 Content Safety" }, - { id: "nvidia/nemotron-mini-4b-instruct", name: "Nemotron Mini 4B Instruct" }, - { - id: "nvidia/nemotron-nano-12b-v2-vl", - name: "Nemotron Nano 12B v2 VL", - supportsReasoning: true, - supportsVision: true, - }, - { - id: "nvidia/nvidia-nemotron-nano-9b-v2", - name: "NVIDIA Nemotron Nano 9B v2", - supportsReasoning: true, - }, - { id: "nvidia/riva-translate-4b-instruct-v1.1", name: "Riva Translate 4B Instruct v1.1" }, - { - id: "qwen/qwen3-next-80b-a3b-instruct", - name: "Qwen3 Next 80B A3B Instruct", - supportsReasoning: true, - }, - { id: "sarvamai/sarvam-m", name: "Sarvam M" }, - { id: "stockmark/stockmark-2-100b-instruct", name: "Stockmark 2 100B Instruct" }, - { id: "upstage/solar-10.7b-instruct", name: "Solar 10.7B Instruct" }, + { id: "openai/gpt-oss-120b", name: "GPT OSS 120B", toolCalling: false }, ], }; diff --git a/src/lib/providers/nvidiaValidationModel.ts b/src/lib/providers/nvidiaValidationModel.ts index bde123944b..07ef6291b4 100644 --- a/src/lib/providers/nvidiaValidationModel.ts +++ b/src/lib/providers/nvidiaValidationModel.ts @@ -9,11 +9,12 @@ * probe HANG until the validation timeout, which surfaces as a misleading "Upstream * Error" on an otherwise-valid key. * - * `meta/llama-3.1-8b-instruct` is a long-lived, universally-available NIM model (no - * special permission), so it is a far more reliable auth probe. A connection may still - * override it via `providerSpecificData.validationModelId`. + * The default must stay inside the current NVIDIA hosted-model catalog. Nemotron 3.5 + * Lightning is the smallest retained general chat model, which keeps the auth probe + * lightweight. A connection may still override it via + * `providerSpecificData.validationModelId`. */ -export const NVIDIA_DEFAULT_VALIDATION_MODEL = "meta/llama-3.1-8b-instruct"; +export const NVIDIA_DEFAULT_VALIDATION_MODEL = "nvidia/nemotron-3.5-lightning-30b-a3b"; export function resolveNvidiaValidationModel(providerSpecificData?: { validationModelId?: unknown; diff --git a/tests/integration/freeModelBenchmarkShared.ts b/tests/integration/freeModelBenchmarkShared.ts index caf5cd8e78..e732c6e0b6 100644 --- a/tests/integration/freeModelBenchmarkShared.ts +++ b/tests/integration/freeModelBenchmarkShared.ts @@ -56,8 +56,8 @@ export const FREE_MODELS: FreeModelSpec[] = [ displayName: "Gemini 3.1 Flash-Lite", }, { provider: "gemini", model: "gemini/gemma-4-31b-it", displayName: "Gemma 4 31B (Gemini)" }, - { provider: "nvidia", model: "nvidia/openai/gpt-oss-20b", displayName: "GPT OSS 20B (NVIDIA)" }, - { provider: "nvidia", model: "nvidia/z-ai/glm-5.1", displayName: "GLM 5.1 (NVIDIA)" }, + { provider: "nvidia", model: "nvidia/openai/gpt-oss-120b", displayName: "GPT OSS 120B (NVIDIA)" }, + { provider: "nvidia", model: "nvidia/moonshotai/kimi-k3", displayName: "Kimi K3 (NVIDIA)" }, { provider: "nvidia", model: "nvidia/google/gemma-4-31b-it", diff --git a/tests/unit/catalog-updates-v3x.test.ts b/tests/unit/catalog-updates-v3x.test.ts index e95132bec6..e233380490 100644 --- a/tests/unit/catalog-updates-v3x.test.ts +++ b/tests/unit/catalog-updates-v3x.test.ts @@ -23,18 +23,21 @@ test("Pollinations catalog mirrors the current public text model lineup", () => ); }); -test("NVIDIA catalog includes the verified 2026 additions and GPT OSS 20B alias resolution", () => { +test("NVIDIA catalog includes the current hosted models and GPT OSS 120B alias resolution", () => { const ids = new Set(getModelsByProviderId("nvidia").map((model) => model.id)); - assert.ok(ids.has("openai/gpt-oss-20b")); + assert.ok(ids.has("moonshotai/kimi-k3")); + assert.ok(ids.has("deepseek-ai/deepseek-v4-pro-0813")); + assert.ok(ids.has("deepseek-ai/deepseek-v4-flash-0731")); + assert.ok(ids.has("nvidia/nemotron-3.5-lightning-30b-a3b")); + assert.ok(ids.has("meta/muse-glimmer-30b")); + assert.ok(ids.has("google/diffusiongemma-26b-a4b-it")); + assert.ok(ids.has("openai/gpt-oss-120b")); assert.ok(ids.has("nvidia/nemotron-3-super-120b-a12b")); - assert.ok(ids.has("mistralai/mistral-large-3-675b-instruct-2512")); - assert.ok(ids.has("qwen/qwen3.5-397b-a17b")); - assert.ok(ids.has("mistralai/devstral-2-123b-instruct-2512")); - assert.deepEqual(resolveCanonicalProviderModel("nvidia", "gpt-oss-20b"), { + assert.deepEqual(resolveCanonicalProviderModel("nvidia", "gpt-oss-120b"), { provider: "nvidia", - model: "openai/gpt-oss-20b", + model: "openai/gpt-oss-120b", }); }); diff --git a/tests/unit/clinepass-thinking-budget.test.ts b/tests/unit/clinepass-thinking-budget.test.ts index b1452aedaa..7883294d0d 100644 --- a/tests/unit/clinepass-thinking-budget.test.ts +++ b/tests/unit/clinepass-thinking-budget.test.ts @@ -83,11 +83,11 @@ test("bumps undersized max_tokens for a non-clinepass reasoning provider (gate r // Nemotron Nano with supportsReasoning in the NVIDIA registry. const executor = new DefaultExecutor("nvidia"); const body = { - model: "nvidia/nvidia-nemotron-nano-9b-v2", + model: "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", reasoning_effort: "high", max_tokens: 100, } as Record; - executor.ensureThinkingBudget(body, "nvidia/nvidia-nemotron-nano-9b-v2"); + executor.ensureThinkingBudget(body, "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning"); assert.equal(body.max_tokens, 4096); }); diff --git a/tests/unit/free-models.test.ts b/tests/unit/free-models.test.ts index e857db1bdf..0d4ae6a566 100644 --- a/tests/unit/free-models.test.ts +++ b/tests/unit/free-models.test.ts @@ -57,15 +57,12 @@ test("isFreeModel: a model id listed in the free catalog for that provider is fr assert.equal(isFreeModel(sample.provider, { id: sample.modelId }), true); }); -test("isFreeModel: NVIDIA GLM 5.2 is included in the reviewed trial catalog", () => { - assert.equal(isFreeModel("nvidia", { id: "z-ai/glm-5.2" }), true); +test("isFreeModel: NVIDIA GPT OSS 120B remains in the reviewed trial catalog", () => { + assert.equal(isFreeModel("nvidia", { id: "openai/gpt-oss-120b" }), true); }); test("selectModelsForImport: passthrough when importFreeOnly is false", () => { - const models = [ - { id: "a:free" }, - { id: "b", pricing: { prompt: "0.01", completion: "0.02" } }, - ]; + const models = [{ id: "a:free" }, { id: "b", pricing: { prompt: "0.01", completion: "0.02" } }]; const result = selectModelsForImport("openrouter", models, false); assert.equal(result.models.length, 2); assert.equal(result.freeFilterEmpty, false); @@ -130,8 +127,14 @@ test("sortModelsFreeFirst: deterministic (alphabetical) within each group, regar ], { isFree: (m) => m.isFree, key: (m) => m.id } ); - assert.deepEqual(a.map((m) => m.id), ["a", "b", "c"]); - assert.deepEqual(b.map((m) => m.id), ["a", "b", "c"]); + assert.deepEqual( + a.map((m) => m.id), + ["a", "b", "c"] + ); + assert.deepEqual( + b.map((m) => m.id), + ["a", "b", "c"] + ); }); test("sortModelsFreeFirst: does not mutate the input array", () => { @@ -141,5 +144,8 @@ test("sortModelsFreeFirst: does not mutate the input array", () => { ]; const before = items.map((m) => m.id); sortModelsFreeFirst(items, { isFree: (m) => m.isFree, key: (m) => m.id }); - assert.deepEqual(items.map((m) => m.id), before); + assert.deepEqual( + items.map((m) => m.id), + before + ); }); diff --git a/tests/unit/guardrails/combo-vision-provider-id-12112.test.ts b/tests/unit/guardrails/combo-vision-provider-id-12112.test.ts index 734e13e2fe..c4ddaf6fe9 100644 --- a/tests/unit/guardrails/combo-vision-provider-id-12112.test.ts +++ b/tests/unit/guardrails/combo-vision-provider-id-12112.test.ts @@ -3,29 +3,27 @@ import assert from "node:assert/strict"; process.env.DATA_DIR = `/tmp/omniroute-test-12112-${Date.now()}`; -const { getComboVisionBridgeDecision } = await import( - "../../../src/lib/guardrails/visionBridge.ts" -); +const { getComboVisionBridgeDecision } = + await import("../../../src/lib/guardrails/visionBridge.ts"); const combosDb = await import("../../../src/lib/db/combos.ts"); const core = await import("../../../src/lib/db/core.ts"); -const { isVisionIncompatibleTarget } = await import( - "../../../open-sse/services/combo/comboStructure.ts" -); +const { isVisionIncompatibleTarget } = + await import("../../../open-sse/services/combo/comboStructure.ts"); import type { ResolvedComboTarget } from "../../../open-sse/services/combo/types.ts"; test.after(() => { core.resetDbInstance(); }); -test("#12112: checkComboVision respects providerId for namespaced vision models (e.g. nvidia/nemotron-nano-12b-v2-vl)", async () => { - // Model 'nvidia/nemotron-nano-12b-v2-vl' is declared with supportsVision: true in nvidia provider registry. +test("#12112: checkComboVision respects providerId for namespaced vision models (e.g. nvidia/nemotron-3-nano-omni-30b-a3b-reasoning)", async () => { + // Model 'nvidia/nemotron-3-nano-omni-30b-a3b-reasoning' is declared with supportsVision: true in nvidia provider registry. // It has a slash in model id and requires providerId="nvidia" to resolve capabilities. await combosDb.createCombo({ name: "nvidia-vision-combo-12112", models: [ { providerId: "nvidia", - model: "nvidia/nemotron-nano-12b-v2-vl", + model: "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", weight: 1, }, ], @@ -44,7 +42,7 @@ test("#12112: isVisionIncompatibleTarget passes providerId to resolve vision cap kind: "model", stepId: "step-1", executionKey: "step-1", - modelStr: "nvidia/nemotron-nano-12b-v2-vl", + modelStr: "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", provider: "nvidia", providerId: "nvidia", connectionId: "conn-1", @@ -64,6 +62,6 @@ test("#12112: isVisionIncompatibleTarget passes providerId to resolve vision cap assert.equal( incompatible, false, - "Target with providerId='nvidia' and model='nvidia/nemotron-nano-12b-v2-vl' must be vision-compatible" + "Target with providerId='nvidia' and model='nvidia/nemotron-3-nano-omni-30b-a3b-reasoning' must be vision-compatible" ); }); diff --git a/tests/unit/model-capabilities-registry.test.ts b/tests/unit/model-capabilities-registry.test.ts index 7bb82681db..98adbbf21a 100644 --- a/tests/unit/model-capabilities-registry.test.ts +++ b/tests/unit/model-capabilities-registry.test.ts @@ -203,7 +203,7 @@ test("GPT OSS and DeepSeek Reasoner models support tool calling", () => { // GPT OSS models should not be blocked by the heuristic assert.equal(modelCapabilities.supportsToolCalling("fake-provider/gpt-oss-120b"), true); assert.equal(modelCapabilities.supportsToolCalling("gpt-oss-120b"), true); - assert.equal(modelCapabilities.supportsToolCalling("nvidia/openai/gpt-oss-20b"), false); // in registry + assert.equal(modelCapabilities.supportsToolCalling("nvidia/openai/gpt-oss-120b"), false); // in registry // DeepSeek Reasoner supports tool calling assert.equal(modelCapabilities.supportsToolCalling("deepseek-reasoner"), true); diff --git a/tests/unit/nvidia-410-model-scope.test.ts b/tests/unit/nvidia-410-model-scope.test.ts index 0d72fe1345..9865e1f5f8 100644 --- a/tests/unit/nvidia-410-model-scope.test.ts +++ b/tests/unit/nvidia-410-model-scope.test.ts @@ -15,7 +15,7 @@ const auth = await import("../../src/sse/services/auth.ts"); const fallback = await import("../../open-sse/services/accountFallback.ts"); const DEAD_MODEL = "deepseek-ai/deepseek-v4-pro"; -const HEALTHY_MODEL = "z-ai/glm-5.2"; +const HEALTHY_MODEL = "nvidia/nemotron-3.5-lightning-30b-a3b"; const GONE_BODY = JSON.stringify({ type: "about:blank", diff --git a/tests/unit/nvidia-eol-catalog.test.ts b/tests/unit/nvidia-eol-catalog.test.ts index ea8708d19e..f5319d08e0 100644 --- a/tests/unit/nvidia-eol-catalog.test.ts +++ b/tests/unit/nvidia-eol-catalog.test.ts @@ -13,20 +13,15 @@ const documentedFreeIds = new Set( const reviewedIds = new Set(reviewedLiveIds); -test("NVIDIA registry excludes retired DeepSeek V4 models", () => { - assert.ok( - !registryIds.has("deepseek-ai/deepseek-v4-pro"), - "retired deepseek-ai/deepseek-v4-pro must not be advertised" - ); - - assert.ok( - !registryIds.has("deepseek-ai/deepseek-v4-flash"), - "retired deepseek-ai/deepseek-v4-flash must not be advertised" - ); -}); - -test("NVIDIA static lifecycle metadata excludes known EOL models", () => { - for (const modelId of ["z-ai/glm-5.1", "deepseek-ai/deepseek-v4-pro"]) { +test("NVIDIA static catalog metadata excludes superseded model ids", () => { + for (const modelId of [ + "z-ai/glm-5.1", + "z-ai/glm-5.2", + "deepseek-ai/deepseek-v4-pro", + "deepseek-ai/deepseek-v4-flash", + "minimaxai/minimax-m2.7", + ]) { + assert.ok(!registryIds.has(modelId), `${modelId} must not remain in the NVIDIA registry`); assert.ok( !reviewedIds.has(modelId), `${modelId} must not remain in the reviewed NVIDIA hosted-model snapshot` @@ -39,16 +34,9 @@ test("NVIDIA static lifecycle metadata excludes known EOL models", () => { } }); -test("NVIDIA cleanup preserves the healthy GLM replacement", () => { - assert.ok(registryIds.has("z-ai/glm-5.2"), "z-ai/glm-5.2 must remain in the NVIDIA registry"); - - assert.ok( - reviewedIds.has("z-ai/glm-5.2"), - "z-ai/glm-5.2 must remain in the reviewed NVIDIA hosted-model snapshot" - ); - - assert.ok( - documentedFreeIds.has("z-ai/glm-5.2"), - "z-ai/glm-5.2 must remain in the NVIDIA free-model catalog" - ); +test("NVIDIA reviewed snapshot matches the registry and trial entries remain valid", () => { + assert.deepEqual([...reviewedIds], [...registryIds]); + for (const modelId of documentedFreeIds) { + assert.ok(registryIds.has(modelId), `${modelId} must exist in the NVIDIA hosted catalog`); + } }); diff --git a/tests/unit/nvidia-minimax-m3-removed-3329.test.ts b/tests/unit/nvidia-minimax-m3-removed-3329.test.ts index 77e0ce4822..f0c8b6d502 100644 --- a/tests/unit/nvidia-minimax-m3-removed-3329.test.ts +++ b/tests/unit/nvidia-minimax-m3-removed-3329.test.ts @@ -4,16 +4,16 @@ import assert from "node:assert/strict"; const { getRegistryEntry } = await import("../../open-sse/config/providerRegistry.ts"); // #3329: `minimaxai/minimax-m3` was registered in the nvidia (NVIDIA NIM) tier, -// but NVIDIA NIM does not host it — every request returns `404 page not found`, -// while sibling models on the same provider (e.g. `minimaxai/minimax-m2.7`) -// work. Advertising a model that 404s is a catalog bug; it is removed from the -// nvidia tier until NVIDIA actually serves it. It remains on the tiers that do -// (minimax / minimax-cn / opencode / etc.). +// but NVIDIA NIM does not host it — every request returns `404 page not found`. +// Advertising a model that 404s is a catalog bug; it stays absent from the +// NVIDIA tier while remaining available from providers that actually serve it. test("nvidia tier does not advertise minimaxai/minimax-m3 (404 upstream) (#3329)", () => { const nvidia = getRegistryEntry("nvidia"); assert.ok(nvidia, "nvidia registry entry must exist"); const ids = (nvidia.models ?? []).map((m) => m.id); assert.ok(!ids.includes("minimaxai/minimax-m3"), "minimaxai/minimax-m3 must not be in nvidia"); - // sanity: the working sibling stays listed - assert.ok(ids.includes("minimaxai/minimax-m2.7"), "minimaxai/minimax-m2.7 stays available"); + assert.ok( + !ids.includes("minimaxai/minimax-m2.7"), + "removed minimaxai/minimax-m2.7 must stay out" + ); }); diff --git a/tests/unit/nvidia-nim-catalog-expansion-2373.test.ts b/tests/unit/nvidia-nim-catalog-expansion-2373.test.ts deleted file mode 100644 index 32d14a8a00..0000000000 --- a/tests/unit/nvidia-nim-catalog-expansion-2373.test.ts +++ /dev/null @@ -1,81 +0,0 @@ -import test from "node:test"; -import assert from "node:assert/strict"; - -import { nvidiaProvider } from "../../open-sse/config/providers/registry/nvidia/index.ts"; - -// Port of decolua/9router#2373 ("fix(nvidia): expand NIM chat model catalog"). Upstream's -// PR also added a per-model `thinkingFormat`/`kind` capability shape in a legacy -// open-sse/providers/capabilities.js file that has no equivalent in OmniRoute — reasoning -// translation here is per-PROVIDER (open-sse/translator/paramSupport.ts, -// executors/default.ts, both gated on `this.provider === "nvidia"`), not per-model, so -// only the catalog (RegistryModel.supportsReasoning/supportsVision) needed porting. -// Embedding/ASR/TTS entries from the same upstream PR are already covered by -// open-sse/config/embeddingRegistry.ts and audioRegistry.ts, so they are not duplicated -// here. `minimaxai/minimax-m3` is intentionally excluded — see the #3329 guard -// (nvidia-minimax-m3-removed-3329.test.ts). -const modelIds = new Set(nvidiaProvider.models.map((m) => m.id)); - -test("#2373: NVIDIA NIM registry gains the newly-observed chat-completions models", () => { - for (const id of [ - "abacusai/dracarys-llama-3.1-70b-instruct", - "google/gemma-2-2b-it", - "google/gemma-3n-e2b-it", - "meta/llama-3.1-8b-instruct", - "meta/llama-3.2-11b-vision-instruct", - "meta/llama-4-maverick-17b-128e-instruct", - "meta/llama-guard-4-12b", - "mistralai/ministral-14b-instruct-2512", - "mistralai/mistral-medium-3.5-128b", - "mistralai/mistral-nemotron", - "mistralai/mixtral-8x7b-instruct-v0.1", - "nvidia/ising-calibration-1-35b-a3b", - "nvidia/llama-3.1-nemoguard-8b-content-safety", - "nvidia/llama-3.3-nemotron-super-49b-v1.5", - "nvidia/nemotron-3-nano-30b-a3b", - "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", - "nvidia/nemotron-nano-12b-v2-vl", - "nvidia/nvidia-nemotron-nano-9b-v2", - "qwen/qwen3-next-80b-a3b-instruct", - "sarvamai/sarvam-m", - "stockmark/stockmark-2-100b-instruct", - "upstage/solar-10.7b-instruct", - ]) { - assert.ok(modelIds.has(id), `expected nvidia registry to include ${id}`); - } -}); - -test("#2373: reasoning-capable NVIDIA-hosted models are flagged supportsReasoning", () => { - const reasoningIds = [ - "nvidia/ising-calibration-1-35b-a3b", - "nvidia/nemotron-3-nano-30b-a3b", - "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", - "nvidia/nemotron-nano-12b-v2-vl", - "nvidia/nvidia-nemotron-nano-9b-v2", - "qwen/qwen3-next-80b-a3b-instruct", - ]; - for (const id of reasoningIds) { - const model = nvidiaProvider.models.find((m) => m.id === id); - assert.ok(model, `model ${id} must exist`); - assert.equal(model?.supportsReasoning, true, `${id} must be supportsReasoning: true`); - } -}); - -test("#2373/#3329: minimaxai/minimax-m3 stays excluded from the nvidia tier", () => { - assert.ok( - !modelIds.has("minimaxai/minimax-m3"), - "minimaxai/minimax-m3 must not be re-added to the nvidia registry (404 upstream, #3329)" - ); - // sanity: the working sibling stays listed - assert.ok(modelIds.has("minimaxai/minimax-m2.7"), "minimaxai/minimax-m2.7 stays available"); -}); - -test("#2373: non-chat model kinds (NER/diffusion) from the upstream PR are not ported into the chat registry", () => { - assert.ok( - !modelIds.has("nvidia/gliner-pii"), - "nvidia/gliner-pii is an NER/PII tagger, not a chat-completions model" - ); - assert.ok( - !modelIds.has("google/diffusiongemma-26b-a4b-it"), - "google/diffusiongemma-26b-a4b-it is a diffusion model, not a chat-completions model" - ); -}); diff --git a/tests/unit/nvidia-nim-registry-6108.test.ts b/tests/unit/nvidia-nim-registry-6108.test.ts index b9c12d9992..702747656d 100644 --- a/tests/unit/nvidia-nim-registry-6108.test.ts +++ b/tests/unit/nvidia-nim-registry-6108.test.ts @@ -3,21 +3,45 @@ import assert from "node:assert/strict"; import { nvidiaProvider } from "../../open-sse/config/providers/registry/nvidia/index.ts"; -// Regression guard for #6108: the static NVIDIA NIM model registry had gone -// stale — z-ai/glm-5.1 was EOL'd (410) 2026-07-02, while glm-5.2 and -// nvidia/nemotron-3-ultra-550b-a55b were absent. minimaxai/minimax-m3 stays -// excluded per the #3329 guard (nvidia-minimax-m3-removed-3329.test.ts) — the -// single 200 probe in #6108 wasn't reproducible enough to override it. -const modelIds = new Set(nvidiaProvider.models.map((m) => m.id)); +const EXPECTED_MODEL_IDS = [ + "moonshotai/kimi-k3", + "deepseek-ai/deepseek-v4-pro-0813", + "deepseek-ai/deepseek-v4-flash-0731", + "meta/muse-glimmer-30b", + "poolside/laguna-xs-2.1", + "google/gemma-4-31b-it", + "google/diffusiongemma-26b-a4b-it", + "nvidia/nemotron-3-ultra-550b-a55b", + "nvidia/nemotron-3-super-120b-a12b", + "nvidia/nemotron-3.5-lightning-30b-a3b", + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", + "openai/gpt-oss-120b", +] as const; -test("#6108: NVIDIA NIM registry contains the refreshed live models", () => { - assert.ok(modelIds.has("z-ai/glm-5.2"), "z-ai/glm-5.2 must be present"); - assert.ok( - modelIds.has("nvidia/nemotron-3-ultra-550b-a55b"), - "nvidia/nemotron-3-ultra-550b-a55b must be present" +test("NVIDIA NIM registry exactly matches the current hosted-model catalog", () => { + assert.deepEqual( + nvidiaProvider.models.map((model) => model.id), + EXPECTED_MODEL_IDS ); }); -test("#6108: NVIDIA NIM registry no longer lists EOL z-ai/glm-5.1", () => { - assert.ok(!modelIds.has("z-ai/glm-5.1"), "EOL z-ai/glm-5.1 must be removed"); +test("NVIDIA NIM registry preserves known model capabilities", () => { + const byId = new Map(nvidiaProvider.models.map((model) => [model.id, model])); + + for (const id of [ + "deepseek-ai/deepseek-v4-pro-0813", + "deepseek-ai/deepseek-v4-flash-0731", + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", + ]) { + assert.equal(byId.get(id)?.supportsReasoning, true, `${id} must support reasoning`); + } + + const omni = byId.get("nvidia/nemotron-3-nano-omni-30b-a3b-reasoning"); + assert.equal(omni?.supportsVision, true, "Nemotron 3 Nano Omni must support vision"); + + assert.equal( + byId.get("openai/gpt-oss-120b")?.toolCalling, + false, + "openai/gpt-oss-120b must keep tool calling disabled" + ); }); diff --git a/tests/unit/nvidia-nim-validator.test.ts b/tests/unit/nvidia-nim-validator.test.ts index f6f82e44f6..b6c2c5518d 100644 --- a/tests/unit/nvidia-nim-validator.test.ts +++ b/tests/unit/nvidia-nim-validator.test.ts @@ -135,7 +135,7 @@ test("nvidia specialty validator falls back to stable chat validation model", as calls.some((u) => u.endsWith("/chat/completions")), `should fall back to /chat/completions, called: ${JSON.stringify(calls)}` ); - assert.equal(payload?.model, "meta/llama-3.1-8b-instruct"); + assert.equal(payload?.model, "nvidia/nemotron-3.5-lightning-30b-a3b"); } ); }); diff --git a/tests/unit/nvidia-passthrough-models-6773.test.ts b/tests/unit/nvidia-passthrough-models-6773.test.ts index 915b541413..9275367845 100644 --- a/tests/unit/nvidia-passthrough-models-6773.test.ts +++ b/tests/unit/nvidia-passthrough-models-6773.test.ts @@ -2,8 +2,8 @@ * Regression test for #6773 — NVIDIA NIM models listed available:true but 404 at router. * * Root cause: the `nvidia` provider registry entry multiplexes many distinct - * third-party vendor models (z-ai/, minimaxai/, deepseek-ai/, qwen/, - * mistralai/, stepfun-ai/, moonshotai/, openai/, nvidia/) behind ONE base URL + * third-party vendor models (moonshotai/, deepseek-ai/, nvidia/, meta/, + * poolside/, google/, openai/) behind ONE base URL * and ONE API key connection — architecturally identical to `modelscope`, * `synthetic`, and `kilo-gateway`, which all set `passthroughModels: true` so * that a single model's 404/429 stays scoped to that model instead of cooling @@ -24,8 +24,8 @@ test("#6773: nvidia registry entry sets passthroughModels", () => { entry?.passthroughModels, true, "nvidia multiplexes many third-party vendor models behind one connection " + - "(z-ai/, minimaxai/, deepseek-ai/, qwen/, mistralai/, stepfun-ai/, " + - "moonshotai/, openai/, nvidia/) — it should set passthroughModels: true " + + "(moonshotai/, deepseek-ai/, nvidia/, meta/, poolside/, google/, " + + "openai/) — it should set passthroughModels: true " + "like modelscope/synthetic/kilo-gateway, so a single stale model 404 " + "does not cool down the whole connection for all other models" ); @@ -33,7 +33,7 @@ test("#6773: nvidia registry entry sets passthroughModels", () => { test("#6773: hasPerModelQuota('nvidia') is true, so a 404 on one nvidia model is model-scoped", () => { assert.equal( - accountFallback.hasPerModelQuota("nvidia", "z-ai/glm-5.2"), + accountFallback.hasPerModelQuota("nvidia", "nvidia/nemotron-3.5-lightning-30b-a3b"), true, "expected nvidia to use per-model lockout (like gemini/github/codex/compatible " + "providers) so a 404 on one model doesn't cool down the other nvidia models" @@ -50,7 +50,7 @@ test("#6773: checkFallbackError + lockModelIfPerModelQuota scope a single-model 404, "Not Found", 0, - "z-ai/glm-5.2", + "nvidia/nemotron-3.5-lightning-30b-a3b", "nvidia", null, null, @@ -66,7 +66,7 @@ test("#6773: checkFallbackError + lockModelIfPerModelQuota scope a single-model const locked = accountFallback.lockModelIfPerModelQuota( "nvidia", "conn-6773", - "z-ai/glm-5.2", + "nvidia/nemotron-3.5-lightning-30b-a3b", "unknown", result.cooldownMs ?? 30_000 ); diff --git a/tests/unit/nvidia-validation-model-3116.test.ts b/tests/unit/nvidia-validation-model-3116.test.ts index 1d52f68059..da7e541fd2 100644 --- a/tests/unit/nvidia-validation-model-3116.test.ts +++ b/tests/unit/nvidia-validation-model-3116.test.ts @@ -2,7 +2,7 @@ * #3116 — NVIDIA key validation probed the first catalog model (`z-ai/glm-5.1`), which * requires the "Public API Endpoints" account permission and can hang/be DEGRADED, * making a *valid* key fail with a misleading "Upstream Error". The probe now defaults to - * the universally-available `meta/llama-3.1-8b-instruct`, with a per-connection override. + * a lightweight model from the current hosted catalog, with a per-connection override. */ import test from "node:test"; import assert from "node:assert/strict"; @@ -12,10 +12,10 @@ import { resolveNvidiaValidationModel, } from "../../src/lib/providers/nvidiaValidationModel.ts"; -test("defaults to a stable, permission-free NVIDIA model (not the gated glm-5.1)", () => { - assert.equal(NVIDIA_DEFAULT_VALIDATION_MODEL, "meta/llama-3.1-8b-instruct"); - assert.equal(resolveNvidiaValidationModel(), "meta/llama-3.1-8b-instruct"); - assert.equal(resolveNvidiaValidationModel({}), "meta/llama-3.1-8b-instruct"); +test("defaults to a lightweight model in the current NVIDIA catalog", () => { + assert.equal(NVIDIA_DEFAULT_VALIDATION_MODEL, "nvidia/nemotron-3.5-lightning-30b-a3b"); + assert.equal(resolveNvidiaValidationModel(), "nvidia/nemotron-3.5-lightning-30b-a3b"); + assert.equal(resolveNvidiaValidationModel({}), "nvidia/nemotron-3.5-lightning-30b-a3b"); assert.notEqual(resolveNvidiaValidationModel(undefined), "z-ai/glm-5.1"); }); @@ -25,5 +25,8 @@ test("honors a per-connection validationModelId override", () => { "nvidia/llama-3.3-nemotron-super-49b" ); // blank/whitespace override falls back to the default - assert.equal(resolveNvidiaValidationModel({ validationModelId: " " }), NVIDIA_DEFAULT_VALIDATION_MODEL); + assert.equal( + resolveNvidiaValidationModel({ validationModelId: " " }), + NVIDIA_DEFAULT_VALIDATION_MODEL + ); }); diff --git a/tests/unit/opencode-go-effort-aliases-8353.test.ts b/tests/unit/opencode-go-effort-aliases-8353.test.ts index 762de5fcd2..c743623f93 100644 --- a/tests/unit/opencode-go-effort-aliases-8353.test.ts +++ b/tests/unit/opencode-go-effort-aliases-8353.test.ts @@ -283,12 +283,3 @@ test("#10788 registry base rows declare the same tiers EFFORT_TIERS parses", () } } }); - -test("#10788 nvidia z-ai/glm-5.2 declares reasoning with an empty tier list (binary switch)", () => { - const entry = REGISTRY["nvidia"]; - assert.ok(entry?.models, "nvidia must expose models"); - const row = entry.models.find((m) => m.id === "z-ai/glm-5.2"); - assert.ok(row, "nvidia z-ai/glm-5.2 must exist"); - assert.equal(row.supportsReasoning, true); - assert.deepEqual(row.supportedThinkingEfforts, []); -}); From 82c64d76d32fa4e4469e37f529e46372bd3e5db5 Mon Sep 17 00:00:00 2001 From: backryun Date: Thu, 3 Sep 2026 19:50:22 +0900 Subject: [PATCH 018/143] feat(providers): modernize CLOVA Studio chat and embeddings (#12277) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com #12524, #12538, #12277 e #12367 sobre o tip de release/v3.8.51: os quatro boardaram sem conflito (áreas disjuntas — zai-web, nvidia, clova, cursor/devin/fable). typecheck:core limpo, check:provider-consistency OK (272 entradas REGISTRY, 355 providers canônicos), check:known-symbols OK, e 305/305 nos testes tocados pelos quatro PRs. Os IDs de modelo adicionados foram conferidos individualmente. Obrigado, @backryun. --- open-sse/config/embeddingRegistry.ts | 128 +- .../providers/registry/clova-studio/index.ts | 72 +- open-sse/executors/clova-studio.ts | 12 + open-sse/executors/index.ts | 1 + open-sse/handlers/embeddingStructuredInput.ts | 52 +- open-sse/handlers/embeddings.ts | 1090 +++++++++-------- open-sse/translator/bootstrap.ts | 2 + open-sse/translator/formats.ts | 2 + .../translator/request/openai-to-clova.ts | 375 ++++++ .../translator/response/clova-to-openai.ts | 354 ++++++ scripts/check/check-known-symbols.ts | 3 + .../constants/providers/apikey/regional.ts | 2 +- src/shared/constants/visionModels.ts | 5 + tests/snapshots/executors/executor-map.json | 7 +- tests/snapshots/provider/translate-path.json | 6 +- tests/unit/embedding-clova-v2.test.ts | 258 ++++ tests/unit/translator-clova-v3.test.ts | 825 +++++++++++++ 17 files changed, 2638 insertions(+), 556 deletions(-) create mode 100644 open-sse/executors/clova-studio.ts create mode 100644 open-sse/translator/request/openai-to-clova.ts create mode 100644 open-sse/translator/response/clova-to-openai.ts create mode 100644 tests/unit/embedding-clova-v2.test.ts create mode 100644 tests/unit/translator-clova-v3.test.ts diff --git a/open-sse/config/embeddingRegistry.ts b/open-sse/config/embeddingRegistry.ts index c4ab2dc2fd..562d2f63c8 100644 --- a/open-sse/config/embeddingRegistry.ts +++ b/open-sse/config/embeddingRegistry.ts @@ -10,6 +10,7 @@ export type EmbeddingModality = "text" | "image" | "audio" | "video" | "document"; export type StructuredEmbeddingProtocol = "jina-v1" | "gemini-embed-content"; +export type SingleTextEmbeddingProtocol = "clova-v2"; export interface EmbeddingModel { id: string; @@ -34,6 +35,13 @@ export interface EmbeddingProvider { models: EmbeddingModel[]; /** Provider-native serializer required for canonical structured input. */ structuredInputProtocol?: StructuredEmbeddingProtocol; + /** + * Set when the endpoint embeds exactly ONE text per request (`{"text": …}` → + * one vector) instead of accepting OpenAI's `input` array. A batched + * `/v1/embeddings` call is then fanned out into N sequential upstream calls and + * merged back into a single OpenAI list response. + */ + singleTextProtocol?: SingleTextEmbeddingProtocol; } export interface EmbeddingProviderNodeRow { @@ -297,6 +305,18 @@ export const EMBEDDING_PROVIDERS: Record = { ], }, + // Naver CLOVA Studio — embedding v2. The endpoint takes a single `{"text": …}` + // body and returns `{status, result:{embedding:[…1024 floats], inputTokens}}`, + // with no batch array and no `usage` object, hence `singleTextProtocol`. + "clova-studio": { + id: "clova-studio", + baseUrl: "https://clovastudio.stream.ntruss.com/v1/api-tools/embedding/v2", + authType: "apikey", + authHeader: "bearer", + singleTextProtocol: "clova-v2", + models: [{ id: "clova-embedding-v2", name: "CLOVA Embedding v2", dimensions: 1024 }], + }, + "jina-ai": { id: "jina-ai", structuredInputProtocol: "jina-v1", @@ -471,6 +491,62 @@ export function getEmbeddingProvider(providerId: string): EmbeddingProvider | nu return EMBEDDING_PROVIDERS[resolveEmbeddingProviderId(providerId)] || null; } +function findDynamicEmbeddingProvider( + modelStr: string, + dynamicProviders: EmbeddingProvider[] | undefined +): { provider: string; model: string } | null { + const match = dynamicProviders?.find((provider) => modelStr.startsWith(`${provider.id}/`)); + return match ? { provider: match.id, model: modelStr.slice(match.id.length + 1) } : null; +} + +function parsePrefixedEmbeddingModel( + modelStr: string, + slashIdx: number, + dynamicProviders: EmbeddingProvider[] | undefined +): { provider: string; model: string } { + const rawProvider = modelStr.slice(0, slashIdx); + const dynamicExact = dynamicProviders?.find((provider) => provider.id === rawProvider); + if (dynamicExact) { + return { provider: rawProvider, model: modelStr.slice(slashIdx + 1) }; + } + + const resolvedProvider = resolveEmbeddingProviderId(rawProvider); + if (EMBEDDING_PROVIDERS[resolvedProvider]) { + return { + provider: resolvedProvider, + model: normalizeProviderScopedModelId(resolvedProvider, modelStr.slice(slashIdx + 1)), + }; + } + + const hardcodedProvider = Object.keys(EMBEDDING_PROVIDERS).find((providerId) => + modelStr.startsWith(`${providerId}/`) + ); + if (hardcodedProvider) { + return { + provider: hardcodedProvider, + model: normalizeProviderScopedModelId( + hardcodedProvider, + modelStr.slice(hardcodedProvider.length + 1) + ), + }; + } + + return ( + findDynamicEmbeddingProvider(modelStr, dynamicProviders) ?? { + provider: rawProvider, + model: modelStr.slice(slashIdx + 1), + } + ); +} + +function findEmbeddingModelProvider(modelStr: string): string | null { + return ( + Object.entries(EMBEDDING_PROVIDERS).find(([, config]) => + config.models.some((model) => model.id === modelStr) + )?.[0] ?? null + ); +} + /** * Derive an OpenAI-compatible embeddings config for a chat provider that has NO * curated EMBEDDING_PROVIDERS entry. Works for any registry provider whose base @@ -517,59 +593,11 @@ export function parseEmbeddingModel( // Check for "provider/model" format const slashIdx = modelStr.indexOf("/"); if (slashIdx > 0) { - const rawProvider = modelStr.slice(0, slashIdx); - - // A configured provider_node whose prefix exactly equals the requested - // provider segment always wins — even when that segment is also an alias - // of a curated provider (a local node must not be hijacked by a registry - // alias). Same exact-match precedence documented for - // EMBEDDING_MODEL_ALIASES above. - const dynamicExact = - dynamicProviders && dynamicProviders.find((dp) => dp.id === rawProvider); - if (dynamicExact) { - return { provider: rawProvider, model: modelStr.slice(slashIdx + 1) }; - } - - const resolvedProvider = resolveEmbeddingProviderId(rawProvider); - - if (EMBEDDING_PROVIDERS[resolvedProvider]) { - return { - provider: resolvedProvider, - model: normalizeProviderScopedModelId(resolvedProvider, modelStr.slice(slashIdx + 1)), - }; - } - - // Phase 1: Try each hardcoded provider prefix - for (const [providerId] of Object.entries(EMBEDDING_PROVIDERS)) { - if (modelStr.startsWith(providerId + "/")) { - return { - provider: providerId, - model: normalizeProviderScopedModelId(providerId, modelStr.slice(providerId.length + 1)), - }; - } - } - // Phase 2: Try dynamic provider_nodes prefix - if (dynamicProviders) { - for (const dp of dynamicProviders) { - if (modelStr.startsWith(dp.id + "/")) { - return { provider: dp.id, model: modelStr.slice(dp.id.length + 1) }; - } - } - } - // Phase 3: Fallback — first segment is provider - const provider = modelStr.slice(0, slashIdx); - const model = modelStr.slice(slashIdx + 1); - return { provider, model }; + return parsePrefixedEmbeddingModel(modelStr, slashIdx, dynamicProviders); } // No provider prefix — search hardcoded providers for the model - for (const [providerId, config] of Object.entries(EMBEDDING_PROVIDERS)) { - if (config.models.some((m) => m.id === modelStr)) { - return { provider: providerId, model: modelStr }; - } - } - - return { provider: null, model: modelStr }; + return { provider: findEmbeddingModelProvider(modelStr), model: modelStr }; } /** diff --git a/open-sse/config/providers/registry/clova-studio/index.ts b/open-sse/config/providers/registry/clova-studio/index.ts index 8344ebe911..dd371a0336 100644 --- a/open-sse/config/providers/registry/clova-studio/index.ts +++ b/open-sse/config/providers/registry/clova-studio/index.ts @@ -1,17 +1,75 @@ import type { RegistryEntry } from "../../shared.ts"; +/** + * Naver CLOVA Studio — Chat Completions **v3** (native API). + * + * Previously this entry pointed at Naver's OpenAI-compatibility shim + * (`/v1/openai/chat/completions`), which meant `format: "openai"` and a + * pass-through `DefaultExecutor`. The v3 API is Naver's own wire format, so the + * entry now uses `format: "clova"` and the translator pair + * (`openai-to-clova` / `clova-to-openai`). + * + * v3 moves the model into the URL path (`/v3/chat-completions/{modelName}`), uses + * camelCase sampling params, and returns a `{status, result}` envelope instead of + * an OpenAI `choices[]` body — see the translators for the exact mapping. + * + * All three v3 models are live-verified against the real API (2026-09-01): + * + * | Model | Surface | Notes | + * | ------------- | -------- | -------------------------------------------------------- | + * | HCX-007 | thinking | rejects `maxTokens` (use `maxCompletionTokens`); no vision | + * | HCX-005 | text+img | vision via public URL **or** inline base64 data URI | + * | HCX-DASH-002 | text | lightweight, text only | + * + * Docs: https://api.ncloud-docs.com/docs/clovastudio-chatcompletionsv3 + * https://api.ncloud-docs.com/docs/clovastudio-chatcompletionsv3-thinking + */ export const clova_studioProvider: RegistryEntry = { id: "clova-studio", alias: "clova", - format: "openai", - executor: "default", - baseUrl: "https://clovastudio.stream.ntruss.com/v1/openai/chat/completions", + format: "clova", + executor: "clova-studio", + baseUrl: "https://clovastudio.stream.ntruss.com/v3/chat-completions", authType: "apikey", authHeader: "bearer", + /** + * The v3 API does answer non-streaming requests (`Accept: application/json`), + * but only the streaming surface is expressed in the translator: CLOVA's SSE + * frames carry incremental `token` events plus a terminal `result` event that + * repeats the full text. Forcing the upstream stream lets OmniRoute consume + * that single, well-tested path and accumulate it into a JSON body for + * non-streaming clients, instead of maintaining a second parser for the + * `{status, result}` envelope. + */ + forceStream: true, models: [ - // HCX-007 stays first so it remains the provider default (deep-reasoning - // flagship); HCX-005 is the multimodal option. - { id: "HCX-007", name: "HCX-007" }, - { id: "HCX-005", name: "HCX-005" }, + { + // Reasoning flagship. Input+output ≤ 128000 tokens; the output cap counts + // thinking tokens too, so `maxCompletionTokens` may be up to 32768. + id: "HCX-007", + name: "HCX-007", + contextLength: 128000, + maxOutputTokens: 32768, + supportsReasoning: true, + }, + { + // HyperCLOVA X vision model. Input+output ≤ 128000 tokens, output ≤ 4096, + // up to 5 images per request (1 per turn). Accepts a public URL or an + // inline base64 data URI — the data URI must keep its + // `data:;base64,` prefix inside `dataUri.data` or the request is + // rejected with `40001 Invalid parameter`. + id: "HCX-005", + name: "HCX-005", + contextLength: 128000, + maxOutputTokens: 4096, + supportsVision: true, + }, + { + // Lightweight model. Input+output ≤ 32000 tokens, output ≤ 4096, text only. + id: "HCX-DASH-002", + name: "HCX-DASH-002", + contextLength: 32000, + maxOutputTokens: 4096, + }, ], }; diff --git a/open-sse/executors/clova-studio.ts b/open-sse/executors/clova-studio.ts new file mode 100644 index 0000000000..28997a78b9 --- /dev/null +++ b/open-sse/executors/clova-studio.ts @@ -0,0 +1,12 @@ +import { DefaultExecutor } from "./default.ts"; + +/** CLOVA Chat Completions v3 places the selected model in the URL path. */ +export class ClovaStudioExecutor extends DefaultExecutor { + constructor() { + super("clova-studio"); + } + + buildUrl(model: string): string { + return `${this.config.baseUrl}/${encodeURIComponent(model)}`; + } +} diff --git a/open-sse/executors/index.ts b/open-sse/executors/index.ts index d1c3b9ced0..fed0a651c1 100644 --- a/open-sse/executors/index.ts +++ b/open-sse/executors/index.ts @@ -180,6 +180,7 @@ const lazyExecutors: Record Promise> = { xai: () => import("./xai.ts").then((m) => new m.XaiExecutor()), "xai-oauth": () => import("./xai.ts").then((m) => new m.XaiExecutor("xai-oauth")), xao: () => import("./xai.ts").then((m) => new m.XaiExecutor("xai-oauth")), + "clova-studio": () => import("./clova-studio.ts").then((m) => new m.ClovaStudioExecutor()), "conol-web": () => import("./conol-web.ts").then((m) => new m.ConolWebExecutor()), cnl: () => import("./conol-web.ts").then((m) => new m.ConolWebExecutor()), // Alias }; diff --git a/open-sse/handlers/embeddingStructuredInput.ts b/open-sse/handlers/embeddingStructuredInput.ts index 1183c9e5a3..1cd7571a38 100644 --- a/open-sse/handlers/embeddingStructuredInput.ts +++ b/open-sse/handlers/embeddingStructuredInput.ts @@ -128,10 +128,7 @@ export async function prepareJinaMixedEmbeddingInput( continue; } if (isCanonicalEmbeddingItem(item)) { - const [translated] = await prepareJinaInput( - [item as EmbeddingMultimodalItem], - fetchMedia - ); + const [translated] = await prepareJinaInput([item as EmbeddingMultimodalItem], fetchMedia); out.push(translated); continue; } @@ -163,7 +160,9 @@ function embeddingValues(entry: unknown): unknown[] { return Array.isArray(values) ? values : []; } -function normalizeGeminiEmbedContentResponse(data: Record): Record { +function normalizeGeminiEmbedContentResponse( + data: Record +): Record { return { object: "list", data: [{ object: "embedding", embedding: embeddingValues(data.embedding), index: 0 }], @@ -263,10 +262,7 @@ async function itemToGeminiContent( return { parts: [await jinaDocToGeminiPart(item, fetchMedia)] }; } if (isCanonicalEmbeddingItem(item)) { - const [part] = await prepareGeminiParts( - [item as EmbeddingMultimodalItem], - fetchMedia - ); + const [part] = await prepareGeminiParts([item as EmbeddingMultimodalItem], fetchMedia); return { parts: [part] }; } throw new Error("Unsupported Gemini embedding input item"); @@ -346,3 +342,41 @@ export async function prepareStructuredEmbeddingRequest( } throw new Error(`Provider ${provider.id} has no structured embedding input translator`); } + +/** + * Normalize a single-text embedding endpoint's response into OpenAI's + * `/v1/embeddings` list shape. + * + * CLOVA Studio's embedding v2 answers: + * + * ``` + * {"status":{"code":"20000","message":"OK"}, + * "result":{"embedding":[…1024 floats],"inputTokens":4}} + * ``` + * + * There is no `data[]` and no `usage` object, so both are synthesized. `index` is + * left at 0 here — the batching loop in `embeddings.ts` rewrites it to the + * caller's position before the response is returned. + * + * A non-20000 status or malformed success envelope throws so an HTTP-200 error + * envelope can never be exposed as an empty successful embedding response. + */ +export function normalizeClovaEmbeddingV2Response( + rawData: Record +): Record { + const statusCode = (rawData?.status as { code?: unknown } | undefined)?.code; + if (String(statusCode) !== "20000") { + throw new Error("CLOVA Studio embedding v2 returned an unsuccessful status"); + } + + const result = (rawData?.result ?? {}) as Record; + if (!Array.isArray(result.embedding)) { + throw new Error("CLOVA Studio embedding v2 response is missing an embedding vector"); + } + + const inputTokens = Number(result.inputTokens) || 0; + return { + data: [{ object: "embedding", index: 0, embedding: result.embedding }], + usage: { prompt_tokens: inputTokens, total_tokens: inputTokens }, + }; +} diff --git a/open-sse/handlers/embeddings.ts b/open-sse/handlers/embeddings.ts index 7cf322610d..c60842b09f 100644 --- a/open-sse/handlers/embeddings.ts +++ b/open-sse/handlers/embeddings.ts @@ -1,16 +1,8 @@ /** * Embedding Handler * - * Handles POST /v1/embeddings requests. - * Proxies to upstream embedding providers using OpenAI-compatible format. - * - * Request format (OpenAI-compatible): - * { - * "model": "nebius/Qwen/Qwen3-Embedding-8B", - * "input": "text" | ["text1", "text2"], - * "dimensions": 4096, // optional - * "encoding_format": "float" // optional - * } + * Handles POST /v1/embeddings requests and normalizes provider responses to the + * OpenAI embedding shape. */ import { @@ -32,6 +24,7 @@ import { stripTrailingSlashes } from "../utils/urlSanitize.ts"; import { fetchRemoteImage } from "@/shared/network/remoteImageFetch"; import { hasStructuredEmbeddingInput, + normalizeClovaEmbeddingV2Response, prepareJinaMixedEmbeddingInput, prepareStructuredEmbeddingRequest, } from "./embeddingStructuredInput.ts"; @@ -53,17 +46,82 @@ interface ClientRawRequest { headers: Record; } -/** - * Flatten a single embedding item's vector to the OpenAI-spec `number[]` shape. - * - * Some OpenAI-compatible embedding backends — notably a llama.cpp - * `llama-server --embedding --pooling ...` instance — return each vector wrapped in one - * extra array level: `[[...floats]]` instead of `[...floats]` for a single input. That - * extra level is silently spec-breaking, since a standard OpenAI-SDK consumer reading - * `response.data[i].embedding` gets a length-1 array holding the real vector instead of - * the vector itself. Unwrap only that single redundant level; vectors that are already - * flat (or genuinely multi-row) are left untouched. See issue #9089. - */ +interface EmbeddingCredentials { + apiKey?: string | null; + accessToken?: string | null; + providerSpecificData?: Record | null; +} + +interface EmbeddingLog { + info: (...args: unknown[]) => void; + error: (...args: unknown[]) => void; +} + +interface HandleEmbeddingParams { + body: Record; + credentials: EmbeddingCredentials | null; + log?: EmbeddingLog; + resolvedProvider?: EmbeddingProvider | null; + resolvedModel?: string | null; + clientRawRequest?: ClientRawRequest | null; + apiKeyId?: string | null; + apiKeyName?: string | null; + connectionId?: string | null; +} + +interface EmbeddingFailure { + success: false; + status: number; + error: string; + headers?: Headers; + data?: never; +} + +interface EmbeddingSuccess { + success: true; + data: Record; + headers: Headers; + status?: never; + error?: never; +} + +type EmbeddingResult = EmbeddingSuccess | EmbeddingFailure; + +interface ResolvedEmbedding { + provider: string | null; + model: string | null; + providerConfig: EmbeddingProvider | null; +} + +type RequestLogger = Awaited>; +type ProviderResponseNormalizer = + ((data: Record) => Record) | null; + +interface EmbeddingRuntime extends HandleEmbeddingParams { + provider: string; + model: string | null; + providerConfig: EmbeddingProvider; + startTime: number; + detailedLoggingEnabled: boolean; + reqLogger: RequestLogger; + logRequestBody: Record; +} + +interface PreparedEmbeddingRequest { + upstreamBody: Record; + upstreamUrl: string; + headers: Record; + normalizeProviderResponse: ProviderResponseNormalizer; +} + +interface ParsedEmbeddingResponse { + data?: unknown[] | unknown; + usage?: { prompt_tokens?: number; total_tokens?: number }; +} + +const KNOWN_EMBEDDING_FIELDS = new Set(["model", "input", "dimensions", "encoding_format"]); + +/** Unwrap one redundant row around an otherwise flat vector. */ function flattenSingleRowEmbedding(item: unknown): void { if (!item || typeof item !== "object" || !("embedding" in item)) return; const record = item as { embedding: unknown }; @@ -78,103 +136,81 @@ function flattenSingleRowEmbedding(item: unknown): void { } } -/** - * Handle embedding request. - * Supports both hardcoded cloud providers and dynamic local provider_nodes. - * When resolvedProvider is passed, uses it directly (injection pattern from route handler). - * Falls back to hardcoded registry lookup for backward compatibility. - */ -export async function handleEmbedding({ - body, - credentials, - log, - resolvedProvider = null, - resolvedModel = null, - clientRawRequest = null, - apiKeyId = null, - apiKeyName = null, - connectionId = null, -}: { - body: Record; - credentials: { - apiKey?: string | null; - accessToken?: string | null; - providerSpecificData?: Record | null; - } | null; - log?: { info: (...args: unknown[]) => void; error: (...args: unknown[]) => void }; - resolvedProvider?: EmbeddingProvider | null; - resolvedModel?: string | null; - clientRawRequest?: ClientRawRequest | null; - apiKeyId?: string | null; - apiKeyName?: string | null; - connectionId?: string | null; -}) { - // Use pre-resolved provider/model from route handler if available (supports dynamic provider_nodes). - let provider: string | null; - let model: string | null; - let providerConfig: EmbeddingProvider | null; +function failure(status: number, error: string, headers?: Headers): EmbeddingFailure { + return { success: false, status, error, ...(headers ? { headers } : {}) }; +} - if (resolvedProvider) { - provider = resolvedProvider.id; - model = resolvedModel; - providerConfig = resolvedProvider; - } else { - const parsed = parseEmbeddingModel(body.model as string); - provider = parsed.provider; - model = parsed.model; - providerConfig = provider ? getEmbeddingProvider(provider) : null; +function resolveEmbedding(params: HandleEmbeddingParams): ResolvedEmbedding { + if (params.resolvedProvider) { + return { + provider: params.resolvedProvider.id, + model: params.resolvedModel ?? null, + providerConfig: params.resolvedProvider, + }; } + const parsed = parseEmbeddingModel(params.body.model as string); + return { + provider: parsed.provider, + model: parsed.model, + providerConfig: parsed.provider ? getEmbeddingProvider(parsed.provider) : null, + }; +} - const startTime = Date.now(); - - // Set up request logger for pipeline artifact capture +async function createEmbeddingRuntime( + params: HandleEmbeddingParams, + resolved: ResolvedEmbedding +): Promise { const detailedLoggingEnabled = await isDetailedLoggingEnabled(); - const captureStreamChunks = getCallLogPipelineCaptureStreamChunks(); const reqLogger = await createRequestLogger( - provider || "openai", + resolved.provider || "openai", "openai", - body.model as string, + params.body.model as string, { enabled: detailedLoggingEnabled, - captureStreamChunks, - connectionId: connectionId || undefined, - model: model || (body.model as string), - provider: provider || undefined, + captureStreamChunks: getCallLogPipelineCaptureStreamChunks(), + connectionId: params.connectionId || undefined, + model: resolved.model || (params.body.model as string), + provider: resolved.provider || undefined, } ); - // Log client raw request - if (clientRawRequest) { + if (params.clientRawRequest) { reqLogger.logClientRawRequest( - clientRawRequest.endpoint, - clientRawRequest.body, - clientRawRequest.headers + params.clientRawRequest.endpoint, + params.clientRawRequest.body, + params.clientRawRequest.headers ); } + if (!resolved.provider) { + return failure( + 400, + `Invalid embedding model: ${params.body.model}. Use format: provider/model` + ); + } + if (!resolved.providerConfig) { + return failure(400, `Unknown embedding provider: ${resolved.provider}`); + } - // Summarized request body for call log (avoid storing large embedding input arrays) - const logRequestBody = { - model: body.model, - input_count: Array.isArray(body.input) ? body.input.length : 1, - dimensions: body.dimensions || undefined, + return { + ...params, + provider: resolved.provider, + model: resolved.model, + providerConfig: resolved.providerConfig, + startTime: Date.now(), + detailedLoggingEnabled, + reqLogger, + logRequestBody: { + model: params.body.model, + input_count: Array.isArray(params.body.input) ? params.body.input.length : 1, + dimensions: params.body.dimensions || undefined, + }, }; +} - if (!provider) { - return { - success: false, - status: 400, - error: `Invalid embedding model: ${body.model}. Use format: provider/model`, - }; - } - - if (!providerConfig) { - return { - success: false, - status: 400, - error: `Unknown embedding provider: ${provider}`, - }; - } - +function collectRequestedModalities(body: Record): { + structuredItems: Array<{ type: EmbeddingModality }>; + nativeModalities: EmbeddingModality[]; +} { const structuredItems = Array.isArray(body.input) ? body.input.filter( (item): item is { type: EmbeddingModality } => @@ -184,409 +220,493 @@ export async function handleEmbedding({ const nativeModalities = [ ...(isJinaNativeEmbeddingInput(body.input) ? collectJinaNativeModalities(body.input) : []), ...(isGeminiNativeEmbeddingInput(body.input) ? collectGeminiNativeModalities(body.input) : []), - ].filter((modality) => modality !== "text"); - if (structuredItems.length > 0 || nativeModalities.length > 0) { - const supportedModalities = getEmbeddingModelModalities(providerConfig, model); - if (!supportedModalities) { - return { - success: false, - status: 400, - error: `Embedding model ${body.model} does not advertise structured embedding input support`, - }; - } - const unsupportedCanonical = structuredItems.find( - (item) => !supportedModalities.includes(item.type) + ].filter((modality): modality is EmbeddingModality => modality !== "text"); + return { structuredItems, nativeModalities }; +} + +function validateRequestedModalities(runtime: EmbeddingRuntime): EmbeddingFailure | null { + const { structuredItems, nativeModalities } = collectRequestedModalities(runtime.body); + if (structuredItems.length === 0 && nativeModalities.length === 0) return null; + + const supported = getEmbeddingModelModalities(runtime.providerConfig, runtime.model); + if (!supported) { + return failure( + 400, + `Embedding model ${runtime.body.model} does not advertise structured embedding input support` ); - if (unsupportedCanonical) { - return { - success: false, - status: 400, - error: `Embedding model ${body.model} does not support ${unsupportedCanonical.type} input`, - }; - } - const unsupportedNative = nativeModalities.find( - (modality) => !supportedModalities.includes(modality) - ); - if (unsupportedNative) { - return { - success: false, - status: 400, - error: `Embedding model ${body.model} does not support ${unsupportedNative} input`, - }; - } } + const unsupportedCanonical = structuredItems.find((item) => !supported.includes(item.type)); + if (unsupportedCanonical) { + return failure( + 400, + `Embedding model ${runtime.body.model} does not support ${unsupportedCanonical.type} input` + ); + } + const unsupportedNative = nativeModalities.find((modality) => !supported.includes(modality)); + return unsupportedNative + ? failure( + 400, + `Embedding model ${runtime.body.model} does not support ${unsupportedNative} input` + ) + : null; +} - // Build upstream request — start with standard fields, then forward extra fields - // the client sent (e.g. input_type, user, truncate for NVIDIA NIM asymmetric models). - const KNOWN_FIELDS = new Set(["model", "input", "dimensions", "encoding_format"]); - - let upstreamBody: Record = { - model: model, - input: body.input, +function buildUpstreamBody(runtime: EmbeddingRuntime): Record { + const upstreamBody: Record = { + model: runtime.model, + input: runtime.body.input, }; - - if (body.dimensions !== undefined) upstreamBody.dimensions = body.dimensions; - if (body.encoding_format !== undefined) upstreamBody.encoding_format = body.encoding_format; - - for (const [key, value] of Object.entries(body)) { - if (!KNOWN_FIELDS.has(key) && value !== undefined) { - upstreamBody[key] = value; - } + if (runtime.body.dimensions !== undefined) upstreamBody.dimensions = runtime.body.dimensions; + if (runtime.body.encoding_format !== undefined) { + upstreamBody.encoding_format = runtime.body.encoding_format; + } + for (const [key, value] of Object.entries(runtime.body)) { + if (!KNOWN_EMBEDDING_FIELDS.has(key) && value !== undefined) upstreamBody[key] = value; } - // Gemini embedding models (gemini-embedding-001 / -2-preview / text-embedding-004) - // default to 3072-dim vectors. Clients targeting pgvector-style schemas typically - // request a smaller size (e.g. 1536) via OpenAI's `dimensions` field, but Google's - // OpenAI-compatibility shim at /v1beta/openai/embeddings does not document the - // `dimensions` → `outputDimensionality` translation. Mirror the request value into - // the Gemini-native `outputDimensionality` field so the upstream actually returns - // the requested vector size. Ported from upstream decolua/9router#1366. - if (provider === "gemini" && upstreamBody.outputDimensionality === undefined) { - const outputDimensionality = Number(body.dimensions); + if (runtime.provider === "gemini" && upstreamBody.outputDimensionality === undefined) { + const outputDimensionality = Number(runtime.body.dimensions); if (Number.isFinite(outputDimensionality) && outputDimensionality > 0) { upstreamBody.outputDimensionality = outputDimensionality; } } - - // Inject model-level default params (e.g. NVIDIA NIM asymmetric models require - // `input_type`) only for keys the client did not already supply, so a - // client-sent value is never overwritten. Symmetric models carry no defaults - // and are unaffected. See issue #1378. - const defaultParams = getEmbeddingModelDefaultParams(providerConfig, model); - if (defaultParams) { - for (const [key, value] of Object.entries(defaultParams)) { - if (upstreamBody[key] === undefined) { - upstreamBody[key] = value; - } - } + const defaultParams = getEmbeddingModelDefaultParams(runtime.providerConfig, runtime.model); + for (const [key, value] of Object.entries(defaultParams ?? {})) { + if (upstreamBody[key] === undefined) upstreamBody[key] = value; } + return upstreamBody; +} - let upstreamUrl = providerConfig.baseUrl; - if (provider === "ollama-local" || provider === "lmstudio") { - // Keyless local servers (#2824 ollama-local, #11233 lmstudio): honor the - // configured connection's baseUrl when one was hydrated, and fall back to - // the static localhost registry default otherwise. - const configuredBaseUrl = credentials?.providerSpecificData?.baseUrl; - const rawBaseUrl = - typeof configuredBaseUrl === "string" && configuredBaseUrl.trim().length > 0 - ? configuredBaseUrl - : providerConfig.baseUrl; - // Use the shared O(n) helper instead of `/\/+$/` — that regex is - // vulnerable to polynomial backtracking on adversarial input - // (CodeQL js/polynomial-redos) since baseUrl is operator-configured - // per-connection data. See open-sse/utils/urlSanitize.ts. - const normalizedBaseUrl = stripTrailingSlashes(rawBaseUrl.trim()); - const localServerHost = normalizedBaseUrl - .replace(/\/v1\/(?:chat\/completions|embeddings)$/i, "") - .replace(/\/api\/chat$/i, "") - .replace(/\/v1$/i, ""); - upstreamUrl = `${localServerHost}/v1/embeddings`; - } - let normalizeProviderResponse: - ((data: Record) => Record) | null = null; +function resolveLocalEmbeddingUrl(runtime: EmbeddingRuntime): string { + const configuredBaseUrl = runtime.credentials?.providerSpecificData?.baseUrl; + const rawBaseUrl = + typeof configuredBaseUrl === "string" && configuredBaseUrl.trim() + ? configuredBaseUrl + : runtime.providerConfig.baseUrl; + const localServerHost = stripTrailingSlashes(rawBaseUrl.trim()) + .replace(/\/v1\/(?:chat\/completions|embeddings)$/i, "") + .replace(/\/api\/chat$/i, "") + .replace(/\/v1$/i, ""); + return `${localServerHost}/v1/embeddings`; +} - // Build headers - const headers: Record = { - "Content-Type": "application/json", - }; +function resolveUpstreamUrl(runtime: EmbeddingRuntime): string { + return runtime.provider === "ollama-local" || runtime.provider === "lmstudio" + ? resolveLocalEmbeddingUrl(runtime) + : runtime.providerConfig.baseUrl; +} - // Skip credential injection for local providers (authType: "none") +function buildAuth( + runtime: EmbeddingRuntime +): { headers: Record; token: string | null } | EmbeddingFailure { + const headers: Record = { "Content-Type": "application/json" }; const token = - providerConfig.authType === "none" ? null : credentials?.apiKey || credentials?.accessToken; - if (token) { - if (providerConfig.authHeader === "bearer") { - headers["Authorization"] = `Bearer ${token}`; - } else if (providerConfig.authHeader === "x-api-key") { - headers["x-api-key"] = token; - } - } else if (providerConfig.authType !== "none") { - return { - success: false, - status: 401, - error: `No valid authentication token for provider ${provider}. Check provider credentials.`, - }; - } - - // Jina v5 Omni native docs ({ text }, { image: url|base64 }, { content: [...] }) - // must reach api.jina.ai unchanged. Do not fetch those image URLs or collapse - // to string[]. Canonical { type, source } items still go through the translator. - const jinaNative = isJinaNativeEmbeddingInput(body.input); - const geminiNative = isGeminiNativeEmbeddingInput(body.input); - const canonicalStructured = hasStructuredEmbeddingInput(body.input); - const passThroughJinaNative = - providerConfig.structuredInputProtocol === "jina-v1" && jinaNative && !canonicalStructured; - // gemini-embedding-2 aggregates a string[] on Google's OpenAI shim into one - // vector. Always use embedContent / batchEmbedContents so N input items - // become N embeddings. Native multimodal parts take the same path. - const useGeminiNativeTransport = - providerConfig.structuredInputProtocol === "gemini-embed-content" && - (isGeminiEmbedding2Family(model) || canonicalStructured || geminiNative || jinaNative); - - if (providerConfig.structuredInputProtocol === "jina-v1" && jinaNative && canonicalStructured) { - try { - const mixed = Array.isArray(body.input) ? body.input : [body.input]; - upstreamBody.input = await prepareJinaMixedEmbeddingInput(mixed, async (url) => { - const result = await fetchRemoteImage(url, { - guard: "public-only", - maxBytes: MAX_EMBEDDING_INLINE_ITEM_BYTES, - pinDns: true, - }); - return { buffer: result.buffer, contentType: result.contentType || null }; - }); - } catch (error) { - return { success: false, status: 400, error: sanitizeErrorMessage(error) }; - } - } else if (useGeminiNativeTransport || (!passThroughJinaNative && canonicalStructured)) { - if (!model) { - return { - success: false, - status: 400, - error: `Invalid embedding model: ${body.model}. Use format: provider/model`, - }; - } - try { - const prepared = await prepareStructuredEmbeddingRequest( - providerConfig, - model, - body, - token ?? "", - { - fetchMedia: async (url) => { - const result = await fetchRemoteImage(url, { - guard: "public-only", - maxBytes: MAX_EMBEDDING_INLINE_ITEM_BYTES, - pinDns: true, - }); - return { buffer: result.buffer, contentType: result.contentType || null }; - }, - } - ); - upstreamBody = prepared.body; - upstreamUrl = prepared.url; - normalizeProviderResponse = prepared.normalizeResponse ?? null; - if (prepared.authHeader) { - delete headers.Authorization; - delete headers["x-api-key"]; - headers[prepared.authHeader.name] = prepared.authHeader.value; - } - } catch (error) { - return { success: false, status: 400, error: sanitizeErrorMessage(error) }; - } - } - - if (log) { - log.info( - "EMBED", - `${provider}/${model} | input: ${Array.isArray(body.input) ? body.input.length + " items" : "1 item"}` + runtime.providerConfig.authType === "none" + ? null + : runtime.credentials?.apiKey || runtime.credentials?.accessToken || null; + if (!token && runtime.providerConfig.authType !== "none") { + return failure( + 401, + `No valid authentication token for provider ${runtime.provider}. Check provider credentials.` ); } + if (token && runtime.providerConfig.authHeader === "bearer") { + headers.Authorization = `Bearer ${token}`; + } else if (token && runtime.providerConfig.authHeader === "x-api-key") { + headers["x-api-key"] = token; + } + return { headers, token }; +} - try { - // Quota share enforcement (fail-open: errors allow the request through) - if (apiKeyId && connectionId && provider) { - try { - const { enforceQuotaShare } = await import("@/lib/quota/enforce"); - const quotaDecision = await enforceQuotaShare({ - apiKeyId, - connectionId, - provider, - // Per-(key,model) cap — resolved embedding model id (same scope used in logs/routing). - model: model || undefined, - }); - if (quotaDecision.kind === "block") { - return { - success: false, - status: quotaDecision.httpStatus ?? 429, - error: quotaDecision.reason || "Quota share limit reached", - }; - } - } catch { - // fail-open per B16 - } - } +async function fetchEmbeddingMedia( + url: string +): Promise<{ buffer: Buffer; contentType: string | null }> { + const result = await fetchRemoteImage(url, { + guard: "public-only", + maxBytes: MAX_EMBEDDING_INLINE_ITEM_BYTES, + pinDns: true, + }); + return { buffer: result.buffer, contentType: result.contentType || null }; +} - // Log provider request - reqLogger.logTargetRequest(upstreamUrl, headers, upstreamBody); +async function prepareMixedJinaInput( + runtime: EmbeddingRuntime, + prepared: PreparedEmbeddingRequest +): Promise { + const mixed = Array.isArray(runtime.body.input) ? runtime.body.input : [runtime.body.input]; + prepared.upstreamBody.input = await prepareJinaMixedEmbeddingInput(mixed, fetchEmbeddingMedia); +} - const response = await fetch(upstreamUrl, { - method: "POST", - headers, - body: JSON.stringify(upstreamBody), - }); - - if (!response.ok) { - const errorText = await response.text(); - if (log) { - log.error("EMBED", `${provider} error ${response.status}: ${errorText.slice(0, 200)}`); - } - - // Log provider response - reqLogger.logProviderResponse(response.status, "", response.headers, errorText.slice(0, 500)); - - // Build client error response - const clientErrorBody = toJsonErrorPayload( - errorText.slice(0, 500), - "Embedding provider error" - ); - reqLogger.logConvertedResponse(clientErrorBody); - - const pipelinePayloads = detailedLoggingEnabled ? reqLogger.getPipelinePayloads() : null; - - // Save error call log for Logger panel - saveCallLog({ - method: "POST", - path: "/v1/embeddings", - status: response.status, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: errorText.slice(0, 500), - requestBody: logRequestBody, - pipelinePayloads, - apiKeyId, - apiKeyName, - connectionId, - }).catch(() => {}); - - // #10347 — persist a connection-level failure marker on a hard upstream failure so - // the dead account is not re-selected and re-hit on the next embed request (chat - // parity). markAccountUnavailable classifies the status via checkFallbackError: a - // payment-required 402 becomes the TERMINAL state credits_exhausted (the terminal - // marker excludes the account from selection until an operator resets it), benign - // 4xx are a no-op, and terminal statuses are never overwritten. honors per-connection - // disableCooling. The write must never break the error response path, so it is - // best-effort. - if (connectionId) { - try { - await markAccountUnavailable(connectionId, response.status, errorText, provider, model); - } catch { - // swallow — the upstream error response takes priority - } - } - - return { - success: false, - status: response.status, - error: errorText, - headers: stripStaleEncodingHeaders(response.headers), - }; - } - - const rawData = (await response.json()) as Record; - const data = (normalizeProviderResponse ? normalizeProviderResponse(rawData) : rawData) as { - data?: unknown[] | unknown; - usage?: { prompt_tokens?: number; total_tokens?: number }; - }; - - // Log provider response - reqLogger.logProviderResponse(response.status, "", response.headers, data); - - // OpenAI-spec compliance (#9089): each item's `embedding` must be a flat number[]. - // Some OpenAI-compatible backends (e.g. a llama.cpp `llama-server --embedding` - // instance) return the vector wrapped in one extra array level — `[[...floats]]` - // instead of `[...floats]` — for a single input, which silently breaks any standard - // OpenAI-SDK consumer doing `response.data[i].embedding`. Flatten that one redundant - // level without touching providers that already return flat vectors. - const responseItems = data.data || data; - if (Array.isArray(responseItems)) { - for (const item of responseItems) { - flattenSingleRowEmbedding(item); - } - } - - // Normalize response to OpenAI format - const normalizedResponse = { - object: "list", - data: data.data || data, - model: `${provider}/${model}`, - usage: data.usage || { prompt_tokens: 0, total_tokens: 0 }, - }; - - // Log client response - reqLogger.logConvertedResponse(normalizedResponse); - - const pipelinePayloads = detailedLoggingEnabled ? reqLogger.getPipelinePayloads() : null; - - // Save success call log for Logger panel - // Embeddings only have input tokens (prompt_tokens + total_tokens), no output/completion tokens - saveCallLog({ - method: "POST", - path: "/v1/embeddings", - status: 200, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - tokens: { - prompt_tokens: data.usage?.prompt_tokens || data.usage?.total_tokens || 0, - completion_tokens: 0, - }, - requestBody: logRequestBody, - responseBody: { - usage: data.usage || null, - object: "list", - data_count: Array.isArray(data.data) ? data.data.length : 0, - }, - pipelinePayloads, - apiKeyId, - apiKeyName, - connectionId, - }).catch(() => {}); - - // Record quota consumption (fire-and-forget, never blocks) - if (apiKeyId && connectionId && provider) { - try { - const { scheduleRecordConsumption } = await import("@/lib/quota/spendRecorder"); - scheduleRecordConsumption({ - apiKeyId, - connectionId, - provider, - // Per-(key,model) cap accounting — same resolved model id used at enforce time. - model: model || undefined, - cost: { - tokens: data.usage?.prompt_tokens || data.usage?.total_tokens || 0, - requests: 1, - }, - }); - } catch { - // fail-open per B29 - } - } - - return { - success: true, - data: normalizedResponse, - headers: stripStaleEncodingHeaders(response.headers), - }; - } catch (err) { - if (log) { - log.error("EMBED", `${provider} fetch error: ${err.message}`); - } - - // Log error - reqLogger.logError(err, upstreamBody); - - const pipelinePayloads = detailedLoggingEnabled ? reqLogger.getPipelinePayloads() : null; - - // Save exception call log for Logger panel - saveCallLog({ - method: "POST", - path: "/v1/embeddings", - status: 502, - model: `${provider}/${model}`, - provider, - duration: Date.now() - startTime, - error: err.message, - requestBody: logRequestBody, - pipelinePayloads, - apiKeyId, - apiKeyName, - connectionId, - }).catch(() => {}); - - return { - success: false, - status: 502, - error: `Embedding provider error: ${sanitizeErrorMessage(err.message)}`, - }; +async function prepareNativeTransport( + runtime: EmbeddingRuntime, + prepared: PreparedEmbeddingRequest, + token: string | null +): Promise { + if (!runtime.model) { + throw new Error(`Invalid embedding model: ${runtime.body.model}. Use format: provider/model`); + } + const native = await prepareStructuredEmbeddingRequest( + runtime.providerConfig, + runtime.model, + runtime.body, + token ?? "", + { fetchMedia: fetchEmbeddingMedia } + ); + prepared.upstreamBody = native.body; + prepared.upstreamUrl = native.url; + prepared.normalizeProviderResponse = native.normalizeResponse ?? null; + if (native.authHeader) { + delete prepared.headers.Authorization; + delete prepared.headers["x-api-key"]; + prepared.headers[native.authHeader.name] = native.authHeader.value; } } + +async function applyStructuredTransport( + runtime: EmbeddingRuntime, + prepared: PreparedEmbeddingRequest, + token: string | null +): Promise { + const jinaNative = isJinaNativeEmbeddingInput(runtime.body.input); + const geminiNative = isGeminiNativeEmbeddingInput(runtime.body.input); + const canonical = hasStructuredEmbeddingInput(runtime.body.input); + const isJinaProtocol = runtime.providerConfig.structuredInputProtocol === "jina-v1"; + const passThroughJina = isJinaProtocol && jinaNative && !canonical; + const useGeminiNative = + runtime.providerConfig.structuredInputProtocol === "gemini-embed-content" && + (isGeminiEmbedding2Family(runtime.model) || canonical || geminiNative || jinaNative); + + if (isJinaProtocol && jinaNative && canonical) { + await prepareMixedJinaInput(runtime, prepared); + } else if (useGeminiNative || (!passThroughJina && canonical)) { + await prepareNativeTransport(runtime, prepared, token); + } +} + +async function prepareEmbeddingRequest( + runtime: EmbeddingRuntime +): Promise { + const auth = buildAuth(runtime); + if ("success" in auth) return auth; + const prepared: PreparedEmbeddingRequest = { + upstreamBody: buildUpstreamBody(runtime), + upstreamUrl: resolveUpstreamUrl(runtime), + headers: auth.headers, + normalizeProviderResponse: null, + }; + try { + await applyStructuredTransport(runtime, prepared, auth.token); + return prepared; + } catch (error) { + return failure(400, sanitizeErrorMessage(error)); + } +} + +async function enforceEmbeddingQuota(runtime: EmbeddingRuntime): Promise { + if (!runtime.apiKeyId || !runtime.connectionId) return null; + try { + const { enforceQuotaShare } = await import("@/lib/quota/enforce"); + const decision = await enforceQuotaShare({ + apiKeyId: runtime.apiKeyId, + connectionId: runtime.connectionId, + provider: runtime.provider, + model: runtime.model || undefined, + }); + return decision.kind === "block" + ? failure(decision.httpStatus ?? 429, decision.reason || "Quota share limit reached") + : null; + } catch { + return null; + } +} + +function resolveSingleTexts(runtime: EmbeddingRuntime): string[] | EmbeddingFailure | null { + if (runtime.providerConfig.singleTextProtocol !== "clova-v2") return null; + const input = Array.isArray(runtime.body.input) ? runtime.body.input : [runtime.body.input]; + if ( + input.length === 0 || + input.some((item) => typeof item !== "string" || item.trim().length === 0) + ) { + return failure(400, "CLOVA Studio embedding v2 accepts non-empty text strings only"); + } + if (runtime.body.encoding_format === "base64") { + return failure(400, "CLOVA Studio embedding v2 supports float encoding only"); + } + if (runtime.body.dimensions !== undefined && Number(runtime.body.dimensions) !== 1024) { + return failure(400, "CLOVA Studio embedding v2 has a fixed dimension of 1024"); + } + return input as string[]; +} + +function appendClovaEmbedding( + parsed: ParsedEmbeddingResponse, + embeddings: Array>, + usage: { prompt_tokens: number; total_tokens: number } +): void { + if (!Array.isArray(parsed.data)) { + throw new Error("CLOVA Studio embedding v2 returned an invalid data list"); + } + for (const item of parsed.data) { + flattenSingleRowEmbedding(item); + if (!item || typeof item !== "object") { + throw new Error("CLOVA Studio embedding v2 returned an invalid embedding item"); + } + (item as { index?: number }).index = embeddings.length; + embeddings.push(item as Record); + } + usage.prompt_tokens += parsed.usage?.prompt_tokens || parsed.usage?.total_tokens || 0; + usage.total_tokens += parsed.usage?.total_tokens || parsed.usage?.prompt_tokens || 0; +} + +async function fetchClovaEmbeddingBatch( + prepared: PreparedEmbeddingRequest, + texts: string[], + reqLogger: RequestLogger +): Promise { + const embeddings: Array> = []; + const usage = { prompt_tokens: 0, total_tokens: 0 }; + let lastHeaders = new Headers(); + for (const text of texts) { + const requestBody = { text }; + reqLogger.logTargetRequest(prepared.upstreamUrl, prepared.headers, requestBody); + const response = await fetch(prepared.upstreamUrl, { + method: "POST", + headers: prepared.headers, + body: JSON.stringify(requestBody), + }); + lastHeaders = response.headers; + if (!response.ok) return response; + const rawData = (await response.json()) as Record; + appendClovaEmbedding(normalizeClovaEmbeddingV2Response(rawData), embeddings, usage); + } + return new Response(JSON.stringify({ data: embeddings, usage }), { + status: 200, + headers: lastHeaders, + }); +} + +async function dispatchEmbeddingRequest( + prepared: PreparedEmbeddingRequest, + singleTexts: string[] | null, + reqLogger: RequestLogger +): Promise { + if (singleTexts) return fetchClovaEmbeddingBatch(prepared, singleTexts, reqLogger); + reqLogger.logTargetRequest(prepared.upstreamUrl, prepared.headers, prepared.upstreamBody); + return fetch(prepared.upstreamUrl, { + method: "POST", + headers: prepared.headers, + body: JSON.stringify(prepared.upstreamBody), + }); +} + +function pipelinePayloads( + runtime: EmbeddingRuntime +): ReturnType | null { + return runtime.detailedLoggingEnabled ? runtime.reqLogger.getPipelinePayloads() : null; +} + +async function handleUpstreamFailure( + runtime: EmbeddingRuntime, + response: Response +): Promise { + const errorText = await response.text(); + runtime.log?.error( + "EMBED", + `${runtime.provider} error ${response.status}: ${errorText.slice(0, 200)}` + ); + runtime.reqLogger.logProviderResponse( + response.status, + "", + response.headers, + errorText.slice(0, 500) + ); + runtime.reqLogger.logConvertedResponse( + toJsonErrorPayload(errorText.slice(0, 500), "Embedding provider error") + ); + saveCallLog({ + method: "POST", + path: "/v1/embeddings", + status: response.status, + model: `${runtime.provider}/${runtime.model}`, + provider: runtime.provider, + duration: Date.now() - runtime.startTime, + error: errorText.slice(0, 500), + requestBody: runtime.logRequestBody, + pipelinePayloads: pipelinePayloads(runtime), + apiKeyId: runtime.apiKeyId, + apiKeyName: runtime.apiKeyName, + connectionId: runtime.connectionId, + }).catch(() => {}); + if (runtime.connectionId) { + try { + await markAccountUnavailable( + runtime.connectionId, + response.status, + errorText, + runtime.provider, + runtime.model + ); + } catch { + // The upstream response has priority over a best-effort cooldown write. + } + } + return failure(response.status, errorText, stripStaleEncodingHeaders(response.headers)); +} + +function normalizeEmbeddingData( + runtime: EmbeddingRuntime, + response: Response, + rawData: Record, + normalizer: ProviderResponseNormalizer +): { data: ParsedEmbeddingResponse; normalizedResponse: Record } { + const data = (normalizer ? normalizer(rawData) : rawData) as ParsedEmbeddingResponse; + runtime.reqLogger.logProviderResponse(response.status, "", response.headers, data); + const responseItems = data.data || data; + if (Array.isArray(responseItems)) responseItems.forEach(flattenSingleRowEmbedding); + return { + data, + normalizedResponse: { + object: "list", + data: data.data || data, + model: `${runtime.provider}/${runtime.model}`, + usage: data.usage || { prompt_tokens: 0, total_tokens: 0 }, + }, + }; +} + +function recordEmbeddingSuccess( + runtime: EmbeddingRuntime, + data: ParsedEmbeddingResponse, + normalizedResponse: Record +): void { + runtime.reqLogger.logConvertedResponse(normalizedResponse); + saveCallLog({ + method: "POST", + path: "/v1/embeddings", + status: 200, + model: `${runtime.provider}/${runtime.model}`, + provider: runtime.provider, + duration: Date.now() - runtime.startTime, + tokens: { + prompt_tokens: data.usage?.prompt_tokens || data.usage?.total_tokens || 0, + completion_tokens: 0, + }, + requestBody: runtime.logRequestBody, + responseBody: { + usage: data.usage || null, + object: "list", + data_count: Array.isArray(data.data) ? data.data.length : 0, + }, + pipelinePayloads: pipelinePayloads(runtime), + apiKeyId: runtime.apiKeyId, + apiKeyName: runtime.apiKeyName, + connectionId: runtime.connectionId, + }).catch(() => {}); +} + +async function recordEmbeddingConsumption( + runtime: EmbeddingRuntime, + data: ParsedEmbeddingResponse, + requestCount: number +): Promise { + if (!runtime.apiKeyId || !runtime.connectionId) return; + try { + const { scheduleRecordConsumption } = await import("@/lib/quota/spendRecorder"); + scheduleRecordConsumption({ + apiKeyId: runtime.apiKeyId, + connectionId: runtime.connectionId, + provider: runtime.provider, + model: runtime.model || undefined, + cost: { + tokens: data.usage?.prompt_tokens || data.usage?.total_tokens || 0, + requests: requestCount, + }, + }); + } catch { + // Quota accounting is fail-open. + } +} + +async function handleUpstreamSuccess( + runtime: EmbeddingRuntime, + prepared: PreparedEmbeddingRequest, + response: Response, + requestCount: number +): Promise { + const rawData = (await response.json()) as Record; + const { data, normalizedResponse } = normalizeEmbeddingData( + runtime, + response, + rawData, + prepared.normalizeProviderResponse + ); + recordEmbeddingSuccess(runtime, data, normalizedResponse); + await recordEmbeddingConsumption(runtime, data, requestCount); + return { + success: true, + data: normalizedResponse, + headers: stripStaleEncodingHeaders(response.headers), + }; +} + +function handleEmbeddingException( + runtime: EmbeddingRuntime, + prepared: PreparedEmbeddingRequest, + error: unknown +): EmbeddingFailure { + const message = error instanceof Error ? error.message : String(error); + runtime.log?.error("EMBED", `${runtime.provider} fetch error: ${message}`); + runtime.reqLogger.logError(error, prepared.upstreamBody); + saveCallLog({ + method: "POST", + path: "/v1/embeddings", + status: 502, + model: `${runtime.provider}/${runtime.model}`, + provider: runtime.provider, + duration: Date.now() - runtime.startTime, + error: message, + requestBody: runtime.logRequestBody, + pipelinePayloads: pipelinePayloads(runtime), + apiKeyId: runtime.apiKeyId, + apiKeyName: runtime.apiKeyName, + connectionId: runtime.connectionId, + }).catch(() => {}); + return failure(502, `Embedding provider error: ${sanitizeErrorMessage(message)}`); +} + +async function executeEmbedding( + runtime: EmbeddingRuntime, + prepared: PreparedEmbeddingRequest +): Promise { + const quotaFailure = await enforceEmbeddingQuota(runtime); + if (quotaFailure) return quotaFailure; + const singleTextsOrFailure = resolveSingleTexts(runtime); + if (singleTextsOrFailure && !Array.isArray(singleTextsOrFailure)) return singleTextsOrFailure; + const singleTexts = Array.isArray(singleTextsOrFailure) ? singleTextsOrFailure : null; + try { + const response = await dispatchEmbeddingRequest(prepared, singleTexts, runtime.reqLogger); + return response.ok + ? handleUpstreamSuccess(runtime, prepared, response, singleTexts?.length ?? 1) + : handleUpstreamFailure(runtime, response); + } catch (error) { + return handleEmbeddingException(runtime, prepared, error); + } +} + +/** Handle one OpenAI-compatible embedding request. */ +export async function handleEmbedding(params: HandleEmbeddingParams): Promise { + const resolved = resolveEmbedding(params); + const runtime = await createEmbeddingRuntime(params, resolved); + if ("success" in runtime) return runtime; + const modalityFailure = validateRequestedModalities(runtime); + if (modalityFailure) return modalityFailure; + const prepared = await prepareEmbeddingRequest(runtime); + if ("success" in prepared) return prepared; + runtime.log?.info( + "EMBED", + `${runtime.provider}/${runtime.model} | input: ${ + Array.isArray(runtime.body.input) ? `${runtime.body.input.length} items` : "1 item" + }` + ); + return executeEmbedding(runtime, prepared); +} diff --git a/open-sse/translator/bootstrap.ts b/open-sse/translator/bootstrap.ts index df852d483c..bd870e934a 100644 --- a/open-sse/translator/bootstrap.ts +++ b/open-sse/translator/bootstrap.ts @@ -5,6 +5,7 @@ import "./request/claude-to-openai.ts"; import "./request/openai-to-claude.ts"; +import "./request/openai-to-clova.ts"; import "./request/gemini-to-openai.ts"; import "./request/openai-to-gemini.ts"; import "./request/antigravity-to-openai.ts"; @@ -15,6 +16,7 @@ import "./request/claude-to-gemini.ts"; import "./response/claude-to-openai.ts"; import "./response/openai-to-claude.ts"; +import "./response/clova-to-openai.ts"; import "./response/gemini-to-openai.ts"; import "./response/gemini-to-claude.ts"; import "./response/openai-to-antigravity.ts"; diff --git a/open-sse/translator/formats.ts b/open-sse/translator/formats.ts index 4e0bd391f0..1349d30963 100644 --- a/open-sse/translator/formats.ts +++ b/open-sse/translator/formats.ts @@ -5,6 +5,8 @@ export const FORMATS = { OPENAI_RESPONSE: "openai-response", CLAUDE: "claude", GEMINI: "gemini", + /** Naver CLOVA Studio Chat Completions v3 (native envelope, model in URL path). */ + CLOVA: "clova", CODEX: "codex", ANTIGRAVITY: "antigravity", KIRO: "kiro", diff --git a/open-sse/translator/request/openai-to-clova.ts b/open-sse/translator/request/openai-to-clova.ts new file mode 100644 index 0000000000..62bd519c66 --- /dev/null +++ b/open-sse/translator/request/openai-to-clova.ts @@ -0,0 +1,375 @@ +/** + * OpenAI → Naver CLOVA Studio "Chat Completions v3" request translator. + * + * Wire format: `POST https://clovastudio.stream.ntruss.com/v3/chat-completions/{modelName}` + * + * Everything below that is marked "live-verified" was confirmed against the real + * API on 2026-09-01 — several of these rules contradict a plausible reading of + * the vendor docs, so they are recorded with the evidence. + * + * Vendor docs: + * - text/image: https://api.ncloud-docs.com/docs/clovastudio-chatcompletionsv3 + * - thinking: https://api.ncloud-docs.com/docs/clovastudio-chatcompletionsv3-thinking + * - FC: https://api.ncloud-docs.com/docs/clovastudio-chatcompletionsv3-fc + * - SO: https://api.ncloud-docs.com/docs/clovastudio-chatcompletionsv3-so + */ +import { register } from "../registry.ts"; +import { FORMATS } from "../formats.ts"; + +/** Output cap for the non-reasoning v3 models (HCX-005, HCX-DASH-002). */ +export const CLOVA_V3_MAX_OUTPUT_TOKENS = 4096; + +/** Output cap for the reasoning model (HCX-007) — includes thinking tokens. */ +export const CLOVA_V3_REASONING_MAX_OUTPUT_TOKENS = 32768; + +/** + * Function calling rejects any cap below 1024 (live-verified: `40001 Invalid + * parameter: tools, maxTokens`). + */ +export const CLOVA_V3_MIN_TOOL_TOKENS = 1024; + +export const CLOVA_V3_REASONING_MODELS: ReadonlySet = new Set(["HCX-007"]); + +export const CLOVA_V3_VISION_MODELS: ReadonlySet = new Set(["HCX-005"]); + +/** + * All three v3 models accept function calling (live-verified). HCX-007 needs + * `thinking.effort: "none"` alongside it or the call fails with + * `40001 Invalid parameter: tools, thinking`. + */ +export const CLOVA_V3_FUNCTION_CALLING_MODELS: ReadonlySet = new Set([ + "HCX-005", + "HCX-007", + "HCX-DASH-002", +]); + +/** Structured Outputs is HCX-007 only (live-verified: HCX-005 rejects `thinking`). */ +export const CLOVA_V3_STRUCTURED_OUTPUT_MODELS: ReadonlySet = new Set(["HCX-007"]); + +const CLOVA_THINKING_EFFORTS: ReadonlySet = new Set(["none", "low", "medium", "high"]); + +type JsonRecord = Record; + +function toRecord(value: unknown): JsonRecord | null { + return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : null; +} + +function nonEmptyString(value: unknown): string { + return typeof value === "string" && value.length > 0 ? value : ""; +} + +export function isClovaReasoningModel(model: string): boolean { + return typeof model === "string" && CLOVA_V3_REASONING_MODELS.has(model.toUpperCase()); +} + +export function isClovaVisionModel(model: string): boolean { + return typeof model === "string" && CLOVA_V3_VISION_MODELS.has(model.toUpperCase()); +} + +export function isClovaFunctionCallingModel(model: string): boolean { + return typeof model === "string" && CLOVA_V3_FUNCTION_CALLING_MODELS.has(model.toUpperCase()); +} + +export function isClovaStructuredOutputModel(model: string): boolean { + return typeof model === "string" && CLOVA_V3_STRUCTURED_OUTPUT_MODELS.has(model.toUpperCase()); +} + +function clampNumeric(value: unknown, min: number, max: number): number | null { + const n = typeof value === "string" ? Number(value) : value; + if (typeof n !== "number" || !Number.isFinite(n)) return null; + return Math.min(Math.max(n, min), max); +} + +/** + * Map OpenAI `reasoning_effort` onto CLOVA's `thinking.effort`. + * `minimal` collapses to `low`; unrecognised values are dropped so CLOVA applies + * its own default (`low`). + */ +export function toClovaThinkingEffort(reasoningEffort: unknown): string { + if (typeof reasoningEffort !== "string") return ""; + const effort = reasoningEffort.toLowerCase(); + if (effort === "minimal") return "low"; + return CLOVA_THINKING_EFFORTS.has(effort) ? effort : ""; +} + +/** Flatten OpenAI message content into a single string (text only). */ +function contentToString(content: unknown): string { + if (typeof content === "string") return content; + if (!Array.isArray(content)) return content == null ? "" : String(content); + return content + .map((part) => + part && typeof part === "object" && typeof part.text === "string" ? part.text : "" + ) + .filter(Boolean) + .join("\n"); +} + +/** + * Convert an OpenAI `content` value into CLOVA v3 typed content parts. + * + * Both image transports work (live-verified): a public URL becomes + * `imageUrl.url`, and a `data:` URL becomes `dataUri.data` — which must keep the + * FULL `data:;base64,` prefix or CLOVA rejects the request with + * `40001 Invalid parameter`. + */ +export function toClovaContent( + content: unknown, + supportsImages: boolean +): Array> { + if (typeof content === "string") { + return [{ type: "text", text: content }]; + } + + if (!Array.isArray(content)) { + return [{ type: "text", text: content == null ? "" : String(content) }]; + } + + const parts = content + .map((part) => toClovaContentPart(part, supportsImages)) + .filter((part): part is JsonRecord => part !== null); + + // CLOVA rejects a message with an empty content array, so always emit a part. + return parts.length > 0 ? parts : [{ type: "text", text: "" }]; +} + +function toClovaContentPart(part: unknown, supportsImages: boolean): JsonRecord | null { + const record = toRecord(part); + if (!record) return null; + + const text = nonEmptyString(record.text); + if (record.type === "text" || text) return text ? { type: "text", text } : null; + if (record.type !== "image_url" || !supportsImages) return null; + + const imageUrl = toRecord(record.image_url); + const url = nonEmptyString(imageUrl?.url) || nonEmptyString(record.url); + if (!url) return null; + return url.startsWith("data:") + ? { type: "image_url", dataUri: { data: url } } + : { type: "image_url", imageUrl: { url } }; +} + +/** Parse OpenAI's JSON-string tool arguments into the object CLOVA expects. */ +function toolArgumentsToObject(raw: unknown): Record { + if (raw == null) return {}; + if (typeof raw === "object") return raw as Record; + if (typeof raw !== "string" || !raw.trim()) return {}; + try { + const parsed = JSON.parse(raw); + return parsed && typeof parsed === "object" ? (parsed as Record) : {}; + } catch { + return {}; + } +} + +/** + * Convert OpenAI tool declarations into CLOVA's `tools` array. + * The shapes are nearly identical; empty declarations are skipped because CLOVA + * rejects a tool without a name. + */ +export function toClovaTools(tools: unknown): Array> { + if (!Array.isArray(tools)) return []; + return tools.map(toClovaTool).filter((tool): tool is JsonRecord => tool !== null); +} + +function toClovaTool(tool: unknown): JsonRecord | null { + const record = toRecord(tool); + if (!record) return null; + const fn = toRecord(record.function); + const name = nonEmptyString(fn?.name) || nonEmptyString(record.name); + if (!name) return null; + + const description = + nonEmptyString(fn?.description) || nonEmptyString(record.description) || `Tool: ${name}`; + const parameters = fn?.parameters ?? record.parameters; + return { + type: "function", + function: { + name, + description, + ...(parameters ? { parameters } : {}), + }, + }; +} + +/** + * Which mutually-exclusive v3 mode does this request use? + * + * CLOVA forbids combining function calling with thinking or images, and forbids + * combining structured outputs with either. Exactly one mode is chosen. + */ +export function resolveClovaMode( + model: string, + body: Record +): "tools" | "structured" | "plain" { + const tools = toClovaTools(body?.tools); + if (tools.length > 0 && isClovaFunctionCallingModel(model)) return "tools"; + + const format = toRecord(body?.response_format); + const wantsSchema = format && (format.type === "json_schema" || format.type === "json_object"); + if (wantsSchema && isClovaStructuredOutputModel(model)) return "structured"; + + return "plain"; +} + +type ClovaMode = "tools" | "structured" | "plain"; + +function normalizeMessageRole(role: unknown): "assistant" | "system" | "user" { + return role === "assistant" || role === "system" ? role : "user"; +} + +function toClovaToolCall(call: unknown): JsonRecord { + const record = toRecord(call) ?? {}; + const fn = toRecord(record.function); + return { + id: record.id ?? "", + type: "function", + function: { + name: fn?.name ?? record.name ?? "", + arguments: toolArgumentsToObject(fn?.arguments ?? record.arguments), + }, + }; +} + +function toClovaToolModeMessage(message: unknown): JsonRecord { + const record = toRecord(message) ?? {}; + if (record.role === "tool") { + return { + role: "tool", + content: contentToString(record.content), + ...(record.tool_call_id ? { toolCallId: String(record.tool_call_id) } : {}), + }; + } + + const toolCalls = Array.isArray(record.tool_calls) ? record.tool_calls : []; + if (record.role === "assistant" && toolCalls.length > 0) { + return { + role: "assistant", + content: "", + toolCalls: toolCalls.map(toClovaToolCall), + }; + } + return { + role: normalizeMessageRole(record.role), + content: contentToString(record.content), + }; +} + +function toClovaPlainMessage(message: unknown, supportsImages: boolean): JsonRecord { + const record = toRecord(message) ?? {}; + return { + role: normalizeMessageRole(record.role), + content: toClovaContent(record.content, supportsImages), + }; +} + +function toClovaMessages(body: JsonRecord, mode: ClovaMode, supportsImages: boolean): JsonRecord[] { + const messages = Array.isArray(body.messages) ? body.messages : []; + return messages.map((message) => + mode === "tools" + ? toClovaToolModeMessage(message) + : toClovaPlainMessage(message, supportsImages) + ); +} + +function applyThinking(payload: JsonRecord, body: JsonRecord, reasoning: boolean, mode: ClovaMode) { + if (!reasoning) return; + const effort = toClovaThinkingEffort(body.reasoning_effort); + if (mode === "tools" || mode === "structured") { + payload.thinking = { effort: "none" }; + } else if (effort) { + payload.thinking = { effort }; + } +} + +function applySampling(payload: JsonRecord, body: JsonRecord): void { + const temperature = clampNumeric(body.temperature, 0, 1); + if (temperature !== null) payload.temperature = temperature; + const topP = clampNumeric(body.top_p, 0, 1); + if (topP !== null && topP > 0) payload.topP = topP; + const topK = clampNumeric(body.top_k, 0, 128); + if (topK !== null && topK > 0) payload.topK = topK; + const penalty = clampNumeric(body.repetition_penalty, 0, 2); + if (penalty !== null && penalty > 0) payload.repetitionPenalty = penalty; +} + +function applyOutputCap( + payload: JsonRecord, + body: JsonRecord, + reasoning: boolean, + mode: ClovaMode +): void { + const cap = reasoning ? CLOVA_V3_REASONING_MAX_OUTPUT_TOKENS : CLOVA_V3_MAX_OUTPUT_TOKENS; + const key = reasoning ? "maxCompletionTokens" : "maxTokens"; + let tokens = clampNumeric(body.max_completion_tokens ?? body.max_tokens, 1, cap); + if (mode === "tools") { + const floor = Math.min(CLOVA_V3_MIN_TOOL_TOKENS, cap); + tokens = tokens === null ? floor : Math.max(tokens, floor); + } + if (tokens !== null) payload[key] = tokens; +} + +function responseSchema(body: JsonRecord): unknown { + const format = toRecord(body.response_format); + const jsonSchema = toRecord(format?.json_schema); + return jsonSchema?.schema ?? format?.schema; +} + +function applyModeFields(payload: JsonRecord, body: JsonRecord, mode: ClovaMode): void { + if (mode === "tools") { + payload.tools = toClovaTools(body.tools); + if (body.tool_choice === "none") payload.toolChoice = "none"; + if (body.tool_choice === "auto" || body.tool_choice === "required") { + payload.toolChoice = "auto"; + } + return; + } + if (mode !== "structured") return; + const schema = responseSchema(body); + if (schema && typeof schema === "object") { + payload.responseFormat = { type: "json", schema }; + } else { + delete payload.thinking; + } +} + +function applyPlainOptions( + payload: JsonRecord, + body: JsonRecord, + reasoning: boolean, + mode: ClovaMode +): void { + if (mode === "plain" && !reasoning) { + if (Array.isArray(body.stop) && body.stop.length > 0) { + payload.stop = body.stop.filter((value) => typeof value === "string"); + } else if (typeof body.stop === "string" && body.stop) { + payload.stop = [body.stop]; + } + } + const seed = clampNumeric(body.seed, 0, 4294967295); + if (seed !== null && seed > 0) payload.seed = Math.floor(seed); + if (body.include_ai_filters === true) payload.includeAiFilters = true; +} + +/** Build the CLOVA Studio v3 request body from an OpenAI Chat Completions body. */ +export function buildClovaPayload( + model: string, + body: Record, + stream: boolean, + credentials?: Record | null +): Record { + void stream; + void credentials; + const reasoning = isClovaReasoningModel(model); + const mode = resolveClovaMode(model, body); + const supportsImages = mode === "plain" && isClovaVisionModel(model); + const payload: JsonRecord = { messages: toClovaMessages(body, mode, supportsImages) }; + + applyThinking(payload, body, reasoning, mode); + applySampling(payload, body); + applyOutputCap(payload, body, reasoning, mode); + applyModeFields(payload, body, mode); + applyPlainOptions(payload, body, reasoning, mode); + return payload; +} + +register(FORMATS.OPENAI, FORMATS.CLOVA, buildClovaPayload, null); diff --git a/open-sse/translator/response/clova-to-openai.ts b/open-sse/translator/response/clova-to-openai.ts new file mode 100644 index 0000000000..4d2b9cee54 --- /dev/null +++ b/open-sse/translator/response/clova-to-openai.ts @@ -0,0 +1,354 @@ +/** + * Naver CLOVA Studio "Chat Completions v3" → OpenAI response translator. + * + * CLOVA v3 streams as SSE with **named events**: + * + * ``` + * id: + * event: token + * data: {"message":{"role":"assistant","content":"안"},"finishReason":null,...} + * + * id: + * event: result + * data: {"message":{"role":"assistant","content":"안녕"},"finishReason":"stop", + * "usage":{"promptTokens":20,"completionTokens":5,"totalTokens":25}} + * ``` + * + * Three traps this translator exists to defuse: + * + * 1. **`event: token` carries an incremental delta, but `event: result` repeats + * the COMPLETE text.** Concatenating both duplicates the whole answer at the + * end of the stream, so the result event is treated as a terminal snapshot: + * it contributes `finish_reason` + `usage` only. + * 2. **Function-calling streams deliver arguments as `partialJson` fragments.** + * The first token carries the tool `id` + `name`; every later token carries + * only a JSON fragment (`{`, `"location`, `":`, ` "`, `Se`, `oul`, `"}`), + * which have to be reassembled into OpenAI's `tool_calls[].function.arguments` + * string. The terminal frame repeats the finished call, so — same rule as the + * text snapshot — it is not re-emitted. + * 3. **Failures can arrive as an in-stream payload** whose `status.code` is not + * `20000`, not just as an HTTP error. Those are surfaced through + * `state.upstreamError` so stream.ts fails the request out and combo fallback + * can run, mirroring the Gemini translator. + * + * Docs: https://api.ncloud-docs.com/docs/clovastudio-chatcompletionsv3 + */ +import { register } from "../registry.ts"; +import { FORMATS } from "../formats.ts"; + +/** CLOVA's success status code (a string, not an HTTP number). */ +const CLOVA_STATUS_OK = "20000"; + +type JsonRecord = Record; + +interface ClovaStreamState extends JsonRecord { + responseId?: string; + created?: number; + model?: string; + chunkIndex?: number; + usage?: { prompt_tokens: number; completion_tokens: number; total_tokens: number }; + upstreamError?: { status: number; type: string; code: string; message: string }; + toolCallStarted?: boolean; + finishReason?: string; +} + +function toRecord(value: unknown): JsonRecord | null { + return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : null; +} + +/** Map a CLOVA `finishReason` onto the OpenAI vocabulary. */ +function mapFinishReason(reason: unknown): string { + switch (String(reason || "")) { + case "length": + return "length"; + case "tool_calls": + return "tool_calls"; + case "content_filter": + return "content_filter"; + default: + return "stop"; + } +} + +/** + * Map a CLOVA string status code onto an HTTP status for error surfacing. + * Codes are 5-digit strings: `2xxxx` success, `4xxxx` client, `5xxxx` server. + */ +function httpStatusFromClovaCode(code: unknown): number { + const first = String(code || "").charAt(0); + if (first === "4") return 400; + return 502; +} + +/** + * Parse one raw SSE frame into `{ event, data }`. + * CLOVA emits `id:` / `event:` / `data:` lines per frame. + */ +export function parseClovaSseFrame(raw: string): { event: string; data: unknown } | null { + if (typeof raw !== "string" || !raw.trim()) return null; + + let event = ""; + let dataLine = ""; + + for (const line of raw.split("\n")) { + const trimmed = line.trim(); + if (trimmed.startsWith("event:")) { + event = trimmed.slice(6).trim(); + } else if (trimmed.startsWith("data:")) { + dataLine = trimmed.slice(5).trim(); + } + } + + if (!dataLine) return null; + + try { + return { event, data: JSON.parse(dataLine) }; + } catch { + return null; + } +} + +function baseChunk(state: ClovaStreamState): Record { + return { + id: state.responseId, + object: "chat.completion.chunk", + created: state.created, + model: state.model || "clova", + }; +} + +/** + * Build one OpenAI delta chunk. + * + * `field` selects the delta key: `"content"` for the visible answer and + * `"reasoning_content"` for CLOVA's `thinkingContent` (HCX-007). + */ +function deltaChunk( + state: ClovaStreamState, + content: string, + field = "content" +): Record { + const chunk = baseChunk(state); + chunk.choices = [ + { + index: 0, + delta: { + ...((state.chunkIndex ?? 0) === 0 ? { role: "assistant" } : {}), + [field]: content, + }, + finish_reason: null, + }, + ]; + state.chunkIndex = (state.chunkIndex ?? 0) + 1; + return chunk; +} + +/** First tool-call chunk: carries id + name and opens an empty argument string. */ +function toolCallStartChunk( + state: ClovaStreamState, + id: string, + name: string +): Record { + const chunk = baseChunk(state); + chunk.choices = [ + { + index: 0, + delta: { + ...((state.chunkIndex ?? 0) === 0 ? { role: "assistant" } : {}), + tool_calls: [ + { + index: 0, + id: id || `call_${state.responseId}`, + type: "function", + function: { name, arguments: "" }, + }, + ], + }, + finish_reason: null, + }, + ]; + state.chunkIndex = (state.chunkIndex ?? 0) + 1; + return chunk; +} + +/** Subsequent tool-call chunk: appends one `partialJson` fragment. */ +function toolCallArgumentsChunk( + state: ClovaStreamState, + fragment: string +): Record { + const chunk = baseChunk(state); + chunk.choices = [ + { + index: 0, + delta: { tool_calls: [{ index: 0, function: { arguments: fragment } }] }, + finish_reason: null, + }, + ]; + state.chunkIndex = (state.chunkIndex ?? 0) + 1; + return chunk; +} + +function terminalChunk(state: ClovaStreamState, finishReason: string): Record { + const chunk = baseChunk(state); + chunk.choices = [{ index: 0, delta: {}, finish_reason: finishReason }]; + if (state.usage) chunk.usage = state.usage; + return chunk; +} + +function recordUsage(state: ClovaStreamState, usage: unknown): void { + const record = toRecord(usage); + if (!record) return; + const prompt = Number(record.promptTokens) || 0; + const completion = Number(record.completionTokens) || 0; + const total = Number(record.totalTokens) || prompt + completion; + state.usage = { + prompt_tokens: prompt, + completion_tokens: completion, + total_tokens: total, + }; +} + +function recordUpstreamError(state: ClovaStreamState, code: unknown, message: unknown): void { + const status = httpStatusFromClovaCode(code); + state.upstreamError = { + status, + type: status === 429 ? "rate_limit_error" : "server_error", + code: String(code || "clova_error"), + message: typeof message === "string" && message ? message : "CLOVA Studio upstream failure", + }; +} + +interface DecodedClovaChunk { + event: string; + data: JsonRecord; +} + +function initializeState(state: ClovaStreamState): void { + if (state.responseId) return; + state.responseId = `chatcmpl-${Date.now()}`; + state.created = Math.floor(Date.now() / 1000); + state.chunkIndex = 0; +} + +function decodeClovaChunk(chunk: unknown): DecodedClovaChunk | null { + if (typeof chunk === "string") { + const frame = parseClovaSseFrame(chunk); + const data = toRecord(frame?.data); + return frame && data ? { event: frame.event, data } : null; + } + const data = toRecord(chunk); + if (!data) return null; + return { event: String(data.event || data._eventType || ""), data }; +} + +function handleErrorEnvelope(state: ClovaStreamState, event: string, data: JsonRecord): boolean { + const status = toRecord(data.status); + const statusCode = status?.code ?? data.statusCode; + if (statusCode != null && String(statusCode) !== CLOVA_STATUS_OK) { + recordUpstreamError(state, statusCode, status?.message ?? data.message); + return true; + } + + const error = toRecord(data.error); + if (event !== "error" && !error) return false; + const source = error ?? data; + const errorStatus = toRecord(source.status); + recordUpstreamError( + state, + errorStatus?.code ?? source.code, + errorStatus?.message ?? source.message + ); + return true; +} + +function toolCallDelta(state: ClovaStreamState, call: unknown): Record | null { + const record = toRecord(call); + const fn = toRecord(record?.function); + if (!record || !fn) return null; + const id = typeof record.id === "string" ? record.id : ""; + const name = typeof fn.name === "string" ? fn.name : ""; + if (id || name) { + if (state.toolCallStarted) return null; + state.toolCallStarted = true; + return toolCallStartChunk(state, id, name); + } + return typeof fn.partialJson === "string" && fn.partialJson + ? toolCallArgumentsChunk(state, fn.partialJson) + : null; +} + +function toolCallDeltas( + state: ClovaStreamState, + message: JsonRecord +): Record | Array> | null { + if (!Array.isArray(message.toolCalls) || message.toolCalls.length === 0) return null; + const out = message.toolCalls + .map((call) => toolCallDelta(state, call)) + .filter((chunk): chunk is Record => chunk !== null); + if (out.length === 0) return null; + return out.length === 1 ? out[0] : out; +} + +function convertTokenEvent( + state: ClovaStreamState, + data: JsonRecord +): Record | Array> | null { + const message = toRecord(data.message) ?? data; + const toolDeltas = toolCallDeltas(state, message); + if (toolDeltas) return toolDeltas; + const thinking = message.thinkingContent ?? data.thinkingContent; + if (thinking) return deltaChunk(state, String(thinking), "reasoning_content"); + const content = message.content ?? data.content; + return content ? deltaChunk(state, String(content)) : null; +} + +function shouldEmitResultSnapshot( + state: ClovaStreamState, + isResultEvent: boolean, + snapshot: unknown +): snapshot is string { + return ( + !isResultEvent && (state.chunkIndex ?? 0) === 0 && typeof snapshot === "string" && !!snapshot + ); +} + +function convertResultEvent( + state: ClovaStreamState, + event: string, + data: JsonRecord +): Record | Array> | null { + const isResultEvent = event === "result" || event === "stop"; + const resultEnvelope = toRecord(data.result); + if (!isResultEvent && (event || !resultEnvelope)) return null; + + const result = resultEnvelope ?? data; + const message = toRecord(result.message); + recordUsage(state, result.usage); + const hasToolCalls = Array.isArray(message?.toolCalls) && message.toolCalls.length > 0; + const finishReason = hasToolCalls ? "tool_calls" : mapFinishReason(result.finishReason); + state.finishReason = finishReason; + + const snapshot = message?.content ?? result.content; + if (shouldEmitResultSnapshot(state, isResultEvent, snapshot)) { + return [deltaChunk(state, snapshot), terminalChunk(state, finishReason)]; + } + return terminalChunk(state, finishReason); +} + +/** Convert one CLOVA stream frame or JSON envelope into OpenAI chunk(s). */ +export function convertClovaToOpenAI( + chunk: unknown, + state: Record +): Record | Array> | null { + if (chunk == null) return null; + const streamState = state as ClovaStreamState; + initializeState(streamState); + const decoded = decodeClovaChunk(chunk); + if (!decoded) return null; + if (handleErrorEnvelope(streamState, decoded.event, decoded.data)) return null; + return decoded.event === "token" + ? convertTokenEvent(streamState, decoded.data) + : convertResultEvent(streamState, decoded.event, decoded.data); +} + +register(FORMATS.CLOVA, FORMATS.OPENAI, null, convertClovaToOpenAI); diff --git a/scripts/check/check-known-symbols.ts b/scripts/check/check-known-symbols.ts index e655c169bb..769afa5493 100644 --- a/scripts/check/check-known-symbols.ts +++ b/scripts/check/check-known-symbols.ts @@ -211,6 +211,8 @@ export const KNOWN_TRANSLATOR_PAIRS: readonly string[] = [ "antigravity:openai", "claude:gemini", "claude:openai", + // Naver CLOVA Studio Chat Completions v3 (native envelope, model in URL path). + "clova:openai", "cursor:openai", "gemini:claude", "gemini:openai", @@ -218,6 +220,7 @@ export const KNOWN_TRANSLATOR_PAIRS: readonly string[] = [ "openai-responses:openai", "openai:antigravity", "openai:claude", + "openai:clova", "openai:cursor", "openai:gemini", "openai:kiro", diff --git a/src/shared/constants/providers/apikey/regional.ts b/src/shared/constants/providers/apikey/regional.ts index a9c5701dfd..2c437cc4b3 100644 --- a/src/shared/constants/providers/apikey/regional.ts +++ b/src/shared/constants/providers/apikey/regional.ts @@ -494,7 +494,7 @@ export const APIKEY_PROVIDERS_REGIONAL = { textIcon: "CS", website: "https://api.ncloud-docs.com/docs/en/ai-naver-clovastudio-summary", apiHint: - "CLOVA Studio (HyperCLOVA X) is OpenAI-compatible on /v1/openai. OmniRoute probes /v1/openai/models and routes chat traffic to /v1/openai/chat/completions. Uses the current clovastudio.stream.ntruss.com host — the legacy clovastudio.apigw.ntruss.com endpoint is being deprecated.", + "OmniRoute routes chat traffic to the native Chat Completions v3 API (/v3/chat-completions/{model}), not the OpenAI-compatibility shim. All three v3 models are served: HCX-007 (reasoning, text only), HCX-005 (vision — accepts both public image URLs and inline base64 images), and HCX-DASH-002 (lightweight, text only). Requests stream upstream and are accumulated into a JSON body when the client asks for a non-streaming response.", }, internlm: { id: "internlm", diff --git a/src/shared/constants/visionModels.ts b/src/shared/constants/visionModels.ts index da7b755490..bcad514058 100644 --- a/src/shared/constants/visionModels.ts +++ b/src/shared/constants/visionModels.ts @@ -57,6 +57,11 @@ export const VISION_MODEL_ID_FRAGMENTS = [ "mistral-medium-3", "minimax-m3", "kimi-k2.", + // Naver CLOVA Studio: HCX-005 is the only v3 model with image input. Listed by + // exact id (not a family fragment) to stay conservative — live-verified on + // 2026-09-01 that it answers image prompts over both a public URL and a + // base64 data URI, while HCX-007 and HCX-DASH-002 reject images. + "hcx-005", "-vision", "multimodal", ] as const; diff --git a/tests/snapshots/executors/executor-map.json b/tests/snapshots/executors/executor-map.json index 94d70fb770..8f4956e814 100644 --- a/tests/snapshots/executors/executor-map.json +++ b/tests/snapshots/executors/executor-map.json @@ -135,6 +135,11 @@ "configSource": "", "provider": "cloudflare-playground" }, + "clova-studio": { + "className": "ClovaStudioExecutor", + "configSource": "clova-studio", + "provider": "clova-studio" + }, "cmd": { "className": "CommandCodeExecutor", "configSource": "", @@ -671,6 +676,6 @@ "provider": "zai-web" } }, - "keyCount": 134, + "keyCount": 135, "sharedInstances": [] } diff --git a/tests/snapshots/provider/translate-path.json b/tests/snapshots/provider/translate-path.json index 43150e2b22..64fd60cfa2 100644 --- a/tests/snapshots/provider/translate-path.json +++ b/tests/snapshots/provider/translate-path.json @@ -1217,7 +1217,7 @@ } }, "clova-studio": { - "format": "openai", + "format": "clova", "headers": { "apiKey": { "Accept": "text/event-stream", @@ -1235,8 +1235,8 @@ } }, "url": { - "nonStream": "https://clovastudio.stream.ntruss.com/v1/openai/chat/completions", - "stream": "https://clovastudio.stream.ntruss.com/v1/openai/chat/completions" + "nonStream": "https://clovastudio.stream.ntruss.com/v3/chat-completions", + "stream": "https://clovastudio.stream.ntruss.com/v3/chat-completions" } }, "codebuddy-cn": { diff --git a/tests/unit/embedding-clova-v2.test.ts b/tests/unit/embedding-clova-v2.test.ts new file mode 100644 index 0000000000..9a97a1c5ee --- /dev/null +++ b/tests/unit/embedding-clova-v2.test.ts @@ -0,0 +1,258 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { mkdtempSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; + +// Naver CLOVA Studio embedding v2. +// +// The endpoint embeds exactly ONE text per request (`{"text": …}` → one vector) +// and answers `{status, result:{embedding:[…1024 floats], inputTokens}}`, so a +// batched `/v1/embeddings` call has to be fanned out into N upstream calls and +// merged back into OpenAI's list shape. +// +// Live-verified against the API on 2026-09-01: 1024 dimensions, ~100ms per call, +// and an empty string is rejected with `40004 Text empty`. + +const TEST_DATA_DIR = mkdtempSync(join(tmpdir(), "omniroute-clova-embeddings-")); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.SQLITE_FILE = join(TEST_DATA_DIR, "storage.sqlite"); + +const registry = await import("../../open-sse/config/embeddingRegistry.ts"); +const { normalizeClovaEmbeddingV2Response } = + await import("../../open-sse/handlers/embeddingStructuredInput.ts"); +const { handleEmbedding } = await import("../../open-sse/handlers/embeddings.ts"); +const { resetDbInstance } = await import("../../src/lib/db/core.ts"); + +test.after(async () => { + // handleEmbedding records call logs asynchronously; let those writes settle + // before closing the singleton so a late write cannot reopen the test DB. + await new Promise((resolve) => setTimeout(resolve, 50)); + resetDbInstance(); +}); + +// --------------------------------------------------------------------------- +// Registry +// --------------------------------------------------------------------------- + +test("clova embedding v2 is registered with the single-text protocol", () => { + const provider = registry.getEmbeddingProvider("clova-studio"); + assert.ok(provider, "clova-studio must be an embedding provider"); + assert.equal(provider.baseUrl, "https://clovastudio.stream.ntruss.com/v1/api-tools/embedding/v2"); + assert.equal(provider.singleTextProtocol, "clova-v2"); + assert.equal(provider.authType, "apikey"); + assert.equal(provider.authHeader, "bearer"); + assert.deepEqual(provider.models, [ + { id: "clova-embedding-v2", name: "CLOVA Embedding v2", dimensions: 1024 }, + ]); +}); + +test("clova embedding v2 resolves its model and dimension", () => { + assert.deepEqual(registry.parseEmbeddingModel("clova-studio/clova-embedding-v2"), { + provider: "clova-studio", + model: "clova-embedding-v2", + }); + assert.equal(registry.getEmbeddingDimension("clova-studio/clova-embedding-v2"), 1024); +}); + +// --------------------------------------------------------------------------- +// Response normalisation +// --------------------------------------------------------------------------- + +test("a success envelope is normalised into OpenAI list shape", () => { + const normalized = normalizeClovaEmbeddingV2Response({ + status: { code: "20000", message: "OK" }, + result: { embedding: [0.1, -0.2, 0.3], inputTokens: 4 }, + }); + assert.deepEqual(normalized, { + data: [{ object: "embedding", index: 0, embedding: [0.1, -0.2, 0.3] }], + usage: { prompt_tokens: 4, total_tokens: 4 }, + }); +}); + +test("a failure envelope is rejected instead of becoming an empty success", () => { + assert.throws( + () => + normalizeClovaEmbeddingV2Response({ + status: { code: "40004", message: "Text empty" }, + }), + /unsuccessful status/ + ); +}); + +test("a payload without an embedding vector is rejected", () => { + assert.throws( + () => + normalizeClovaEmbeddingV2Response({ + status: { code: "20000" }, + result: { inputTokens: 0 }, + }), + /missing an embedding vector/ + ); +}); + +// --------------------------------------------------------------------------- +// Batch fan-out through the real handler +// --------------------------------------------------------------------------- + +const originalFetch = globalThis.fetch; + +function mockClova(calls: Array>): void { + globalThis.fetch = (async (_url: string, init: RequestInit) => { + calls.push(JSON.parse(String(init.body))); + const text = String((calls[calls.length - 1] as { text?: string }).text ?? ""); + return new Response( + JSON.stringify({ + status: { code: "20000", message: "OK" }, + result: { embedding: [text.length, 1, 2], inputTokens: text.length }, + }), + { status: 200, headers: { "Content-Type": "application/json" } } + ); + }) as typeof fetch; +} + +function mockClovaEnvelope( + calls: Array>, + envelope: Record +): void { + globalThis.fetch = (async (_url: string, init: RequestInit) => { + calls.push(JSON.parse(String(init.body))); + return new Response(JSON.stringify(envelope), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + }) as typeof fetch; +} + +test("a batched input is fanned out into one upstream call per text", async () => { + const calls: Array> = []; + mockClova(calls); + try { + const result = await handleEmbedding({ + body: { model: "clova-studio/clova-embedding-v2", input: ["alpha", "beta", "gamma"] }, + credentials: { apiKey: "test-key" }, + }); + + assert.equal(result.success, true, JSON.stringify(result)); + // One request per text — the endpoint cannot batch. + assert.deepEqual( + calls.map((c) => c.text), + ["alpha", "beta", "gamma"] + ); + + const data = (result as { data: Record }).data; + assert.equal(data.object, "list"); + assert.equal(data.model, "clova-studio/clova-embedding-v2"); + assert.equal((data.data as unknown[]).length, 3); + // Indexes must reflect the caller's positions, not each upstream call's 0. + assert.deepEqual( + (data.data as Array<{ index: number }>).map((d) => d.index), + [0, 1, 2] + ); + // Token usage is summed across the fan-out. + assert.equal((data.usage as { prompt_tokens: number }).prompt_tokens, 5 + 4 + 5); + } finally { + globalThis.fetch = originalFetch; + } +}); + +test("an empty text rejects the batch without changing response indexes", async () => { + const calls: Array> = []; + mockClova(calls); + try { + const result = await handleEmbedding({ + body: { model: "clova-studio/clova-embedding-v2", input: ["", " ", "real"] }, + credentials: { apiKey: "test-key" }, + }); + assert.equal(result.success, false); + assert.equal((result as { status: number }).status, 400); + assert.equal(calls.length, 0); + } finally { + globalThis.fetch = originalFetch; + } +}); + +test("an all-empty input fails without calling upstream", async () => { + const calls: Array> = []; + mockClova(calls); + try { + const result = await handleEmbedding({ + body: { model: "clova-studio/clova-embedding-v2", input: ["", ""] }, + credentials: { apiKey: "test-key" }, + }); + assert.equal(result.success, false); + assert.equal((result as { status: number }).status, 400); + assert.equal(calls.length, 0); + } finally { + globalThis.fetch = originalFetch; + } +}); + +test("a single string input takes the fan-out path too", async () => { + const calls: Array> = []; + mockClova(calls); + try { + const result = await handleEmbedding({ + body: { model: "clova-studio/clova-embedding-v2", input: "solo" }, + credentials: { apiKey: "test-key" }, + }); + assert.equal(result.success, true, JSON.stringify(result)); + assert.deepEqual( + calls.map((c) => c.text), + ["solo"] + ); + assert.equal(((result as { data: Record }).data.data as unknown[]).length, 1); + } finally { + globalThis.fetch = originalFetch; + } +}); + +test("an HTTP-200 CLOVA error envelope becomes a provider failure", async () => { + const calls: Array> = []; + mockClovaEnvelope(calls, { status: { code: "40004", message: "Text empty" } }); + try { + const result = await handleEmbedding({ + body: { model: "clova-studio/clova-embedding-v2", input: "text" }, + credentials: { apiKey: "test-key" }, + }); + assert.equal(result.success, false); + assert.equal((result as { status: number }).status, 502); + assert.equal(calls.length, 1); + } finally { + globalThis.fetch = originalFetch; + } +}); + +test("token-array input is rejected instead of being silently dropped", async () => { + const calls: Array> = []; + mockClova(calls); + try { + const result = await handleEmbedding({ + body: { model: "clova-studio/clova-embedding-v2", input: [101, 202] }, + credentials: { apiKey: "test-key" }, + }); + assert.equal(result.success, false); + assert.equal((result as { status: number }).status, 400); + assert.equal(calls.length, 0); + } finally { + globalThis.fetch = originalFetch; + } +}); + +test("unsupported output options are rejected instead of ignored", async () => { + for (const extra of [{ encoding_format: "base64" }, { dimensions: 1536 }]) { + const calls: Array> = []; + mockClova(calls); + try { + const result = await handleEmbedding({ + body: { model: "clova-studio/clova-embedding-v2", input: "text", ...extra }, + credentials: { apiKey: "test-key" }, + }); + assert.equal(result.success, false); + assert.equal((result as { status: number }).status, 400); + assert.equal(calls.length, 0); + } finally { + globalThis.fetch = originalFetch; + } + } +}); diff --git a/tests/unit/translator-clova-v3.test.ts b/tests/unit/translator-clova-v3.test.ts new file mode 100644 index 0000000000..fb9851f7d3 --- /dev/null +++ b/tests/unit/translator-clova-v3.test.ts @@ -0,0 +1,825 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +// Naver CLOVA Studio "Chat Completions v3" translator pair. +// +// The guard that matters most here is the stream-duplication case: `event: token` +// carries an INCREMENTAL delta while the terminal `event: result` repeats the +// COMPLETE text. Concatenating both doubles the whole answer at the end of the +// stream, so the result event must contribute finish_reason + usage only. +// +// Wire docs: https://api.ncloud-docs.com/docs/clovastudio-chatcompletionsv3 + +const request = await import("../../open-sse/translator/request/openai-to-clova.ts"); +const response = await import("../../open-sse/translator/response/clova-to-openai.ts"); +const registry = await import("../../open-sse/translator/registry.ts"); +const { FORMATS } = await import("../../open-sse/translator/formats.ts"); +const { getExecutor } = await import("../../open-sse/executors/index.ts"); + +// --------------------------------------------------------------------------- +// Request: OpenAI → CLOVA v3 +// --------------------------------------------------------------------------- + +test("clova v3: registers the request and response translator pair", () => { + assert.ok(registry.getRequestTranslator(FORMATS.OPENAI, FORMATS.CLOVA)); + assert.ok(registry.getResponseTranslator(FORMATS.CLOVA, FORMATS.OPENAI)); +}); + +test("clova v3: its executor appends and URL-encodes the selected model", async () => { + const executor = await getExecutor("clova-studio"); + assert.equal( + executor.buildUrl("HCX 005", true), + "https://clovastudio.stream.ntruss.com/v3/chat-completions/HCX%20005" + ); +}); + +test("clova v3: string content becomes a typed text part", () => { + const body = { messages: [{ role: "user", content: "hello" }] }; + const payload = request.buildClovaPayload("HCX-005", body, true, null); + assert.deepEqual(payload.messages[0], { + role: "user", + content: [{ type: "text", text: "hello" }], + }); +}); + +test("clova v3: sampling params are camelCased", () => { + const payload = request.buildClovaPayload( + "HCX-005", + { + messages: [{ role: "user", content: "hi" }], + max_tokens: 512, + top_p: 0.8, + top_k: 4, + temperature: 0.5, + repetition_penalty: 1.15, + seed: 42, + stop: ["END"], + }, + true, + null + ); + assert.equal(payload.maxTokens, 512); + assert.equal(payload.topP, 0.8); + assert.equal(payload.topK, 4); + assert.equal(payload.temperature, 0.5); + assert.equal(payload.repetitionPenalty, 1.15); + assert.equal(payload.seed, 42); + assert.deepEqual(payload.stop, ["END"]); + // snake_case must not leak upstream. + assert.equal(payload.max_tokens, undefined); + assert.equal(payload.top_p, undefined); +}); + +test("clova v3: output tokens are clamped to the documented 4096 cap", () => { + const payload = request.buildClovaPayload( + "HCX-005", + { messages: [{ role: "user", content: "hi" }], max_tokens: 100000 }, + true, + null + ); + assert.equal(payload.maxTokens, request.CLOVA_V3_MAX_OUTPUT_TOKENS); +}); + +test("clova v3: max_completion_tokens on a text model still maps to maxTokens", () => { + // Only reasoning models speak `maxCompletionTokens`; for text models the cap is + // `maxTokens` regardless of which OpenAI alias the client used. + const payload = request.buildClovaPayload( + "HCX-005", + { messages: [{ role: "user", content: "hi" }], max_completion_tokens: 1024 }, + true, + null + ); + assert.equal(payload.maxTokens, 1024); + assert.equal(payload.maxCompletionTokens, undefined); +}); + +test("clova v3: model and stream are not sent in the body", () => { + const payload = request.buildClovaPayload( + "HCX-005", + { model: "HCX-005", stream: true, messages: [{ role: "user", content: "hi" }] }, + true, + null + ); + // The model travels in the URL path and streaming is driven by Accept. + assert.equal(payload.model, undefined); + assert.equal(payload.stream, undefined); +}); + +// --------------------------------------------------------------------------- +// Function calling (v3-fc) — same endpoint, different body fields +// --------------------------------------------------------------------------- + +test("clova v3: tools are translated and toolChoice auto is forwarded", () => { + const payload = request.buildClovaPayload( + "HCX-005", + { + messages: [{ role: "user", content: "Weather in Seoul?" }], + tools: [ + { + type: "function", + function: { + name: "get_weather", + description: "Get the weather for a city", + parameters: { + type: "object", + properties: { location: { type: "string" } }, + required: ["location"], + }, + }, + }, + ], + tool_choice: "auto", + }, + true, + null + ); + assert.deepEqual(payload.tools, [ + { + type: "function", + function: { + name: "get_weather", + description: "Get the weather for a city", + parameters: { + type: "object", + properties: { location: { type: "string" } }, + required: ["location"], + }, + }, + }, + ]); + assert.equal(payload.toolChoice, "auto"); +}); + +test("clova v3: toolChoice none is forwarded; a forced choice is dropped", () => { + // Live-verified: `toolChoice: {type:"function", function:{name}}` returns + // `40009 Unsupported function` — CLOVA only accepts "auto" and "none". + const none = request.buildClovaPayload( + "HCX-005", + { + messages: [{ role: "user", content: "hi" }], + tools: [{ type: "function", function: { name: "f" } }], + tool_choice: "none", + }, + true, + null + ); + assert.equal(none.toolChoice, "none"); + + const forced = request.buildClovaPayload( + "HCX-005", + { + messages: [{ role: "user", content: "hi" }], + tools: [{ type: "function", function: { name: "f" } }], + tool_choice: { type: "function", function: { name: "f" } }, + }, + true, + null + ); + assert.equal(forced.toolChoice, undefined); +}); + +test("clova v3: function calling raises the cap to the documented 1024 minimum", () => { + // Live-verified: any cap below 1024 fails with + // `40001 Invalid parameter: tools, maxTokens`. + const below = request.buildClovaPayload( + "HCX-005", + { + messages: [{ role: "user", content: "hi" }], + tools: [{ type: "function", function: { name: "f" } }], + max_tokens: 128, + }, + true, + null + ); + assert.equal(below.maxTokens, request.CLOVA_V3_MIN_TOOL_TOKENS); + + const absent = request.buildClovaPayload( + "HCX-005", + { + messages: [{ role: "user", content: "hi" }], + tools: [{ type: "function", function: { name: "f" } }], + }, + true, + null + ); + assert.equal(absent.maxTokens, request.CLOVA_V3_MIN_TOOL_TOKENS); + + const above = request.buildClovaPayload( + "HCX-005", + { + messages: [{ role: "user", content: "hi" }], + tools: [{ type: "function", function: { name: "f" } }], + max_tokens: 2048, + }, + true, + null + ); + assert.equal(above.maxTokens, 2048); +}); + +test("clova v3: function calling forces thinking.effort none on the reasoning model", () => { + // Live-verified: HCX-007 without it returns + // `40001 Invalid parameter: tools, thinking`. + const payload = request.buildClovaPayload( + "HCX-007", + { + messages: [{ role: "user", content: "hi" }], + tools: [{ type: "function", function: { name: "f" } }], + max_tokens: 2048, + }, + true, + null + ); + assert.deepEqual(payload.thinking, { effort: "none" }); + assert.equal(payload.maxCompletionTokens, 2048); + assert.equal(payload.maxTokens, undefined); +}); + +test("clova v3: non-reasoning models never receive a thinking field", () => { + // Regression guard: HCX-005 and HCX-DASH-002 reject `thinking` outright + // (live-verified: `40001 Invalid parameter: thinking`), even with tools. + for (const model of ["HCX-005", "HCX-DASH-002"]) { + const withTools = request.buildClovaPayload( + model, + { + messages: [{ role: "user", content: "hi" }], + tools: [{ type: "function", function: { name: "f" } }], + }, + true, + null + ); + assert.equal(withTools.thinking, undefined, `${model} must not receive thinking`); + assert.ok(Array.isArray(withTools.tools)); + + const askingForReasoning = request.buildClovaPayload( + model, + { messages: [{ role: "user", content: "hi" }], reasoning_effort: "high" }, + true, + null + ); + assert.equal(askingForReasoning.thinking, undefined, `${model} ignores reasoning_effort`); + } +}); + +test("clova v3: images are dropped in function-calling mode", () => { + const payload = request.buildClovaPayload( + "HCX-005", + { + messages: [ + { + role: "user", + content: [ + { type: "text", text: "describe" }, + { type: "image_url", image_url: { url: "https://example.com/a.png" } }, + ], + }, + ], + tools: [{ type: "function", function: { name: "f" } }], + }, + true, + null + ); + assert.deepEqual(payload.messages[0], { role: "user", content: "describe" }); + assert.ok(!JSON.stringify(payload).includes("imageUrl")); +}); + +test("clova v3: a tool result round-trips as role tool with toolCallId", () => { + const payload = request.buildClovaPayload( + "HCX-005", + { + messages: [ + { role: "user", content: "Weather in Seoul?" }, + { + role: "assistant", + content: "", + tool_calls: [ + { + id: "call_abc", + type: "function", + function: { name: "get_weather", arguments: '{"location":"Seoul"}' }, + }, + ], + }, + { role: "tool", tool_call_id: "call_abc", content: '{"temp":17}' }, + ], + tools: [{ type: "function", function: { name: "get_weather" } }], + }, + true, + null + ); + + assert.deepEqual(payload.messages[1], { + role: "assistant", + content: "", + toolCalls: [ + { + id: "call_abc", + type: "function", + function: { name: "get_weather", arguments: { location: "Seoul" } }, + }, + ], + }); + // CLOVA wants `arguments` as an object; OpenAI sends a JSON string. + assert.equal(typeof payload.messages[1].toolCalls[0].function.arguments, "object"); + + assert.deepEqual(payload.messages[2], { + role: "tool", + content: '{"temp":17}', + toolCallId: "call_abc", + }); +}); + +// --------------------------------------------------------------------------- +// Structured Outputs (v3-so) — HCX-007 only +// --------------------------------------------------------------------------- + +test("clova v3: json_schema maps onto responseFormat", () => { + const schema = { + type: "object", + properties: { temp_high_c: { type: "number" } }, + required: ["temp_high_c"], + }; + const payload = request.buildClovaPayload( + "HCX-007", + { + messages: [{ role: "user", content: "..." }], + response_format: { type: "json_schema", json_schema: { name: "weather", schema } }, + }, + true, + null + ); + assert.deepEqual(payload.responseFormat, { type: "json", schema }); + // Structured Outputs cannot be combined with reasoning (live-verified). + assert.deepEqual(payload.thinking, { effort: "none" }); +}); + +test("clova v3: structured outputs are dropped off the HCX-007-only path", () => { + // HCX-005 rejects `thinking` outright, so SO is unavailable there + // (live-verified: `40001 Invalid parameter: thinking`). + const payload = request.buildClovaPayload( + "HCX-005", + { + messages: [{ role: "user", content: "..." }], + response_format: { + type: "json_schema", + json_schema: { name: "x", schema: { type: "object" } }, + }, + }, + true, + null + ); + assert.equal(payload.responseFormat, undefined); + assert.equal(payload.thinking, undefined); +}); + +test("clova v3: function calling wins over structured outputs", () => { + const payload = request.buildClovaPayload( + "HCX-007", + { + messages: [{ role: "user", content: "..." }], + tools: [{ type: "function", function: { name: "f" } }], + response_format: { + type: "json_schema", + json_schema: { name: "x", schema: { type: "object" } }, + }, + }, + true, + null + ); + assert.ok(Array.isArray(payload.tools)); + assert.equal(payload.responseFormat, undefined); +}); + +test("clova v3: a public image URL maps to imageUrl.url", () => { + const payload = request.buildClovaPayload( + "HCX-005", + { + messages: [ + { + role: "user", + content: [ + { type: "text", text: "describe" }, + { type: "image_url", image_url: { url: "https://example.com/a.png" } }, + ], + }, + ], + }, + true, + null + ); + const parts = payload.messages[0].content; + assert.deepEqual(parts[1], { + type: "image_url", + imageUrl: { url: "https://example.com/a.png" }, + }); +}); + +test("clova v3: base64 images keep their full data-URI prefix in dataUri.data", () => { + // Regression guard: the prefix MUST survive. Sending only the base64 payload + // (prefix stripped) makes CLOVA reject the whole request with + // `40001 Invalid parameter`, while the complete data-URI string is accepted. + // Live-verified 2026-09-01 with PNG and JPEG at 16x16, 64x64 and full size. + const payload = request.buildClovaPayload( + "HCX-005", + { + messages: [ + { + role: "user", + content: [ + { type: "image_url", image_url: { url: "data:image/png;base64,AAAABBBB" } }, + { type: "text", text: "what is this" }, + ], + }, + ], + }, + true, + null + ); + assert.deepEqual(payload.messages[0].content[0], { + type: "image_url", + dataUri: { data: "data:image/png;base64,AAAABBBB" }, + }); + assert.deepEqual(payload.messages[0].content[1], { type: "text", text: "what is this" }); +}); + +test("clova v3: a data: image never leaks into imageUrl.url", () => { + const payload = request.buildClovaPayload( + "HCX-005", + { + messages: [ + { + role: "user", + content: [{ type: "image_url", image_url: { url: "data:image/jpeg;base64,ZZZZ" } }], + }, + ], + }, + true, + null + ); + const part = payload.messages[0].content[0]; + assert.equal(part.imageUrl, undefined); + assert.deepEqual(part.dataUri, { data: "data:image/jpeg;base64,ZZZZ" }); +}); + +test("clova v3: images are stripped for a text-only model", () => { + const payload = request.buildClovaPayload( + "HCX-DASH-002", + { + messages: [ + { + role: "user", + content: [ + { type: "image_url", image_url: { url: "https://example.com/a.png" } }, + { type: "text", text: "describe" }, + ], + }, + ], + }, + true, + null + ); + assert.deepEqual(payload.messages[0].content, [{ type: "text", text: "describe" }]); +}); + +// --------------------------------------------------------------------------- +// Reasoning model (HCX-007) contract +// --------------------------------------------------------------------------- + +test("clova v3: reasoning models use maxCompletionTokens, never maxTokens", () => { + // Live-verified: HCX-007 answers 40001 "Invalid parameter: maxTokens" when the + // cap is sent as `maxTokens`, and succeeds with `maxCompletionTokens`. + const withMaxTokens = request.buildClovaPayload( + "HCX-007", + { messages: [{ role: "user", content: "hi" }], max_tokens: 1024 }, + true, + null + ); + assert.equal(withMaxTokens.maxCompletionTokens, 1024); + assert.equal(withMaxTokens.maxTokens, undefined); + + const withMaxCompletion = request.buildClovaPayload( + "HCX-007", + { messages: [{ role: "user", content: "hi" }], max_completion_tokens: 2048 }, + true, + null + ); + assert.equal(withMaxCompletion.maxCompletionTokens, 2048); +}); + +test("clova v3: reasoning output cap is 32768, not the 4096 text-model cap", () => { + const payload = request.buildClovaPayload( + "HCX-007", + { messages: [{ role: "user", content: "hi" }], max_tokens: 999999 }, + true, + null + ); + assert.equal(payload.maxCompletionTokens, request.CLOVA_V3_REASONING_MAX_OUTPUT_TOKENS); + + const textModel = request.buildClovaPayload( + "HCX-005", + { messages: [{ role: "user", content: "hi" }], max_tokens: 999999 }, + true, + null + ); + assert.equal(textModel.maxTokens, request.CLOVA_V3_MAX_OUTPUT_TOKENS); +}); + +test("clova v3: stop is dropped for reasoning models", () => { + // The vendor docs state `stop` cannot be used while thinking. + const reasoning = request.buildClovaPayload( + "HCX-007", + { messages: [{ role: "user", content: "hi" }], stop: ["END"] }, + true, + null + ); + assert.equal(reasoning.stop, undefined); + + const text = request.buildClovaPayload( + "HCX-005", + { messages: [{ role: "user", content: "hi" }], stop: ["END"] }, + true, + null + ); + assert.deepEqual(text.stop, ["END"]); +}); + +test("clova v3: images are stripped for the reasoning model (HCX-007 has no vision)", () => { + const payload = request.buildClovaPayload( + "HCX-007", + { + messages: [ + { + role: "user", + content: [ + { type: "image_url", image_url: { url: "https://example.com/a.png" } }, + { type: "text", text: "describe" }, + ], + }, + ], + }, + true, + null + ); + assert.deepEqual(payload.messages[0].content, [{ type: "text", text: "describe" }]); +}); + +test("clova v3: reasoning_effort maps onto thinking.effort", () => { + assert.equal(request.toClovaThinkingEffort("low"), "low"); + assert.equal(request.toClovaThinkingEffort("high"), "high"); + // OpenAI's `minimal` has no CLOVA equivalent; `low` is the closest. + assert.equal(request.toClovaThinkingEffort("minimal"), "low"); + // Unknown values are omitted so CLOVA applies its own default. + assert.equal(request.toClovaThinkingEffort("bogus"), ""); + + const payload = request.buildClovaPayload( + "HCX-007", + { messages: [{ role: "user", content: "hi" }], reasoning_effort: "high" }, + true, + null + ); + assert.deepEqual(payload.thinking, { effort: "high" }); + + const noEffort = request.buildClovaPayload( + "HCX-007", + { messages: [{ role: "user", content: "hi" }] }, + true, + null + ); + assert.equal(noEffort.thinking, undefined); + + // Non-reasoning models must never receive the thinking envelope. + const textModel = request.buildClovaPayload( + "HCX-005", + { messages: [{ role: "user", content: "hi" }], reasoning_effort: "high" }, + true, + null + ); + assert.equal(textModel.thinking, undefined); +}); + +// --------------------------------------------------------------------------- +// Response: CLOVA v3 → OpenAI +// --------------------------------------------------------------------------- + +function tokenFrame(text: string): string { + return ( + `id: aabb\n` + + `event: token\n` + + `data: ${JSON.stringify({ message: { role: "assistant", content: text }, finishReason: null, created: 1 })}\n\n` + ); +} + +function resultFrame(fullText: string): string { + return ( + `id: aabb\n` + + `event: result\n` + + `data: ${JSON.stringify({ + message: { role: "assistant", content: fullText }, + finishReason: "stop", + created: 1, + usage: { promptTokens: 20, completionTokens: 5, totalTokens: 25 }, + })}\n\n` + ); +} + +test("clova v3: a token frame emits an incremental delta", () => { + const state = {}; + const chunk = response.convertClovaToOpenAI(tokenFrame("안"), state); + assert.equal(chunk.choices[0].delta.content, "안"); + // First chunk carries the assistant role, per OpenAI semantics. + assert.equal(chunk.choices[0].delta.role, "assistant"); + assert.equal(chunk.choices[0].finish_reason, null); +}); + +test("clova v3: the result frame does NOT repeat the already-streamed text", () => { + const state = {}; + response.convertClovaToOpenAI(tokenFrame("안"), state); + response.convertClovaToOpenAI(tokenFrame("녕"), state); + const terminal = response.convertClovaToOpenAI(resultFrame("안녕"), state); + + // The snapshot text must not be re-emitted — this is the duplication guard. + assert.equal(terminal.choices[0].delta.content, undefined); + assert.deepEqual(terminal.choices[0].delta, {}); + assert.equal(terminal.choices[0].finish_reason, "stop"); +}); + +test("clova v3: a full token→result stream yields the answer exactly once", () => { + const state = {}; + const frames = [tokenFrame("안"), tokenFrame("녕"), resultFrame("안녕")]; + const text = frames + .map((frame) => response.convertClovaToOpenAI(frame, state)) + .filter(Boolean) + .map((chunk) => chunk.choices?.[0]?.delta?.content ?? "") + .join(""); + + assert.equal(text, "안녕"); + assert.notEqual(text, "안녕안녕"); + assert.deepEqual(state.usage, { + prompt_tokens: 20, + completion_tokens: 5, + total_tokens: 25, + }); +}); + +test("clova v3: an upstream status failure surfaces as state.upstreamError", () => { + const state = {}; + const frame = + `id: aabb\n` + + `event: error\n` + + `data: ${JSON.stringify({ status: { code: "40100", message: "Invalid API key" } })}\n\n`; + + assert.equal(response.convertClovaToOpenAI(frame, state), null); + assert.equal(state.upstreamError.status, 400); + assert.match(state.upstreamError.message, /Invalid API key/); +}); + +test("clova v3: a 5xxxx status maps to a 502 upstream error", () => { + const state = {}; + const payload = { + status: { code: "50000", message: "Internal Server Error" }, + result: null, + }; + assert.equal(response.convertClovaToOpenAI(payload, state), null); + assert.equal(state.upstreamError.status, 502); +}); + +test("clova v3: a non-stream envelope replays its text once, then terminates", () => { + const state = {}; + const out = response.convertClovaToOpenAI( + { + status: { code: "20000", message: "OK" }, + result: { + message: { role: "assistant", content: "hello" }, + usage: { promptTokens: 1, completionTokens: 2, totalTokens: 3 }, + finishReason: "stop", + }, + }, + state + ); + + assert.ok(Array.isArray(out)); + assert.equal(out[0].choices[0].delta.content, "hello"); + assert.equal(out[1].choices[0].finish_reason, "stop"); + assert.equal(state.usage.total_tokens, 3); +}); + +test("clova v3: thinkingContent is emitted as reasoning_content", () => { + const state = {}; + const frame = + `id: aabb\n` + + `event: token\n` + + `data: ${JSON.stringify({ message: { role: "assistant", thinkingContent: "생각" }, finishReason: null })}\n\n`; + + const chunk = response.convertClovaToOpenAI(frame, state); + assert.equal(chunk.choices[0].delta.reasoning_content, "생각"); + assert.equal(chunk.choices[0].delta.content, undefined); +}); + +test("clova v3: reasoning and answer deltas stay on separate delta keys", () => { + const state = {}; + const thinking = response.convertClovaToOpenAI( + `event: token\ndata: ${JSON.stringify({ message: { thinkingContent: "because" } })}\n\n`, + state + ); + const answer = response.convertClovaToOpenAI( + `event: token\ndata: ${JSON.stringify({ message: { content: "391" } })}\n\n`, + state + ); + + assert.equal(thinking.choices[0].delta.reasoning_content, "because"); + assert.equal(answer.choices[0].delta.content, "391"); + assert.equal(answer.choices[0].delta.reasoning_content, undefined); +}); + +test("clova v3: a tool-call stream assembles partialJson fragments", () => { + const state = {}; + const frame = (data: unknown, event = "token") => + `id: x\nevent: ${event}\ndata: ${JSON.stringify(data)}\n\n`; + + // First frame carries id + name; the rest carry only JSON fragments. + const start = response.convertClovaToOpenAI( + frame({ + message: { + role: "assistant", + content: "", + toolCalls: [{ id: "call_abc", type: "function", function: { name: "get_weather" } }], + }, + finishReason: null, + }), + state + ); + assert.deepEqual(start.choices[0].delta.tool_calls, [ + { + index: 0, + id: "call_abc", + type: "function", + function: { name: "get_weather", arguments: "" }, + }, + ]); + + // Fragment order is taken verbatim from a live HCX-005 function-calling + // stream — note the space after the colon, which CLOVA emits as its own chunk. + let args = ""; + for (const fragment of ['{"', "location", '":', ' "', "Se", "oul", '"}']) { + const chunk = response.convertClovaToOpenAI( + frame({ + message: { + role: "assistant", + content: "", + toolCalls: [{ type: "function", function: { partialJson: fragment } }], + }, + finishReason: null, + }), + state + ); + args += chunk.choices[0].delta.tool_calls[0].function.arguments; + } + assert.equal(args, '{"location": "Seoul"}'); + assert.deepEqual(JSON.parse(args), { location: "Seoul" }); +}); + +test("clova v3: the terminal frame reports tool_calls without repeating the call", () => { + const state = {}; + response.convertClovaToOpenAI( + `id: x\nevent: token\ndata: ${JSON.stringify({ message: { content: "", toolCalls: [{ id: "call_abc", type: "function", function: { name: "get_weather" } }] }, finishReason: null })}\n\n`, + state + ); + + const terminal = response.convertClovaToOpenAI( + `id: x\nevent: result\ndata: ${JSON.stringify({ + message: { + role: "assistant", + content: "", + toolCalls: [ + { + id: "call_abc", + type: "function", + function: { name: "get_weather", arguments: { location: "Seoul" } }, + }, + ], + }, + finishReason: "tool_calls", + usage: { promptTokens: 9, completionTokens: 47, totalTokens: 56 }, + })}\n\n`, + state + ); + + // The finished call is a snapshot — it must not be emitted a second time. + assert.equal(terminal.choices[0].delta.tool_calls, undefined); + assert.deepEqual(terminal.choices[0].delta, {}); + assert.equal(terminal.choices[0].finish_reason, "tool_calls"); + assert.equal(terminal.usage.total_tokens, 56); +}); + +test("clova v3: the flush signal and unparseable frames return null", () => { + const state = {}; + assert.equal(response.convertClovaToOpenAI(null, state), null); + assert.equal(response.convertClovaToOpenAI("id: aabb\nevent: ping\ndata: \n\n", state), null); + assert.equal(response.convertClovaToOpenAI("not json at all", state), null); +}); + +test("clova v3: an unknown event type is ignored", () => { + const state = {}; + const frame = `event: signal\ndata: ${JSON.stringify({ data: "keepalive" })}\n\n`; + assert.equal(response.convertClovaToOpenAI(frame, state), null); +}); From 6795783228ca030f96cdb6ca596abe70f2f9a734 Mon Sep 17 00:00:00 2001 From: backryun Date: Thu, 3 Sep 2026 19:50:41 +0900 Subject: [PATCH 019/143] feat(providers): refresh Fable, Cursor, and Devin catalogs (#12367) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com #12524, #12538, #12277 e #12367 sobre o tip de release/v3.8.51: os quatro boardaram sem conflito (áreas disjuntas — zai-web, nvidia, clova, cursor/devin/fable). typecheck:core limpo, check:provider-consistency OK (272 entradas REGISTRY, 355 providers canônicos), check:known-symbols OK, e 305/305 nos testes tocados pelos quatro PRs. Os IDs de modelo adicionados foram conferidos individualmente. Obrigado, @backryun. --- config/quality/eslint-suppressions.json | 24 +- .../config/claudeCodeCompatibleIdentity.ts | 8 +- open-sse/config/context1m.ts | 7 +- .../providers/registry/anthropic/index.ts | 11 + .../providers/registry/bedrock/index.ts | 11 + .../config/providers/registry/claude/index.ts | 11 + .../providers/registry/claude/web/index.ts | 11 + .../config/providers/registry/cursor/index.ts | 488 +++++++------ .../providers/registry/devin/catalog.ts | 257 ++++--- .../config/providers/registry/vertex/index.ts | 1 + .../registry/vertex/partner/index.ts | 1 + .../__tests__/scoresAs-11489.test.ts | 8 +- open-sse/services/modelFamilyFallback.ts | 5 +- open-sse/services/providerCostData.ts | 23 +- open-sse/utils/cursorAgentProtobuf.ts | 659 +++++++++--------- .../requestedModelParameters.ts | 113 +++ open-sse/utils/registeredEffortVariants.ts | 15 +- src/lib/db/models/activeSyncedCatalog.ts | 32 +- src/lib/providerModels/cursorAutoCatalog.ts | 49 +- .../providerModels/cursorAvailableModels.ts | 133 ++-- src/lib/providers/staticModels.ts | 1 + src/shared/constants/cliTools.ts | 2 +- src/shared/constants/modelSpecs.ts | 55 +- .../constants/pricing/default-pricing.ts | 2 + src/shared/constants/pricing/devin.ts | 142 ++++ src/shared/constants/pricing/frontier-labs.ts | 2 + .../constants/pricing/oauth-subscriptions.ts | 2 + src/shared/constants/pricing/shared-tiers.ts | 8 + tests/unit/claude-fable-5-1.test.ts | 172 +++++ .../claude-web-sonnet5-registry-6209.test.ts | 1 + tests/unit/cursor-auto-catalog-entry.test.ts | 27 + tests/unit/cursor-available-models.test.ts | 2 + .../unit/cursor-catalog-combo-compat.test.ts | 4 +- .../cursor-model-effort-suffix-7289.test.ts | 32 + .../cursor-registry-claude-families.test.ts | 209 +++++- tests/unit/devin-cli-catalog.test.ts | 134 +++- .../fixtures/cursor-rewrite-failure-ids.ts | 7 +- .../guardrails/visionBridgeRouter.test.ts | 6 +- ...-model-catalog-reconciliation-8926.test.ts | 16 +- tests/unit/pricing-constants-split.test.ts | 5 +- tests/unit/provider-cost-data.test.ts | 48 ++ tests/unit/provider-models-config.test.ts | 12 +- 42 files changed, 1835 insertions(+), 921 deletions(-) create mode 100644 open-sse/utils/cursorAgentProtobuf/requestedModelParameters.ts create mode 100644 src/shared/constants/pricing/devin.ts create mode 100644 tests/unit/claude-fable-5-1.test.ts create mode 100644 tests/unit/provider-cost-data.test.ts diff --git a/config/quality/eslint-suppressions.json b/config/quality/eslint-suppressions.json index 90f949af47..e10d1f6551 100644 --- a/config/quality/eslint-suppressions.json +++ b/config/quality/eslint-suppressions.json @@ -220,11 +220,6 @@ "count": 26 } }, - "open-sse/handlers/chatCore/clientUsageBuffer.ts": { - "@typescript-eslint/no-unused-vars": { - "count": 1 - } - }, "open-sse/handlers/chatCore/executorHelpers.ts": { "@typescript-eslint/no-unused-vars": { "count": 1 @@ -608,11 +603,6 @@ "count": 1 } }, - "open-sse/services/providerCostData.ts": { - "@typescript-eslint/no-unused-vars": { - "count": 1 - } - }, "open-sse/services/rateLimitManager.ts": { "@typescript-eslint/no-unused-vars": { "count": 2 @@ -763,7 +753,7 @@ }, "open-sse/utils/cursorAgentProtobuf.ts": { "@typescript-eslint/no-unused-vars": { - "count": 3 + "count": 2 } }, "open-sse/utils/earlyStreamKeepalive.ts": { @@ -1722,16 +1712,6 @@ "count": 2 } }, - "src/lib/oneproxyRotator.ts": { - "@typescript-eslint/no-unused-vars": { - "count": 1 - } - }, - "src/lib/oneproxySync.ts": { - "@typescript-eslint/no-unused-vars": { - "count": 1 - } - }, "src/lib/piiSanitizer.ts": { "@typescript-eslint/no-unused-vars": { "count": 1 @@ -4812,7 +4792,7 @@ }, "tests/unit/responses-translation-fixes.test.ts": { "@typescript-eslint/no-explicit-any": { - "count": 35 + "count": 34 } }, "tests/unit/route-edge-coverage.test.ts": { diff --git a/open-sse/config/claudeCodeCompatibleIdentity.ts b/open-sse/config/claudeCodeCompatibleIdentity.ts index 31b7997f2a..b614eea324 100644 --- a/open-sse/config/claudeCodeCompatibleIdentity.ts +++ b/open-sse/config/claudeCodeCompatibleIdentity.ts @@ -9,15 +9,19 @@ export const CLAUDE_CODE_COMPATIBLE_VERSION = CLAUDE_CODE_CLIENT_VERSION; export const CLAUDE_CODE_COMPATIBLE_USER_AGENT = getClaudeCodeUserAgent("sdk-cli"); export const CLAUDE_CODE_COMPATIBLE_STAINLESS_PACKAGE_VERSION = CLAUDE_CODE_SDK_PACKAGE_VERSION; export const CLAUDE_CODE_COMPATIBLE_STAINLESS_RUNTIME_VERSION = CLAUDE_CODE_RUNTIME_VERSION; -const CONTEXT_1M_NATIVE_MODELS = ["claude-opus-5"]; +const CONTEXT_1M_NATIVE_MODELS = ["claude-fable-5-1", "claude-opus-5"]; export function modelHasNativeContext1m(model: string | null | undefined): boolean { const normalizedModel = String(model || "") .trim() .toLowerCase() + .replace(/^.*?(?=claude-)/, "") .replace(/-\d{8}$/, ""); return CONTEXT_1M_NATIVE_MODELS.some( - (supported) => normalizedModel === supported || normalizedModel.startsWith(`${supported}-`) + (supported) => + normalizedModel === supported || + (normalizedModel.startsWith(`${supported}-`) && + !/^\d/.test(normalizedModel.slice(supported.length + 1))) ); } diff --git a/open-sse/config/context1m.ts b/open-sse/config/context1m.ts index dab5c405b1..134df22bf8 100644 --- a/open-sse/config/context1m.ts +++ b/open-sse/config/context1m.ts @@ -34,6 +34,9 @@ export function modelSupportsContext1mBeta(model: string | null | undefined): bo .replace(/-\d{8}$/, ""); return CONTEXT_1M_SUPPORTED_MODELS.some( - (supported) => normalizedModel === supported || normalizedModel.startsWith(`${supported}-`) + (supported) => + normalizedModel === supported || + (normalizedModel.startsWith(`${supported}-`) && + !/^\d/.test(normalizedModel.slice(supported.length + 1))) ); -} \ No newline at end of file +} diff --git a/open-sse/config/providers/registry/anthropic/index.ts b/open-sse/config/providers/registry/anthropic/index.ts index 2a3726276d..2845296b15 100644 --- a/open-sse/config/providers/registry/anthropic/index.ts +++ b/open-sse/config/providers/registry/anthropic/index.ts @@ -16,6 +16,17 @@ export const anthropicProvider: RegistryEntry = { "Anthropic-Beta": ANTHROPIC_BETA_API_KEY, }, models: [ + { + id: "claude-fable-5-1", + name: "Claude Fable 5.1", + contextLength: 1000000, + maxOutputTokens: 128000, + supportsReasoning: true, + supportedThinkingEfforts: ["low", "medium", "high", "xhigh", "max"], + supportsXHighEffort: true, + supportsVision: true, + unsupportedParams: ["temperature", "top_p", "top_k"], + }, { id: "claude-fable-5", name: "Claude Fable 5", diff --git a/open-sse/config/providers/registry/bedrock/index.ts b/open-sse/config/providers/registry/bedrock/index.ts index bb9273fda0..ce70091a27 100644 --- a/open-sse/config/providers/registry/bedrock/index.ts +++ b/open-sse/config/providers/registry/bedrock/index.ts @@ -9,6 +9,17 @@ export const bedrockProvider: RegistryEntry = { authHeader: "bearer", defaultContextLength: 200000, models: [ + { + id: "anthropic.claude-fable-5-1", + name: "Claude Fable 5.1 (Bedrock)", + toolCalling: true, + supportsReasoning: true, + supportedThinkingEfforts: ["low", "medium", "high", "xhigh", "max"], + supportsXHighEffort: true, + supportsVision: true, + contextLength: 1000000, + maxOutputTokens: 128000, + }, { id: "anthropic.claude-sonnet-4-6", name: "Claude Sonnet 4.6 (Bedrock)", diff --git a/open-sse/config/providers/registry/claude/index.ts b/open-sse/config/providers/registry/claude/index.ts index 481c831c2f..ba3129678b 100644 --- a/open-sse/config/providers/registry/claude/index.ts +++ b/open-sse/config/providers/registry/claude/index.ts @@ -28,6 +28,17 @@ export const claudeProvider: RegistryEntry = { tokenUrl: "https://api.anthropic.com/v1/oauth/token", }, models: [ + { + id: "claude-fable-5-1", + name: "Claude Fable 5.1", + contextLength: 1000000, + maxOutputTokens: 128000, + supportsReasoning: true, + supportedThinkingEfforts: ["low", "medium", "high", "xhigh", "max"], + supportsXHighEffort: true, + supportsVision: true, + unsupportedParams: ["temperature", "top_p", "top_k"], + }, { id: "claude-fable-5", name: "Claude Fable 5", diff --git a/open-sse/config/providers/registry/claude/web/index.ts b/open-sse/config/providers/registry/claude/web/index.ts index c701ff9aa3..07c4872362 100644 --- a/open-sse/config/providers/registry/claude/web/index.ts +++ b/open-sse/config/providers/registry/claude/web/index.ts @@ -9,6 +9,17 @@ export const claude_webProvider: RegistryEntry = { authType: "apikey", authHeader: "cookie", models: [ + { + id: "claude-fable-5-1", + name: "Claude Fable 5.1 (web)", + toolCalling: false, + supportsReasoning: true, + supportedThinkingEfforts: ["low", "medium", "high", "xhigh", "max"], + supportsXHighEffort: true, + supportsVision: true, + contextLength: 1000000, + maxOutputTokens: 128000, + }, { id: "claude-fable-5", name: "Claude Fable 5 (web)", toolCalling: false }, { id: "claude-opus-5", diff --git a/open-sse/config/providers/registry/cursor/index.ts b/open-sse/config/providers/registry/cursor/index.ts index cc5a78f8ce..4b5b3a86de 100644 --- a/open-sse/config/providers/registry/cursor/index.ts +++ b/open-sse/config/providers/registry/cursor/index.ts @@ -1,6 +1,37 @@ -import type { RegistryEntry } from "../../shared.ts"; +import type { RegistryEntry, RegistryModel } from "../../shared.ts"; import { CURSOR_REGISTRY_VERSION, getCursorRegistryHeaders } from "../../shared.ts"; +const CLAUDE_FABLE_5_1_CAPABILITIES = { + maxOutputTokens: 128_000, +} as const; + +const ONE_MILLION_CONTEXT = 1_000_000; + +function withOneMillionContext( + models: RegistryModel[], + familyName: string, + defaultContextLength: number, + liveCatalogId: string, + supportsOneMillion: (model: RegistryModel) => boolean = () => true +): RegistryModel[] { + return models.flatMap((model) => { + const defaultContextModel = { + ...model, + contextLength: defaultContextLength, + liveCatalogIds: model.liveCatalogIds ?? [liveCatalogId], + ...(familyName.startsWith("GPT-") ? {} : { scoresAs: model.scoresAs ?? liveCatalogId }), + }; + if (!supportsOneMillion(model)) return [defaultContextModel]; + const oneMillionModel = { + ...defaultContextModel, + id: `${model.id}-1m`, + name: model.name.replace(familyName, `${familyName} 1M`), + contextLength: ONE_MILLION_CONTEXT, + }; + return [oneMillionModel, defaultContextModel]; + }); +} + export const cursorProvider: RegistryEntry = { id: "cursor", alias: "cu", @@ -18,259 +49,214 @@ export const cursorProvider: RegistryEntry = { { id: "auto-cost", name: "Auto (cost)" }, { id: "auto-balance", name: "Auto (balance)" }, { id: "auto-intelligence", name: "Auto (intelligence)" }, - // Legacy combo ids kept so existing cu/ targets are not orphaned. - { id: "composer-2", name: "Composer 2" }, - { id: "composer-2-fast", name: "Composer 2 Fast" }, - { id: "gpt-5.4-low-fast", name: "GPT 5.4 Low Fast" }, - { id: "gpt-5.3-codex-spark-preview-low", name: "GPT 5.3 Codex Spark Preview Low" }, - { id: "gpt-5.3-codex-spark-preview", name: "GPT 5.3 Codex Spark Preview" }, - { id: "gpt-5.3-codex-spark-preview-high", name: "GPT 5.3 Codex Spark Preview High" }, - { id: "gpt-5.3-codex-spark-preview-xhigh", name: "GPT 5.3 Codex Spark Preview XHigh" }, - // #11489: cursor/agy spell Claude ids - ("claude-4.6-opus-high"); - // the effort splitter strips those to "claude-4.6-opus", which is not a catalog id. - // `scoresAs` points each at the canonical - spelling so quality - // scores are inherited. Operational metadata stays on these entries. - { - id: "claude-4.6-opus-high-thinking-fast", - name: "Claude 4.6 Opus High Thinking Fast", - scoresAs: "claude-opus-4-6", - }, - { - id: "claude-4.6-opus-max-thinking-fast", - name: "Claude 4.6 Opus Max Thinking Fast", - scoresAs: "claude-opus-4-6", - }, - { - id: "claude-4.6-sonnet-medium", - name: "Claude 4.6 Sonnet Medium", - scoresAs: "claude-sonnet-4-6", - }, - { - id: "claude-4.6-sonnet-medium-thinking", - name: "Claude 4.6 Sonnet Medium Thinking", - scoresAs: "claude-sonnet-4-6", - }, - { id: "gemini-3.1-pro", name: "Gemini 3.1 Pro" }, - { id: "gemini-3.7-flash", name: "Gemini 3.7 Flash" }, - { id: "gemini-3-flash", name: "Gemini 3 Flash" }, - { id: "grok-4.6-medium", name: "Grok 4.6 Medium" }, - { id: "grok-4.6-fast-medium", name: "Grok 4.6 Fast Medium" }, - { id: "grok-4.6-high", name: "Grok 4.6 High" }, - { id: "grok-4.6-fast-high", name: "Grok 4.6 Fast High" }, - { id: "grok-4.6-xhigh", name: "Grok 4.6 XHigh" }, - { id: "grok-4.6-fast-xhigh", name: "Grok 4.6 Fast XHigh" }, - { id: "kimi-k3", name: "Kimi K3" }, - { id: "kimi-k2.7-code", name: "Kimi K2.7 Code" }, - { id: "grok-4.3", name: "Grok 4.3" }, - { id: "grok-4.5-medium", name: "Grok 4.5 Medium" }, - { id: "grok-4.5-fast-medium", name: "Grok 4.5 Fast Medium" }, - { id: "grok-4.5-high", name: "Grok 4.5 High" }, - { id: "grok-4.5-fast-high", name: "Grok 4.5 Fast High" }, - { id: "grok-4.5-xhigh", name: "Grok 4.5 XHigh" }, - { id: "grok-4.5-fast-xhigh", name: "Grok 4.5 Fast XHigh" }, - { id: "kimi-k2.5", name: "Kimi K2.5" }, - { id: "gpt-5.3-codex-low", name: "Codex 5.3 Low" }, - { id: "gpt-5.3-codex-low-fast", name: "Codex 5.3 Low Fast" }, - { id: "gpt-5.3-codex", name: "Codex 5.3" }, - { id: "gpt-5.3-codex-fast", name: "Codex 5.3 Fast" }, - { id: "gpt-5.3-codex-high", name: "Codex 5.3 High" }, - { id: "gpt-5.3-codex-high-fast", name: "Codex 5.3 High Fast" }, - { id: "gpt-5.3-codex-xhigh", name: "Codex 5.3 Extra High" }, - { id: "gpt-5.3-codex-xhigh-fast", name: "Codex 5.3 Extra High Fast" }, - { id: "gpt-5.2", name: "GPT-5.2" }, - { id: "cursor-grok-4.5-high", name: "Cursor Grok 4.5" }, - { id: "cursor-grok-4.5-high-fast", name: "Cursor Grok 4.5 Fast" }, - { id: "composer-2.5", name: "Composer 2.5" }, - { id: "claude-opus-5-thinking-high", name: "Opus 5 1M Thinking" }, - { id: "claude-opus-5-thinking-high-fast", name: "Opus 5 1M Thinking Fast" }, - { id: "claude-opus-5-thinking-xhigh", name: "Opus 5 1M Extra High Thinking" }, - { id: "claude-opus-5-thinking-xhigh-fast", name: "Opus 5 1M Extra High Thinking Fast" }, - { id: "claude-opus-4-8-thinking-high", name: "Opus 4.8 1M Thinking" }, - { id: "claude-opus-4-8-thinking-high-fast", name: "Opus 4.8 1M Thinking Fast" }, - { id: "gpt-5.6-sol-high", name: "GPT-5.6 Sol 1M High" }, - { id: "gpt-5.6-sol-high-fast", name: "GPT-5.6 Sol High Fast" }, - { id: "gpt-5.6-sol-xhigh", name: "GPT-5.6 Sol 1M Extra High" }, - { id: "gpt-5.6-sol-xhigh-fast", name: "GPT-5.6 Sol Extra High Fast" }, - { id: "gpt-5.5-high", name: "GPT-5.5 1M High" }, - { id: "gpt-5.5-high-fast", name: "GPT-5.5 High Fast" }, - { id: "claude-fable-5-thinking-high", name: "Fable 5 1M Thinking (NO ZDR)" }, - { id: "claude-fable-5-thinking-xhigh", name: "Fable 5 1M Extra High Thinking (NO ZDR)" }, - { id: "claude-sonnet-5-thinking-high", name: "Sonnet 5 1M Thinking" }, - { id: "claude-sonnet-5-thinking-xhigh", name: "Sonnet 5 1M Extra High Thinking" }, - { id: "kimi-k3-high", name: "Kimi K3 High" }, - { id: "cursor-grok-4.5-low", name: "Cursor Grok 4.5 Low" }, - { id: "cursor-grok-4.5-low-fast", name: "Cursor Grok 4.5 Low Fast" }, - { id: "cursor-grok-4.5-medium", name: "Cursor Grok 4.5 Medium" }, - { id: "cursor-grok-4.5-medium-fast", name: "Cursor Grok 4.5 Medium Fast" }, + { id: "cursor-grok-4.6-xhigh-fast", name: "Cursor Grok 4.6 Xhigh Fast" }, + { id: "cursor-grok-4.6-xhigh", name: "Cursor Grok 4.6 Xhigh" }, + { id: "cursor-grok-4.6-high-fast", name: "Cursor Grok 4.6 High Fast" }, + { id: "cursor-grok-4.6-high", name: "Cursor Grok 4.6 High" }, + { id: "cursor-grok-4.6-medium-fast", name: "Cursor Grok 4.6 Medium Fast" }, + { id: "cursor-grok-4.6-medium", name: "Cursor Grok 4.6 Medium" }, + { id: "cursor-grok-4.6-low-fast", name: "Cursor Grok 4.6 Low Fast" }, + { id: "cursor-grok-4.6-low", name: "Cursor Grok 4.6 Low" }, { id: "composer-2.5-fast", name: "Composer 2.5 Fast" }, - { id: "claude-opus-5-low", name: "Opus 5 1M Low" }, - { id: "claude-opus-5-low-fast", name: "Opus 5 1M Low Fast" }, - { id: "claude-opus-5-medium", name: "Opus 5 1M Medium" }, - { id: "claude-opus-5-medium-fast", name: "Opus 5 1M Medium Fast" }, - { id: "claude-opus-5-high", name: "Opus 5 1M" }, - { id: "claude-opus-5-high-fast", name: "Opus 5 1M Fast" }, - { id: "claude-opus-5-thinking-low", name: "Opus 5 1M Low Thinking" }, - { id: "claude-opus-5-thinking-low-fast", name: "Opus 5 1M Low Thinking Fast" }, - { id: "claude-opus-5-thinking-medium", name: "Opus 5 1M Medium Thinking" }, - { id: "claude-opus-5-thinking-medium-fast", name: "Opus 5 1M Medium Thinking Fast" }, - { id: "claude-opus-5-thinking-max", name: "Opus 5 1M Max Thinking" }, - { id: "claude-opus-5-thinking-max-fast", name: "Opus 5 1M Max Thinking Fast" }, - { id: "claude-opus-4-8-low", name: "Opus 4.8 1M Low" }, - { id: "claude-opus-4-8-low-fast", name: "Opus 4.8 1M Low Fast" }, - { id: "claude-opus-4-8-medium", name: "Opus 4.8 1M Medium" }, - { id: "claude-opus-4-8-medium-fast", name: "Opus 4.8 1M Medium Fast" }, - { id: "claude-opus-4-8-high", name: "Opus 4.8 1M" }, - { id: "claude-opus-4-8-high-fast", name: "Opus 4.8 1M Fast" }, - { id: "claude-opus-4-8-xhigh", name: "Opus 4.8 1M Extra High" }, - { id: "claude-opus-4-8-xhigh-fast", name: "Opus 4.8 1M Extra High Fast" }, - { id: "claude-opus-4-8-max", name: "Opus 4.8 1M Max" }, - { id: "claude-opus-4-8-max-fast", name: "Opus 4.8 1M Max Fast" }, - { id: "claude-opus-4-8-thinking-low", name: "Opus 4.8 1M Low Thinking" }, - { id: "claude-opus-4-8-thinking-low-fast", name: "Opus 4.8 1M Low Thinking Fast" }, - { id: "claude-opus-4-8-thinking-medium", name: "Opus 4.8 1M Medium Thinking" }, - { id: "claude-opus-4-8-thinking-medium-fast", name: "Opus 4.8 1M Medium Thinking Fast" }, - { id: "claude-opus-4-8-thinking-xhigh", name: "Opus 4.8 1M Extra High Thinking" }, - { id: "claude-opus-4-8-thinking-xhigh-fast", name: "Opus 4.8 1M Extra High Thinking Fast" }, - { id: "claude-opus-4-8-thinking-max", name: "Opus 4.8 1M Max Thinking" }, - { id: "claude-opus-4-8-thinking-max-fast", name: "Opus 4.8 1M Max Thinking Fast" }, - { id: "gpt-5.6-sol-none", name: "GPT-5.6 Sol 1M None" }, - { id: "gpt-5.6-sol-none-fast", name: "GPT-5.6 Sol None Fast" }, - { id: "gpt-5.6-sol-low", name: "GPT-5.6 Sol 1M Low" }, - { id: "gpt-5.6-sol-low-fast", name: "GPT-5.6 Sol Low Fast" }, - { id: "gpt-5.6-sol-medium", name: "GPT-5.6 Sol 1M" }, - { id: "gpt-5.6-sol-medium-fast", name: "GPT-5.6 Sol Fast" }, - { id: "gpt-5.6-sol-max", name: "GPT-5.6 Sol 1M Max" }, - { id: "gpt-5.6-sol-max-fast", name: "GPT-5.6 Sol Max Fast" }, - { id: "gpt-5.5-none", name: "GPT-5.5 1M None" }, - { id: "gpt-5.5-none-fast", name: "GPT-5.5 None Fast" }, - { id: "gpt-5.5-low", name: "GPT-5.5 1M Low" }, - { id: "gpt-5.5-low-fast", name: "GPT-5.5 Low Fast" }, - { id: "gpt-5.5-medium", name: "GPT-5.5 1M" }, - { id: "gpt-5.5-medium-fast", name: "GPT-5.5 Fast" }, - { id: "gpt-5.5-extra-high", name: "GPT-5.5 1M Extra High" }, - { id: "gpt-5.5-extra-high-fast", name: "GPT-5.5 Extra High Fast" }, - { id: "claude-fable-5-low", name: "Fable 5 1M Low (NO ZDR)" }, - { id: "claude-fable-5-medium", name: "Fable 5 1M Medium (NO ZDR)" }, - { id: "claude-fable-5-high", name: "Fable 5 1M (NO ZDR)" }, - { id: "claude-fable-5-xhigh", name: "Fable 5 1M Extra High (NO ZDR)" }, - { id: "claude-fable-5-max", name: "Fable 5 1M Max (NO ZDR)" }, - { id: "claude-fable-5-thinking-low", name: "Fable 5 1M Low Thinking (NO ZDR)" }, - { id: "claude-fable-5-thinking-medium", name: "Fable 5 1M Medium Thinking (NO ZDR)" }, - { id: "claude-fable-5-thinking-max", name: "Fable 5 1M Max Thinking (NO ZDR)" }, - { id: "claude-sonnet-5-low", name: "Sonnet 5 1M Low" }, - { id: "claude-sonnet-5-medium", name: "Sonnet 5 1M Medium" }, - { id: "claude-sonnet-5-high", name: "Sonnet 5 1M" }, - { id: "claude-sonnet-5-xhigh", name: "Sonnet 5 1M Extra High" }, - { id: "claude-sonnet-5-max", name: "Sonnet 5 1M Max" }, - { id: "claude-sonnet-5-thinking-low", name: "Sonnet 5 1M Low Thinking" }, - { id: "claude-sonnet-5-thinking-medium", name: "Sonnet 5 1M Medium Thinking" }, - { id: "claude-sonnet-5-thinking-max", name: "Sonnet 5 1M Max Thinking" }, - { id: "gpt-5.6-terra-none", name: "GPT-5.6 Terra 1M None" }, - { id: "gpt-5.6-terra-none-fast", name: "GPT-5.6 Terra None Fast" }, - { id: "gpt-5.6-terra-low", name: "GPT-5.6 Terra 1M Low" }, - { id: "gpt-5.6-terra-low-fast", name: "GPT-5.6 Terra Low Fast" }, - { id: "gpt-5.6-terra-medium", name: "GPT-5.6 Terra 1M" }, - { id: "gpt-5.6-terra-medium-fast", name: "GPT-5.6 Terra Fast" }, - { id: "gpt-5.6-terra-high", name: "GPT-5.6 Terra 1M High" }, - { id: "gpt-5.6-terra-high-fast", name: "GPT-5.6 Terra High Fast" }, - { id: "gpt-5.6-terra-xhigh", name: "GPT-5.6 Terra 1M Extra High" }, - { id: "gpt-5.6-terra-xhigh-fast", name: "GPT-5.6 Terra Extra High Fast" }, - { id: "gpt-5.6-terra-max", name: "GPT-5.6 Terra 1M Max" }, - { id: "gpt-5.6-terra-max-fast", name: "GPT-5.6 Terra Max Fast" }, - { id: "claude-opus-4-7-low", name: "Opus 4.7 1M Low" }, - { id: "claude-opus-4-7-low-fast", name: "Opus 4.7 1M Low Fast" }, - { id: "claude-opus-4-7-medium", name: "Opus 4.7 1M Medium" }, - { id: "claude-opus-4-7-medium-fast", name: "Opus 4.7 1M Medium Fast" }, - { id: "claude-opus-4-7-high", name: "Opus 4.7 1M High" }, - { id: "claude-opus-4-7-high-fast", name: "Opus 4.7 1M High Fast" }, - { id: "claude-opus-4-7-xhigh", name: "Opus 4.7 1M" }, - { id: "claude-opus-4-7-xhigh-fast", name: "Opus 4.7 1M Fast" }, - { id: "claude-opus-4-7-max", name: "Opus 4.7 1M Max" }, - { id: "claude-opus-4-7-max-fast", name: "Opus 4.7 1M Max Fast" }, - { id: "claude-opus-4-7-thinking-low", name: "Opus 4.7 1M Low Thinking" }, - { id: "claude-opus-4-7-thinking-low-fast", name: "Opus 4.7 1M Low Thinking Fast" }, - { id: "claude-opus-4-7-thinking-medium", name: "Opus 4.7 1M Medium Thinking" }, - { id: "claude-opus-4-7-thinking-medium-fast", name: "Opus 4.7 1M Medium Thinking Fast" }, - { id: "claude-opus-4-7-thinking-high", name: "Opus 4.7 1M High Thinking" }, - { id: "claude-opus-4-7-thinking-high-fast", name: "Opus 4.7 1M High Thinking Fast" }, - { id: "claude-opus-4-7-thinking-xhigh", name: "Opus 4.7 1M Thinking" }, - { id: "claude-opus-4-7-thinking-xhigh-fast", name: "Opus 4.7 1M Thinking Fast" }, - { id: "claude-opus-4-7-thinking-max", name: "Opus 4.7 1M Max Thinking" }, - { id: "claude-opus-4-7-thinking-max-fast", name: "Opus 4.7 1M Max Thinking Fast" }, - { id: "gpt-5.4-low", name: "GPT-5.4 1M Low" }, - { id: "gpt-5.4-medium", name: "GPT-5.4 1M" }, - { id: "gpt-5.4-medium-fast", name: "GPT-5.4 Fast" }, - { id: "gpt-5.4-high", name: "GPT-5.4 1M High" }, - { id: "gpt-5.4-high-fast", name: "GPT-5.4 High Fast" }, - { id: "gpt-5.4-xhigh", name: "GPT-5.4 1M Extra High" }, - { id: "gpt-5.4-xhigh-fast", name: "GPT-5.4 Extra High Fast" }, - // #11489: cursor/agy spell Claude ids - ("claude-4.6-opus-high"); - // the effort splitter strips those to "claude-4.6-opus", which is not a catalog id. - // `scoresAs` points each at the canonical - spelling so quality - // scores are inherited. Operational metadata stays on these entries. - { id: "claude-4.6-opus-high", name: "Opus 4.6 1M", scoresAs: "claude-opus-4-6" }, - { id: "claude-4.6-opus-max", name: "Opus 4.6 1M Max", scoresAs: "claude-opus-4-6" }, - { - id: "claude-4.6-opus-high-thinking", - name: "Opus 4.6 1M Thinking", - scoresAs: "claude-opus-4-6", - }, - { - id: "claude-4.6-opus-max-thinking", - name: "Opus 4.6 1M Max Thinking", - scoresAs: "claude-opus-4-6", - }, - { id: "claude-4.5-opus-high", name: "Opus 4.5", scoresAs: "claude-opus-4-5" }, - { id: "claude-4.5-opus-high-thinking", name: "Opus 4.5 Thinking", scoresAs: "claude-opus-4-5" }, - { id: "gpt-5.2-low", name: "GPT-5.2 Low" }, - { id: "gpt-5.2-low-fast", name: "GPT-5.2 Low Fast" }, - { id: "gpt-5.2-fast", name: "GPT-5.2 Fast" }, - { id: "gpt-5.2-high", name: "GPT-5.2 High" }, - { id: "gpt-5.2-high-fast", name: "GPT-5.2 High Fast" }, - { id: "gpt-5.2-xhigh", name: "GPT-5.2 Extra High" }, - { id: "gpt-5.2-xhigh-fast", name: "GPT-5.2 Extra High Fast" }, - { id: "gpt-5.6-luna-none", name: "GPT-5.6 Luna 1M None" }, - { id: "gpt-5.6-luna-none-fast", name: "GPT-5.6 Luna None Fast" }, - { id: "gpt-5.6-luna-low", name: "GPT-5.6 Luna 1M Low" }, - { id: "gpt-5.6-luna-low-fast", name: "GPT-5.6 Luna Low Fast" }, - { id: "gpt-5.6-luna-medium", name: "GPT-5.6 Luna 1M" }, - { id: "gpt-5.6-luna-medium-fast", name: "GPT-5.6 Luna Fast" }, - { id: "gpt-5.6-luna-high", name: "GPT-5.6 Luna 1M High" }, - { id: "gpt-5.6-luna-high-fast", name: "GPT-5.6 Luna High Fast" }, - { id: "gpt-5.6-luna-xhigh", name: "GPT-5.6 Luna 1M Extra High" }, - { id: "gpt-5.6-luna-xhigh-fast", name: "GPT-5.6 Luna Extra High Fast" }, - { id: "gpt-5.6-luna-max", name: "GPT-5.6 Luna 1M Max" }, - { id: "gpt-5.6-luna-max-fast", name: "GPT-5.6 Luna Max Fast" }, - { id: "gemini-3.6-flash-minimal", name: "Gemini 3.6 Flash Minimal" }, - { id: "gemini-3.6-flash-low", name: "Gemini 3.6 Flash Low" }, - { id: "gemini-3.6-flash-medium", name: "Gemini 3.6 Flash Medium" }, - { id: "gemini-3.6-flash-high", name: "Gemini 3.6 Flash" }, - { id: "gpt-5.4-mini-none", name: "GPT-5.4 Mini None" }, - { id: "gpt-5.4-mini-low", name: "GPT-5.4 Mini Low" }, - { id: "gpt-5.4-mini-medium", name: "GPT-5.4 Mini" }, - { id: "gpt-5.4-mini-high", name: "GPT-5.4 Mini High" }, - { id: "gpt-5.4-mini-xhigh", name: "GPT-5.4 Mini Extra High" }, - { id: "gpt-5.4-nano-none", name: "GPT-5.4 Nano None" }, - { id: "gpt-5.4-nano-low", name: "GPT-5.4 Nano Low" }, - { id: "gpt-5.4-nano-medium", name: "GPT-5.4 Nano" }, - { id: "gpt-5.4-nano-high", name: "GPT-5.4 Nano High" }, - { id: "gpt-5.4-nano-xhigh", name: "GPT-5.4 Nano Extra High" }, - { id: "claude-4.5-sonnet", name: "Sonnet 4.5", scoresAs: "claude-sonnet-4-5" }, - { - id: "claude-4.5-sonnet-thinking", - name: "Sonnet 4.5 Thinking", - scoresAs: "claude-sonnet-4-5", - }, - { id: "gpt-5.1-low", name: "GPT-5.1 Low" }, - { id: "gpt-5.1", name: "GPT-5.1" }, - { id: "gpt-5.1-high", name: "GPT-5.1 High" }, - { id: "claude-4-sonnet", name: "Sonnet 4", scoresAs: "claude-sonnet-4" }, - { id: "claude-4-sonnet-thinking", name: "Sonnet 4 Thinking", scoresAs: "claude-sonnet-4" }, - { id: "gpt-5-mini", name: "GPT-5 Mini" }, + { id: "composer-2.5", name: "Composer 2.5" }, + ...withOneMillionContext( + [ + { + id: "claude-fable-5-1-thinking-max", + name: "Claude Fable 5.1 Max Thinking", + ...CLAUDE_FABLE_5_1_CAPABILITIES, + }, + { + id: "claude-fable-5-1-thinking-xhigh", + name: "Claude Fable 5.1 Xhigh Thinking", + ...CLAUDE_FABLE_5_1_CAPABILITIES, + }, + { + id: "claude-fable-5-1-thinking-high", + name: "Claude Fable 5.1 High Thinking", + ...CLAUDE_FABLE_5_1_CAPABILITIES, + }, + { + id: "claude-fable-5-1-thinking-medium", + name: "Claude Fable 5.1 Medium Thinking", + ...CLAUDE_FABLE_5_1_CAPABILITIES, + }, + { + id: "claude-fable-5-1-thinking-low", + name: "Claude Fable 5.1 Low Thinking", + ...CLAUDE_FABLE_5_1_CAPABILITIES, + }, + ], + "Claude Fable 5.1", + 300_000, + "claude-fable-5-1" + ), + ...withOneMillionContext( + [ + { id: "claude-opus-5-thinking-max-fast", name: "Claude Opus 5 Max Thinking Fast" }, + { id: "claude-opus-5-thinking-max", name: "Claude Opus 5 Max Thinking" }, + { + id: "claude-opus-5-thinking-xhigh-fast", + name: "Claude Opus 5 Xhigh Thinking Fast", + }, + { id: "claude-opus-5-thinking-xhigh", name: "Claude Opus 5 Xhigh Thinking" }, + { id: "claude-opus-5-thinking-high-fast", name: "Claude Opus 5 High Thinking Fast" }, + { id: "claude-opus-5-thinking-high", name: "Claude Opus 5 High Thinking" }, + { id: "claude-opus-5-high-fast", name: "Claude Opus 5 High Fast" }, + { id: "claude-opus-5-high", name: "Claude Opus 5 High" }, + { + id: "claude-opus-5-thinking-medium-fast", + name: "Claude Opus 5 Medium Thinking Fast", + }, + { id: "claude-opus-5-thinking-medium", name: "Claude Opus 5 Medium Thinking" }, + { id: "claude-opus-5-medium-fast", name: "Claude Opus 5 Medium Fast" }, + { id: "claude-opus-5-medium", name: "Claude Opus 5 Medium" }, + { id: "claude-opus-5-thinking-low-fast", name: "Claude Opus 5 Low Thinking Fast" }, + { id: "claude-opus-5-thinking-low", name: "Claude Opus 5 Low Thinking" }, + { id: "claude-opus-5-low-fast", name: "Claude Opus 5 Low Fast" }, + { id: "claude-opus-5-low", name: "Claude Opus 5 Low" }, + ], + "Claude Opus 5", + 300_000, + "claude-opus-5" + ), + ...withOneMillionContext( + [ + { id: "claude-opus-4-8-thinking-max-fast", name: "Claude Opus 4.8 Max Thinking Fast" }, + { id: "claude-opus-4-8-thinking-max", name: "Claude Opus 4.8 Max Thinking" }, + { id: "claude-opus-4-8-max-fast", name: "Claude Opus 4.8 Max Fast" }, + { id: "claude-opus-4-8-max", name: "Claude Opus 4.8 Max" }, + { + id: "claude-opus-4-8-thinking-xhigh-fast", + name: "Claude Opus 4.8 Xhigh Thinking Fast", + }, + { id: "claude-opus-4-8-thinking-xhigh", name: "Claude Opus 4.8 Xhigh Thinking" }, + { id: "claude-opus-4-8-xhigh-fast", name: "Claude Opus 4.8 Xhigh Fast" }, + { id: "claude-opus-4-8-xhigh", name: "Claude Opus 4.8 Xhigh" }, + { id: "claude-opus-4-8-thinking-high-fast", name: "Claude Opus 4.8 High Thinking Fast" }, + { id: "claude-opus-4-8-thinking-high", name: "Claude Opus 4.8 High Thinking" }, + { id: "claude-opus-4-8-high-fast", name: "Claude Opus 4.8 High Fast" }, + { id: "claude-opus-4-8-high", name: "Claude Opus 4.8 High" }, + { + id: "claude-opus-4-8-thinking-medium-fast", + name: "Claude Opus 4.8 Medium Thinking Fast", + }, + { id: "claude-opus-4-8-thinking-medium", name: "Claude Opus 4.8 Medium Thinking" }, + { id: "claude-opus-4-8-medium-fast", name: "Claude Opus 4.8 Medium Fast" }, + { id: "claude-opus-4-8-medium", name: "Claude Opus 4.8 Medium" }, + { id: "claude-opus-4-8-thinking-low-fast", name: "Claude Opus 4.8 Low Thinking Fast" }, + { id: "claude-opus-4-8-thinking-low", name: "Claude Opus 4.8 Low Thinking" }, + { id: "claude-opus-4-8-low-fast", name: "Claude Opus 4.8 Low Fast" }, + { id: "claude-opus-4-8-low", name: "Claude Opus 4.8 Low" }, + ], + "Claude Opus 4.8", + 300_000, + "claude-opus-4-8" + ), + ...withOneMillionContext( + [ + { id: "claude-sonnet-5-thinking-max", name: "Claude Sonnet 5 Max Thinking" }, + { id: "claude-sonnet-5-max", name: "Claude Sonnet 5 Max" }, + { id: "claude-sonnet-5-thinking-xhigh", name: "Claude Sonnet 5 Xhigh Thinking" }, + { id: "claude-sonnet-5-xhigh", name: "Claude Sonnet 5 Xhigh" }, + { id: "claude-sonnet-5-thinking-high", name: "Claude Sonnet 5 High Thinking" }, + { id: "claude-sonnet-5-high", name: "Claude Sonnet 5 High" }, + { id: "claude-sonnet-5-thinking-medium", name: "Claude Sonnet 5 Medium Thinking" }, + { id: "claude-sonnet-5-medium", name: "Claude Sonnet 5 Medium" }, + { id: "claude-sonnet-5-thinking-low", name: "Claude Sonnet 5 Low Thinking" }, + { id: "claude-sonnet-5-low", name: "Claude Sonnet 5 Low" }, + ], + "Claude Sonnet 5", + 300_000, + "claude-sonnet-5" + ), + ...withOneMillionContext( + [ + { id: "claude-4.6-sonnet-max-thinking", name: "Claude Sonnet 4.6 Max Thinking" }, + { id: "claude-4.6-sonnet-max", name: "Claude Sonnet 4.6 Max" }, + { id: "claude-4.6-sonnet-high-thinking", name: "Claude Sonnet 4.6 High Thinking" }, + { id: "claude-4.6-sonnet-high", name: "Claude Sonnet 4.6 High" }, + { id: "claude-4.6-sonnet-medium-thinking", name: "Claude Sonnet 4.6 Medium Thinking" }, + { id: "claude-4.6-sonnet-medium", name: "Claude Sonnet 4.6 Medium" }, + { id: "claude-4.6-sonnet-low-thinking", name: "Claude Sonnet 4.6 Low Thinking" }, + { id: "claude-4.6-sonnet-low", name: "Claude Sonnet 4.6 Low" }, + ], + "Claude Sonnet 4.6", + 200_000, + "claude-sonnet-4-6" + ), + { id: "claude-4.5-haiku-thinking", name: "Claude Haiku 4.5 Thinking" }, + { id: "claude-4.5-haiku", name: "Claude Haiku 4.5" }, + ...withOneMillionContext( + [ + { id: "gpt-5.6-sol-max-fast", name: "GPT-5.6 Sol Max Fast" }, + { id: "gpt-5.6-sol-max", name: "GPT-5.6 Sol Max" }, + { id: "gpt-5.6-sol-xhigh-fast", name: "GPT-5.6 Sol Xhigh Fast" }, + { id: "gpt-5.6-sol-xhigh", name: "GPT-5.6 Sol Xhigh" }, + { id: "gpt-5.6-sol-high-fast", name: "GPT-5.6 Sol High Fast" }, + { id: "gpt-5.6-sol-high", name: "GPT-5.6 Sol High" }, + { id: "gpt-5.6-sol-medium-fast", name: "GPT-5.6 Sol Medium Fast" }, + { id: "gpt-5.6-sol-medium", name: "GPT-5.6 Sol Medium" }, + { id: "gpt-5.6-sol-low-fast", name: "GPT-5.6 Sol Low Fast" }, + { id: "gpt-5.6-sol-low", name: "GPT-5.6 Sol Low" }, + { id: "gpt-5.6-sol-none-fast", name: "GPT-5.6 Sol None Fast" }, + { id: "gpt-5.6-sol-none", name: "GPT-5.6 Sol None" }, + ], + "GPT-5.6 Sol", + 272_000, + "gpt-5.6-sol", + (model) => !model.id.endsWith("-fast") + ), + ...withOneMillionContext( + [ + { id: "gpt-5.6-terra-max-fast", name: "GPT-5.6 Terra Max Fast" }, + { id: "gpt-5.6-terra-max", name: "GPT-5.6 Terra Max" }, + { id: "gpt-5.6-terra-xhigh-fast", name: "GPT-5.6 Terra Xhigh Fast" }, + { id: "gpt-5.6-terra-xhigh", name: "GPT-5.6 Terra Xhigh" }, + { id: "gpt-5.6-terra-high-fast", name: "GPT-5.6 Terra High Fast" }, + { id: "gpt-5.6-terra-high", name: "GPT-5.6 Terra High" }, + { id: "gpt-5.6-terra-medium-fast", name: "GPT-5.6 Terra Medium Fast" }, + { id: "gpt-5.6-terra-medium", name: "GPT-5.6 Terra Medium" }, + { id: "gpt-5.6-terra-low-fast", name: "GPT-5.6 Terra Low Fast" }, + { id: "gpt-5.6-terra-low", name: "GPT-5.6 Terra Low" }, + { id: "gpt-5.6-terra-none-fast", name: "GPT-5.6 Terra None Fast" }, + { id: "gpt-5.6-terra-none", name: "GPT-5.6 Terra None" }, + ], + "GPT-5.6 Terra", + 272_000, + "gpt-5.6-terra", + (model) => !model.id.endsWith("-fast") + ), + ...withOneMillionContext( + [ + { id: "gpt-5.6-luna-max-fast", name: "GPT-5.6 Luna Max Fast" }, + { id: "gpt-5.6-luna-max", name: "GPT-5.6 Luna Max" }, + { id: "gpt-5.6-luna-xhigh-fast", name: "GPT-5.6 Luna Xhigh Fast" }, + { id: "gpt-5.6-luna-xhigh", name: "GPT-5.6 Luna Xhigh" }, + { id: "gpt-5.6-luna-high-fast", name: "GPT-5.6 Luna High Fast" }, + { id: "gpt-5.6-luna-high", name: "GPT-5.6 Luna High" }, + { id: "gpt-5.6-luna-medium-fast", name: "GPT-5.6 Luna Medium Fast" }, + { id: "gpt-5.6-luna-medium", name: "GPT-5.6 Luna Medium" }, + { id: "gpt-5.6-luna-low-fast", name: "GPT-5.6 Luna Low Fast" }, + { id: "gpt-5.6-luna-low", name: "GPT-5.6 Luna Low" }, + { id: "gpt-5.6-luna-none-fast", name: "GPT-5.6 Luna None Fast" }, + { id: "gpt-5.6-luna-none", name: "GPT-5.6 Luna None" }, + ], + "GPT-5.6 Luna", + 272_000, + "gpt-5.6-luna", + (model) => !model.id.endsWith("-fast") + ), + { id: "gemini-3.7-flash-high", name: "Gemini 3.7 Flash High" }, + { id: "gemini-3.7-flash-medium", name: "Gemini 3.7 Flash Medium" }, + { id: "gemini-3.7-flash-low", name: "Gemini 3.7 Flash Low" }, + { id: "gemini-3.1-pro", name: "Gemini 3.1 Pro" }, + { id: "kimi-k3-max", name: "Kimi K3 Max" }, + { id: "kimi-k3-high", name: "Kimi K3 High" }, { id: "kimi-k3-low", name: "Kimi K3 Low" }, - { id: "kimi-k3-max", name: "Kimi K3" }, - { id: "glm-5.2-high", name: "GLM 5.2" }, + { id: "kimi-k2.7-code", name: "Kimi K2.7 Code" }, { id: "glm-5.2-max", name: "GLM 5.2 Max" }, + { id: "glm-5.2-high", name: "GLM 5.2 High" }, ], }; diff --git a/open-sse/config/providers/registry/devin/catalog.ts b/open-sse/config/providers/registry/devin/catalog.ts index d1e5c75894..6ab53fda12 100644 --- a/open-sse/config/providers/registry/devin/catalog.ts +++ b/open-sse/config/providers/registry/devin/catalog.ts @@ -1,115 +1,150 @@ import type { RegistryModel } from "../../shared.ts"; +type EffortVariant = readonly [suffix: string, label: string]; + +const QUALITY_EFFORTS: readonly EffortVariant[] = [ + ["max", "Max"], + ["xhigh", "XHigh"], + ["high", "High"], + ["medium", "Medium"], + ["low", "Low"], +]; + +const GPT_EFFORTS: readonly EffortVariant[] = [ + ["max", "Max Thinking"], + ["xhigh", "XHigh Thinking"], + ["high", "High Thinking"], + ["medium", "Medium Thinking"], + ["low", "Low Thinking"], + ["none", "No Thinking"], +]; + +function model( + id: string, + name: string, + maxOutputTokens?: number, + contextLength?: number +): RegistryModel { + return { + id, + name, + ...(maxOutputTokens === undefined ? {} : { maxOutputTokens }), + ...(contextLength === undefined ? {} : { contextLength }), + }; +} + +function effortModels( + id: string, + name: string, + maxOutputTokens: number, + contextLength: number | undefined, + efforts: readonly EffortVariant[] = QUALITY_EFFORTS +): RegistryModel[] { + return efforts.map(([suffix, label]) => + model(`${id}-${suffix}`, `${name} ${label}`, maxOutputTokens, contextLength) + ); +} + +function fastEffortModels( + id: string, + name: string, + maxOutputTokens: number, + contextLength: number +): RegistryModel[] { + return QUALITY_EFFORTS.flatMap(([suffix, label]) => [ + model(`${id}-${suffix}-fast`, `${name} ${label} Fast`, maxOutputTokens, contextLength), + model(`${id}-${suffix}`, `${name} ${label}`, maxOutputTokens, contextLength), + ]); +} + +function gptModels(id: string, name: string): RegistryModel[] { + return GPT_EFFORTS.flatMap(([suffix, label]) => [ + model(`${id}-${suffix}-priority`, `${name} ${label} Fast`, 128_000, 1_000_000), + model(`${id}-${suffix}`, `${name} ${label}`, 128_000, 1_000_000), + ]); +} + +/** + * Curated from the authenticated `devin models list --format json` response on + * 2026-09-02. Keep this deliberately smaller than Devin's full live catalog: + * these are the operator-selected models OmniRoute intends to expose. + */ export const DEVIN_MODEL_CATALOG: RegistryModel[] = [ - // Cognition / SWE — default model family recommended for coding tasks - { id: "swe-1-7-lightning", name: "SWE-1.7 Lightning", contextLength: 202752 }, - { id: "swe-1-7", name: "SWE-1.7", contextLength: 262000 }, - { id: "swe-1-6-fast", name: "SWE-1.6 Fast" }, - { id: "swe-1-6", name: "SWE-1.6" }, - // Claude Fable 5 - { id: "claude-5-fable-max", name: "Claude Fable 5 Max", contextLength: 1000000 }, - { id: "claude-5-fable-xhigh", name: "Claude Fable 5 XHigh", contextLength: 1000000 }, - { id: "claude-5-fable-high", name: "Claude Fable 5 High", contextLength: 1000000 }, - { id: "claude-5-fable-medium", name: "Claude Fable 5 Medium", contextLength: 1000000 }, - { id: "claude-5-fable-low", name: "Claude Fable 5 Low", contextLength: 1000000 }, - // Claude Opus 5 - { id: "claude-opus-5-max", name: "Claude Opus 5 Max", contextLength: 1000000 }, - { id: "claude-opus-5-xhigh", name: "Claude Opus 5 XHigh", contextLength: 1000000 }, - { id: "claude-opus-5-high", name: "Claude Opus 5 High", contextLength: 1000000 }, - { id: "claude-opus-5-medium", name: "Claude Opus 5 Medium", contextLength: 1000000 }, - { id: "claude-opus-5-low", name: "Claude Opus 5 Low", contextLength: 1000000 }, - // Claude Opus 4.8 - { id: "claude-opus-4-8-max", name: "Claude Opus 4.8 Max", contextLength: 1000000 }, - { id: "claude-opus-4-8-xhigh", name: "Claude Opus 4.8 XHigh", contextLength: 1000000 }, - { id: "claude-opus-4-8-high", name: "Claude Opus 4.8 High", contextLength: 1000000 }, - { id: "claude-opus-4-8-medium", name: "Claude Opus 4.8 Medium", contextLength: 1000000 }, - { id: "claude-opus-4-8-low", name: "Claude Opus 4.8 Low", contextLength: 1000000 }, - // Claude Opus 4.7 - { id: "claude-opus-4-7-max", name: "Claude Opus 4.7 Max", contextLength: 1000000 }, - { id: "claude-opus-4-7-xhigh", name: "Claude Opus 4.7 XHigh", contextLength: 1000000 }, - { id: "claude-opus-4-7-high", name: "Claude Opus 4.7 High", contextLength: 1000000 }, - { id: "claude-opus-4-7-medium", name: "Claude Opus 4.7 Medium", contextLength: 1000000 }, - { id: "claude-opus-4-7-low", name: "Claude Opus 4.7 Low", contextLength: 1000000 }, - // Claude Opus 4.6 - { - id: "claude-opus-4-6-thinking-1m", - name: "Claude Opus 4.6 Thinking 1M", - contextLength: 1000000, - }, - { id: "claude-opus-4-6-thinking", name: "Claude Opus 4.6 Thinking", contextLength: 200000 }, - { id: "claude-opus-4-6-1m", name: "Claude Opus 4.6 1M", contextLength: 1000000 }, - { id: "claude-opus-4-6", name: "Claude Opus 4.6", contextLength: 200000 }, - // Claude Sonnet 5 - { id: "claude-sonnet-5-max", name: "Claude Sonnet 5 Max", contextLength: 1000000 }, - { id: "claude-sonnet-5-xhigh", name: "Claude Sonnet 5 XHigh", contextLength: 1000000 }, - { id: "claude-sonnet-5-high", name: "Claude Sonnet 5 High", contextLength: 1000000 }, - { id: "claude-sonnet-5-medium", name: "Claude Sonnet 5 Medium", contextLength: 1000000 }, - { id: "claude-sonnet-5-low", name: "Claude Sonnet 5 Low", contextLength: 1000000 }, - // Claude Sonnet 4.6 - { - id: "claude-sonnet-4-6-thinking-1m", - name: "Claude Sonnet 4.6 Thinking 1M", - contextLength: 1000000, - }, - { - id: "claude-sonnet-4-6-thinking", - name: "Claude Sonnet 4.6 Thinking", - contextLength: 200000, - }, - { id: "claude-sonnet-4-6-1m", name: "Claude Sonnet 4.6 1M", contextLength: 1000000 }, - { id: "claude-sonnet-4-6", name: "Claude Sonnet 4.6", contextLength: 200000 }, - // GPT-5.6 - { id: "gpt-5-6-sol-max", name: "GPT-5.6 Sol Max", contextLength: 1000000 }, - { id: "gpt-5-6-sol-xhigh", name: "GPT-5.6 Sol XHigh", contextLength: 1000000 }, - { id: "gpt-5-6-sol-high", name: "GPT-5.6 Sol High", contextLength: 1000000 }, - { id: "gpt-5-6-sol-medium", name: "GPT-5.6 Sol Medium", contextLength: 1000000 }, - { id: "gpt-5-6-sol-low", name: "GPT-5.6 Sol Low", contextLength: 1000000 }, - /// Terra - { id: "gpt-5-6-terra-max", name: "GPT-5.6 Terra Max", contextLength: 1000000 }, - { id: "gpt-5-6-terra-xhigh", name: "GPT-5.6 Terra XHigh", contextLength: 1000000 }, - { id: "gpt-5-6-terra-high", name: "GPT-5.6 Terra High", contextLength: 1000000 }, - { id: "gpt-5-6-terra-medium", name: "GPT-5.6 Terra Medium", contextLength: 1000000 }, - { id: "gpt-5-6-terra-low", name: "GPT-5.6 Terra Low", contextLength: 1000000 }, - /// Luna - { id: "gpt-5-6-luna-max", name: "GPT-5.6 Luna Max", contextLength: 1000000 }, - { id: "gpt-5-6-luna-xhigh", name: "GPT-5.6 Luna XHigh", contextLength: 1000000 }, - { id: "gpt-5-6-luna-high", name: "GPT-5.6 Luna High", contextLength: 1000000 }, - { id: "gpt-5-6-luna-medium", name: "GPT-5.6 Luna Medium", contextLength: 1000000 }, - { id: "gpt-5-6-luna-low", name: "GPT-5.6 Luna Low", contextLength: 1000000 }, - // GPT-5.5 - { id: "gpt-5-5-xhigh", name: "GPT-5.5 XHigh", contextLength: 272000 }, - { id: "gpt-5-5-high", name: "GPT-5.5 High", contextLength: 272000 }, - { id: "gpt-5-5-medium", name: "GPT-5.5 Medium", contextLength: 272000 }, - { id: "gpt-5-5-low", name: "GPT-5.5 Low", contextLength: 272000 }, - // Gemini - { id: "gemini-3-1-pro-high", name: "Gemini 3.1 Pro High", contextLength: 1048576 }, - { id: "gemini-3-1-pro-low", name: "Gemini 3.1 Pro Low", contextLength: 1048576 }, - { id: "gemini-3-7-flash-high", name: "Gemini 3.7 Flash High" }, - { id: "gemini-3-7-flash-medium", name: "Gemini 3.7 Flash Medium" }, - { id: "gemini-3-7-flash-low", name: "Gemini 3.7 Flash Low" }, - { id: "gemini-3-7-flash-minimal", name: "Gemini 3.7 Flash Minimal" }, - // Grok - { id: "grok-4-5-high", name: "Grok 4.5 High", contextLength: 500000 }, - { id: "grok-4-5-medium", name: "Grok 4.5 Medium", contextLength: 500000 }, - { id: "grok-4-5-low", name: "Grok 4.5 Low", contextLength: 500000 }, - // GLM - { id: "glm-5-2-max-1m", name: "GLM-5.2 Max 1M", contextLength: 1000000 }, - { id: "glm-5-2-max", name: "GLM-5.2 Max" }, - { id: "glm-5-2-1m", name: "GLM-5.2 High 1M", contextLength: 1000000 }, - { id: "glm-5-2", name: "GLM-5.2 High" }, - // Kimi - { id: "kimi-k3-max", name: "Kimi K3 Max" }, - { id: "kimi-k3-high", name: "Kimi K3 High" }, - { id: "kimi-k3-low", name: "Kimi K3 Low" }, - { id: "kimi-k2-7", name: "Kimi K2.7", contextLength: 262144 }, - // Inkling - { id: "inkling-max", name: "Inkling Max" }, - { id: "inkling-xhigh", name: "Inkling XHigh" }, - { id: "inkling-high", name: "Inkling High" }, - { id: "inkling-medium", name: "Inkling Medium" }, - { id: "inkling-low", name: "Inkling Low" }, - { id: "inkling-none", name: "Inkling None" }, - // Others - { id: "deepseek-v4", name: "DeepSeek V4 Pro", contextLength: 1048576 }, - { id: "nemotron-3-ultra-nvfp4", name: "Nemotron 3 Ultra", contextLength: 262144 }, + ...effortModels("claude-fable-5-1", "Claude Fable 5.1", 128_000, 1_000_000), + ...fastEffortModels("claude-opus-5", "Claude Opus 5", 128_000, 1_000_000), + ...fastEffortModels("claude-opus-4-8", "Claude Opus 4.8", 128_000, 1_000_000), + ...effortModels("claude-sonnet-5", "Claude Sonnet 5", 128_000, 1_000_000), + + model("claude-sonnet-4-6-thinking-1m", "Claude Sonnet 4.6 Thinking 1M", 128_000, 1_000_000), + model("claude-sonnet-4-6-1m", "Claude Sonnet 4.6 1M", 128_000, 1_000_000), + model("claude-sonnet-4-6-thinking", "Claude Sonnet 4.6 Thinking", 128_000, 200_000), + model("claude-sonnet-4-6", "Claude Sonnet 4.6", 128_000, 200_000), + model("MODEL_PRIVATE_11", "Claude Haiku 4.5", 64_000, 200_000), + + ...gptModels("gpt-5-6-sol", "GPT-5.6 Sol"), + ...gptModels("gpt-5-6-terra", "GPT-5.6 Terra"), + ...gptModels("gpt-5-6-luna", "GPT-5.6 Luna"), + + ...effortModels("kimi-k3", "Kimi K3", 131_072, 1_048_576, [ + ["max", "Max"], + ["high", "High"], + ["low", "Low"], + ]), + model("kimi-k2-7", "Kimi K2.7", 16_000, 262_144), + + ...effortModels("glm-5-3", "GLM-5.3", 128_000, 1_000_000, [ + ["max", "Max"], + ["high", "High"], + ["low", "Low"], + ]), + ...effortModels("glm-5-3-flash", "GLM-5.3 Flash", 128_000, 1_000_000, [ + ["max", "Max"], + ["high", "High"], + ["low", "Low"], + ]), + + model("swe-1-7", "SWE-1.7 Max", 128_000, 262_000), + model("swe-1-7-medium", "SWE-1.7 Medium", 128_000, 262_000), + model("swe-1-7-lightning", "SWE-1.7 Lightning Max", 96_000, 202_752), + model("swe-1-7-lightning-medium", "SWE-1.7 Lightning Medium", 96_000, 202_752), + model("adaptive", "Adaptive"), + + ...effortModels("grok-4-6", "Grok 4.6", 100_000, 500_000, [ + ["xhigh", "XHigh"], + ["high", "High"], + ["medium", "Medium"], + ["low", "Low"], + ]), + ...effortModels("inkling", "Inkling", 131_072, undefined, [ + ["max", "Max"], + ["xhigh", "X-High"], + ["high", "High"], + ["medium", "Medium"], + ["low", "Low"], + ["none", "None"], + ]), + ...effortModels("deepseek-v4-flash", "DeepSeek V4 Flash", 384_000, 1_000_000, [ + ["max", "Max"], + ["high", "High"], + ["low", "Low"], + ]), + ...effortModels("nemotron-3-ultra", "Nemotron 3 Ultra", 32_768, 262_144, [ + ["high", "High"], + ["medium", "Medium"], + ["none", "None"], + ]), + ...effortModels("gemini-3-7-flash", "Gemini 3.7 Flash", 65_535, 1_048_576, [ + ["high", "High"], + ["medium", "Medium"], + ["low", "Low"], + ]), + ...effortModels("gemini-3-1-pro", "Gemini 3.1 Pro", 65_535, 1_048_576, [ + ["high", "High Thinking"], + ["low", "Low Thinking"], + ]), + ...effortModels("deepseek-v4-pro", "DeepSeek V4 Pro", 384_000, 1_000_000, [ + ["max", "Max"], + ["high", "High"], + ["low", "Low"], + ]), ]; diff --git a/open-sse/config/providers/registry/vertex/index.ts b/open-sse/config/providers/registry/vertex/index.ts index 1eac20fc15..4b72b9f7c4 100644 --- a/open-sse/config/providers/registry/vertex/index.ts +++ b/open-sse/config/providers/registry/vertex/index.ts @@ -27,6 +27,7 @@ export const vertexProvider: RegistryEntry = { { id: "DeepSeek-V4-Pro", name: "DeepSeek V4 Pro (Vertex Partner)" }, { id: "Qwen3.6-35B-A3B", name: "Qwen3.6 35B A3B (Vertex Partner)" }, { id: "GLM-5.1-FP8", name: "GLM-5.1 (Vertex Partner)" }, + { id: "claude-fable-5-1", name: "Claude Fable 5.1 (Vertex)", targetFormat: "claude" }, { id: "claude-fable-5", name: "Claude Fable 5 (Vertex)", targetFormat: "claude" }, { id: "claude-opus-5", name: "Claude Opus 5 (Vertex)", targetFormat: "claude" }, { id: "claude-sonnet-5", name: "Claude Sonnet 5 (Vertex)", targetFormat: "claude" }, diff --git a/open-sse/config/providers/registry/vertex/partner/index.ts b/open-sse/config/providers/registry/vertex/partner/index.ts index 4cdcd5d0b5..6524cd6a2b 100644 --- a/open-sse/config/providers/registry/vertex/partner/index.ts +++ b/open-sse/config/providers/registry/vertex/partner/index.ts @@ -13,6 +13,7 @@ export const vertex_partnerProvider: RegistryEntry = { { id: "DeepSeek-V4-Pro", name: "DeepSeek V4 Pro" }, { id: "Qwen3.6-35B-A3B", name: "Qwen 3.6 35B A3B" }, { id: "GLM-5.1-FP8", name: "GLM 5.1" }, + { id: "claude-fable-5-1", name: "Claude Fable 5.1", targetFormat: "claude" }, { id: "claude-fable-5", name: "Claude Fable 5", targetFormat: "claude" }, { id: "claude-opus-5", name: "Claude Opus 5", targetFormat: "claude" }, { id: "claude-sonnet-5", name: "Claude Sonnet 5", targetFormat: "claude" }, diff --git a/open-sse/services/autoCombo/__tests__/scoresAs-11489.test.ts b/open-sse/services/autoCombo/__tests__/scoresAs-11489.test.ts index b1d8d31b36..b3f719558f 100644 --- a/open-sse/services/autoCombo/__tests__/scoresAs-11489.test.ts +++ b/open-sse/services/autoCombo/__tests__/scoresAs-11489.test.ts @@ -59,11 +59,9 @@ describe("#11489 resolveScoresAs", () => { expect(resolveScoresAs("claude-sonnet-5")).toEqual({ base: "claude-sonnet-5", via: null }); }); - it("resolves the cursor/agy spelling of a Claude model to its canonical id", () => { - // `claude-4.6-opus-high` strips to `claude-4.6-opus`, which is not a catalog - // id — the canonical spelling is `claude-opus-4-6`. Explicit registry data. - expect(resolveScoresAs("claude-4.6-opus-high")).toEqual({ - base: "claude-opus-4-6", + it("resolves curated Cursor Claude variants to their canonical ids", () => { + expect(resolveScoresAs("claude-fable-5-1-thinking-high")).toEqual({ + base: "claude-fable-5-1", via: "explicit", }); expect(resolveScoresAs("claude-4.6-sonnet-medium")).toEqual({ diff --git a/open-sse/services/modelFamilyFallback.ts b/open-sse/services/modelFamilyFallback.ts index 5e80fe0d40..6226c1217b 100644 --- a/open-sse/services/modelFamilyFallback.ts +++ b/open-sse/services/modelFamilyFallback.ts @@ -78,8 +78,9 @@ const FAMILY_FALLBACK_TEMPLATES: Record = { "gemini-2.5-pro": ["gemini-2.5-pro-preview-06-05", "gemini-2.5-pro-exp-03-25"], "gemini-2.5-pro-preview-06-05": ["gemini-2.5-pro", "gemini-2.5-pro-exp-03-25"], - // Claude Mythos family (Fable 5) — flagship falls to the next-best Opus - // tiers before the cheaper Sonnet, matching the Opus family ordering. + // Claude Mythos family — prefer the previous Fable before falling to Opus + // tiers and then the cheaper Sonnet, matching the flagship ordering. + "claude-fable-5-1": ["claude-fable-5", "claude-opus-5", "claude-sonnet-5"], "claude-fable-5": ["claude-opus-4-8", "claude-opus-4-7", "claude-sonnet-5"], // Claude Opus family diff --git a/open-sse/services/providerCostData.ts b/open-sse/services/providerCostData.ts index d9a4e0a8ab..8464635c92 100644 --- a/open-sse/services/providerCostData.ts +++ b/open-sse/services/providerCostData.ts @@ -1,4 +1,4 @@ -import type { TierAssignment } from "./tierTypes"; +import { getPricingForModel as getDefaultPricingForModel } from "@/shared/constants/pricing"; import type { TierConfig } from "./tierTypes"; export interface ModelPricing { @@ -11,6 +11,7 @@ export interface ModelPricing { export const KNOWN_MODEL_PRICING: Record = { "gpt-4o": { inputCostPer1M: 2.5, outputCostPer1M: 10.0, isFree: false }, "gpt-4o-mini": { inputCostPer1M: 0.15, outputCostPer1M: 0.6, isFree: false }, + "claude-fable-5-1": { inputCostPer1M: 10.0, outputCostPer1M: 50.0, isFree: false }, "claude-fable-5": { inputCostPer1M: 15.0, outputCostPer1M: 75.0, isFree: false }, "claude-opus-5": { inputCostPer1M: 5.0, outputCostPer1M: 25.0, isFree: false }, "claude-opus-4-8": { inputCostPer1M: 15.0, outputCostPer1M: 75.0, isFree: false }, @@ -37,14 +38,26 @@ export const KNOWN_MODEL_PRICING: Record = { }; export function getModelPricing(provider: string, model: string): ModelPricing { - const directKey = model.toLowerCase(); - if (KNOWN_MODEL_PRICING[directKey]) { - return KNOWN_MODEL_PRICING[directKey]; - } const providerKey = `${provider}/${model}`.toLowerCase(); if (KNOWN_MODEL_PRICING[providerKey]) { return KNOWN_MODEL_PRICING[providerKey]; } + const providerPricing = getDefaultPricingForModel(provider, model); + if (providerPricing) { + const inputCostPer1M = Number(providerPricing.input); + const outputCostPer1M = Number(providerPricing.output); + if (Number.isFinite(inputCostPer1M) && Number.isFinite(outputCostPer1M)) { + return { + inputCostPer1M, + outputCostPer1M, + isFree: inputCostPer1M === 0 && outputCostPer1M === 0, + }; + } + } + const directKey = model.toLowerCase(); + if (KNOWN_MODEL_PRICING[directKey]) { + return KNOWN_MODEL_PRICING[directKey]; + } return { inputCostPer1M: 5.0, outputCostPer1M: 15.0, isFree: false }; } diff --git a/open-sse/utils/cursorAgentProtobuf.ts b/open-sse/utils/cursorAgentProtobuf.ts index 21164b6ed1..0e3f642682 100644 --- a/open-sse/utils/cursorAgentProtobuf.ts +++ b/open-sse/utils/cursorAgentProtobuf.ts @@ -24,6 +24,10 @@ import { encodeSelectedImageBody, type EncodedImage, } from "./cursorAgentProtobuf/imageEncoding.ts"; +import { + CURSOR_EFFORT_SUFFIXES, + resolveOneMillionContextModel, +} from "./cursorAgentProtobuf/requestedModelParameters.ts"; import { WT_VARINT, WT_LEN, @@ -41,6 +45,7 @@ import { findField, decodeStringField, decodeVarintField, + type Field, } from "./cursorAgentProtobuf/wire.ts"; // ─── Field numbers (from agent.proto descriptor) ─────────────────────────── @@ -311,8 +316,6 @@ export function normalizeCursorModelId(modelId: string): string { // Grok (`cursor-grok-*` / legacy `grok-*`) follows the Claude-style `effort` // parameter. Without the split, ids like `cursor-grok-4.5-high` return empty // turns (same symptom as #7289). Combined `-high-fast` is supported. -const CURSOR_EFFORT_SUFFIXES = ["low", "medium", "high", "xhigh", "max"] as const; - /** * If `normalized` starts with `prefix` and ends with one of the known effort * suffixes, split it into the base model id plus a `{id: paramId, value}` @@ -432,6 +435,8 @@ export function resolveRequestedModel( }; } } + const oneMillionContext = resolveOneMillionContextModel(normalized); + if (oneMillionContext) return oneMillionContext; // Live catalog is authoritative for exact ids (flattened effort variants). if (opts?.liveCatalogIds?.has(normalized)) { return { modelId: normalized, parameters: [] }; @@ -652,6 +657,41 @@ export type DecodedDelta = | { kind: "kv_server_message" } | { kind: "unknown"; field: number }; +type InteractionUpdateDecoder = (field: Field) => DecodedDelta[]; + +const INTERACTION_UPDATE_DECODERS: Partial> = { + [IU_TEXT_DELTA]: (field) => + field.wireType === WT_LEN + ? [{ kind: "text", text: decodeStringField(field.bytes, TDU_TEXT) }] + : [], + [IU_THINKING_DELTA]: (field) => + field.wireType === WT_LEN + ? [{ kind: "thinking", text: decodeStringField(field.bytes, TDU_TEXT) }] + : [], + [IU_THINKING_COMPLETED]: () => [{ kind: "thinking_complete" }], + [IU_TOOL_CALL_STARTED]: () => [{ kind: "tool_call_started" }], + [IU_TOOL_CALL_COMPLETED]: (field) => { + const deltas: DecodedDelta[] = []; + if (field.wireType === WT_LEN) { + const todoWrite = decodeNativeTodoWriteCompletion(field.bytes); + if (todoWrite) deltas.push(todoWrite); + } + deltas.push({ kind: "tool_call_completed" }); + return deltas; + }, + [IU_TOKEN_DELTA]: (field) => + field.wireType === WT_LEN + ? [{ kind: "token_delta", tokens: decodeVarintField(field.bytes, 1) }] + : [], + [IU_HEARTBEAT]: () => [{ kind: "heartbeat" }], + [IU_TURN_ENDED]: () => [{ kind: "turn_ended" }], +}; + +function decodeInteractionUpdate(field: Field): DecodedDelta[] { + const decoder = INTERACTION_UPDATE_DECODERS[field.fieldNumber]; + return decoder ? decoder(field) : [{ kind: "unknown", field: field.fieldNumber }]; +} + export function decodeAgentServerMessage(payload: Buffer): DecodedDelta[] { const out: DecodedDelta[] = []; for (const top of decodeFields(payload)) { @@ -661,45 +701,7 @@ export function decodeAgentServerMessage(payload: Buffer): DecodedDelta[] { } if (top.fieldNumber !== ASM_INTERACTION_UPDATE || top.wireType !== 2) continue; for (const update of decodeFields(top.bytes)) { - if (update.wireType !== 2 && update.wireType !== 0) continue; - switch (update.fieldNumber) { - case IU_TEXT_DELTA: - if (update.wireType === 2) { - out.push({ kind: "text", text: decodeStringField(update.bytes, TDU_TEXT) }); - } - break; - case IU_THINKING_DELTA: - if (update.wireType === 2) { - out.push({ kind: "thinking", text: decodeStringField(update.bytes, TDU_TEXT) }); - } - break; - case IU_THINKING_COMPLETED: - out.push({ kind: "thinking_complete" }); - break; - case IU_TOOL_CALL_STARTED: - out.push({ kind: "tool_call_started" }); - break; - case IU_TOOL_CALL_COMPLETED: - if (update.wireType === 2) { - const todoWrite = decodeNativeTodoWriteCompletion(update.bytes); - if (todoWrite) out.push(todoWrite); - } - out.push({ kind: "tool_call_completed" }); - break; - case IU_TOKEN_DELTA: - if (update.wireType === 2) { - out.push({ kind: "token_delta", tokens: decodeVarintField(update.bytes, 1) }); - } - break; - case IU_HEARTBEAT: - out.push({ kind: "heartbeat" }); - break; - case IU_TURN_ENDED: - out.push({ kind: "turn_ended" }); - break; - default: - out.push({ kind: "unknown", field: update.fieldNumber }); - } + out.push(...decodeInteractionUpdate(update)); } } return out; @@ -750,52 +752,44 @@ export type KvServerEvent = requestMetadata: Buffer | null; }; +function findLengthDelimitedField(fields: Field[], fieldNumber: number): Buffer | null { + const field = findField(fields, fieldNumber); + return field?.wireType === WT_LEN ? field.bytes : null; +} + +function decodeBlobId(payload: Buffer, fieldNumber: number): Buffer { + return findLengthDelimitedField(decodeFields(payload), fieldNumber) ?? Buffer.alloc(0); +} + +function decodeSetBlobArgs(payload: Buffer): { blobId: Buffer; blobData: Buffer } { + const fields = decodeFields(payload); + return { + blobId: findLengthDelimitedField(fields, SBA_BLOB_ID) ?? Buffer.alloc(0), + blobData: findLengthDelimitedField(fields, SBA_BLOB_DATA) ?? Buffer.alloc(0), + }; +} + export function decodeKvServerEvent(payload: Buffer): KvServerEvent | null { - for (const top of decodeFields(payload)) { - if (top.fieldNumber !== ASM_KV_SERVER_MESSAGE || top.wireType !== 2) continue; + const top = findField(decodeFields(payload), ASM_KV_SERVER_MESSAGE); + if (top?.wireType !== WT_LEN) return null; - let kvId = 0; - let getBlobArgs: Buffer | null = null; - let setBlobArgs: Buffer | null = null; - let requestMetadata: Buffer | null = null; - - for (const f of decodeFields(top.bytes)) { - if (f.fieldNumber === KSM_ID && f.wireType === 0) { - kvId = Number(f.varint); - } else if (f.fieldNumber === KSM_GET_BLOB_ARGS && f.wireType === 2) { - getBlobArgs = f.bytes; - } else if (f.fieldNumber === KSM_SET_BLOB_ARGS && f.wireType === 2) { - setBlobArgs = f.bytes; - } else if (f.fieldNumber === KSM_REQUEST_METADATA && f.wireType === 2) { - requestMetadata = f.bytes; - } - } - - if (getBlobArgs) { - // GetBlobArgs { blob_id (1): bytes } - let blobId: Buffer = Buffer.alloc(0); - for (const f of decodeFields(getBlobArgs)) { - if (f.fieldNumber === GBA_BLOB_ID && f.wireType === 2) { - blobId = f.bytes; - } - } - return { kind: "kv_get_blob", kvId, blobId, requestMetadata }; - } - if (setBlobArgs) { - // SetBlobArgs { blob_id (1): bytes, blob_data (2): bytes } - let blobId: Buffer = Buffer.alloc(0); - let blobData: Buffer = Buffer.alloc(0); - for (const f of decodeFields(setBlobArgs)) { - if (f.fieldNumber === SBA_BLOB_ID && f.wireType === 2) { - blobId = f.bytes; - } else if (f.fieldNumber === SBA_BLOB_DATA && f.wireType === 2) { - blobData = f.bytes; - } - } - return { kind: "kv_set_blob", kvId, blobId, blobData, requestMetadata }; - } + const fields = decodeFields(top.bytes); + const idField = findField(fields, KSM_ID); + const kvId = idField?.wireType === WT_VARINT ? Number(idField.varint) : 0; + const requestMetadata = findLengthDelimitedField(fields, KSM_REQUEST_METADATA); + const getBlobArgs = findLengthDelimitedField(fields, KSM_GET_BLOB_ARGS); + if (getBlobArgs) { + return { + kind: "kv_get_blob", + kvId, + blobId: decodeBlobId(getBlobArgs, GBA_BLOB_ID), + requestMetadata, + }; } - return null; + + const setBlobArgs = findLengthDelimitedField(fields, KSM_SET_BLOB_ARGS); + if (!setBlobArgs) return null; + return { kind: "kv_set_blob", kvId, ...decodeSetBlobArgs(setBlobArgs), requestMetadata }; } // ─── Phase 2: full ExecServerMessage variant decoder ─────────────────────── @@ -886,143 +880,121 @@ function decodeShellArgs(payload: Buffer): DecodedShellArgs { return decoded; } -export function decodeExecServerEvent(payload: Buffer): ExecServerEvent | null { - for (const top of decodeFields(payload)) { - if (top.fieldNumber !== ASM_EXEC_SERVER_MESSAGE || top.wireType !== 2) continue; +type ExecEventContext = { + execMsgId: number; + execId: string; + variantBytes: Buffer; +}; - let execMsgId = 0; - let execId = ""; - let variantField = 0; - let variantBytes: Buffer | null = null; +type ExecEventDecoder = (context: ExecEventContext) => ExecServerEvent; +type PathExecKind = "exec_read" | "exec_write" | "exec_delete" | "exec_ls"; +type ShellExecKind = "exec_shell" | "exec_shell_stream" | "exec_bg_shell"; - for (const f of decodeFields(top.bytes)) { - if (f.fieldNumber === ESM_ID && f.wireType === 0) { - execMsgId = Number(f.varint); - } else if (f.fieldNumber === ESM_EXEC_ID && f.wireType === 2) { - execId = f.bytes.toString("utf8"); - } else if (f.wireType === 2) { - // Any other LEN field is the variant payload. Take the first one we - // see — variants don't co-occur in a well-formed message. - if (variantField === 0) { - variantField = f.fieldNumber; - variantBytes = f.bytes; - } - } - } +function createPathExecEvent(kind: PathExecKind, context: ExecEventContext): ExecServerEvent { + return { + kind, + execMsgId: context.execMsgId, + execId: context.execId, + path: decodeStringField(context.variantBytes, ARG_PATH), + }; +} - if (variantBytes === null) continue; +function createShellExecEvent(kind: ShellExecKind, context: ExecEventContext): ExecServerEvent { + return { + kind, + execMsgId: context.execMsgId, + execId: context.execId, + ...decodeShellArgs(context.variantBytes), + }; +} - switch (variantField) { - case ESM_REQUEST_CONTEXT_ARGS: - return { kind: "exec_request_context", execMsgId, execId }; - case ESM_READ_ARGS: - return { - kind: "exec_read", - execMsgId, - execId, - path: decodeStringField(variantBytes, ARG_PATH), - }; - case ESM_WRITE_ARGS: - return { - kind: "exec_write", - execMsgId, - execId, - path: decodeStringField(variantBytes, ARG_PATH), - }; - case ESM_DELETE_ARGS: - return { - kind: "exec_delete", - execMsgId, - execId, - path: decodeStringField(variantBytes, ARG_PATH), - }; - case ESM_LS_ARGS: - return { - kind: "exec_ls", - execMsgId, - execId, - path: decodeStringField(variantBytes, ARG_PATH), - }; - case ESM_GREP_ARGS: - return { kind: "exec_grep", execMsgId, execId }; - case ESM_DIAGNOSTICS_ARGS: - return { kind: "exec_diagnostics", execMsgId, execId }; - case ESM_SHELL_ARGS: { - const shell = decodeShellArgs(variantBytes); - return { - kind: "exec_shell", - execMsgId, - execId, - ...shell, - }; - } - case ESM_SHELL_STREAM_ARGS: { - const shell = decodeShellArgs(variantBytes); - return { - kind: "exec_shell_stream", - execMsgId, - execId, - ...shell, - }; - } - case ESM_BACKGROUND_SHELL_SPAWN: { - const shell = decodeShellArgs(variantBytes); - return { - kind: "exec_bg_shell", - execMsgId, - execId, - ...shell, - }; - } - case ESM_FETCH_ARGS: - return { - kind: "exec_fetch", - execMsgId, - execId, - url: decodeStringField(variantBytes, ARG_FETCH_URL), - }; - case ESM_WRITE_SHELL_STDIN_ARGS: - return { kind: "exec_write_shell_stdin", execMsgId, execId }; - case ESM_MCP_ARGS: { - // McpArgs.args is map; each value is a protobuf- - // encoded google.protobuf.Value. Decode keys and value-bytes here, - // then convert each Value to its JSON shape. - let toolName = ""; - let toolCallId = ""; - const args: Record = {}; - for (const f of decodeFields(variantBytes)) { - if (f.wireType !== 2) continue; - if (f.fieldNumber === MCA_TOOL_NAME) { - toolName = f.bytes.toString("utf8"); - } else if (f.fieldNumber === MCA_NAME && !toolName) { - // tool_name (5) takes precedence; fall back to name (1) - toolName = f.bytes.toString("utf8"); - } else if (f.fieldNumber === MCA_TOOL_CALL_ID) { - toolCallId = f.bytes.toString("utf8"); - } else if (f.fieldNumber === MCA_ARGS) { - // FieldsEntry { key (1): string, value (2): bytes } - let key = ""; - let valueBytes: Buffer | null = null; - for (const entry of decodeFields(f.bytes)) { - if (entry.fieldNumber === MAP_KEY && entry.wireType === 2) { - key = entry.bytes.toString("utf8"); - } else if (entry.fieldNumber === MAP_VALUE && entry.wireType === 2) { - valueBytes = entry.bytes; - } - } - if (key && valueBytes !== null) { - args[key] = decodeProtobufValue(valueBytes); - } - } - } - return { kind: "exec_mcp", execMsgId, execId, toolName, toolCallId, args }; - } - default: - // Unknown variant — return null so caller can keep buffering. - return null; - } +function decodeMcpMapEntry(payload: Buffer): { key: string; value: unknown } | null { + const fields = decodeFields(payload); + const key = findLengthDelimitedField(fields, MAP_KEY)?.toString("utf8") ?? ""; + const valueBytes = findLengthDelimitedField(fields, MAP_VALUE); + return key && valueBytes ? { key, value: decodeProtobufValue(valueBytes) } : null; +} + +function decodeMcpExecEvent(context: ExecEventContext): ExecServerEvent { + const fields = decodeFields(context.variantBytes); + const canonicalName = findLengthDelimitedField(fields, MCA_TOOL_NAME); + const fallbackName = findLengthDelimitedField(fields, MCA_NAME); + const toolName = (canonicalName ?? fallbackName)?.toString("utf8") ?? ""; + const toolCallId = findLengthDelimitedField(fields, MCA_TOOL_CALL_ID)?.toString("utf8") ?? ""; + const args: Record = {}; + for (const field of fields) { + if (field.fieldNumber !== MCA_ARGS || field.wireType !== WT_LEN) continue; + const entry = decodeMcpMapEntry(field.bytes); + if (entry) args[entry.key] = entry.value; } - return null; + return { + kind: "exec_mcp", + execMsgId: context.execMsgId, + execId: context.execId, + toolName, + toolCallId, + args, + }; +} + +const EXEC_EVENT_DECODERS: Partial> = { + [ESM_REQUEST_CONTEXT_ARGS]: ({ execMsgId, execId }) => ({ + kind: "exec_request_context", + execMsgId, + execId, + }), + [ESM_READ_ARGS]: (context) => createPathExecEvent("exec_read", context), + [ESM_WRITE_ARGS]: (context) => createPathExecEvent("exec_write", context), + [ESM_DELETE_ARGS]: (context) => createPathExecEvent("exec_delete", context), + [ESM_LS_ARGS]: (context) => createPathExecEvent("exec_ls", context), + [ESM_GREP_ARGS]: ({ execMsgId, execId }) => ({ kind: "exec_grep", execMsgId, execId }), + [ESM_DIAGNOSTICS_ARGS]: ({ execMsgId, execId }) => ({ + kind: "exec_diagnostics", + execMsgId, + execId, + }), + [ESM_SHELL_ARGS]: (context) => createShellExecEvent("exec_shell", context), + [ESM_SHELL_STREAM_ARGS]: (context) => createShellExecEvent("exec_shell_stream", context), + [ESM_BACKGROUND_SHELL_SPAWN]: (context) => createShellExecEvent("exec_bg_shell", context), + [ESM_FETCH_ARGS]: ({ execMsgId, execId, variantBytes }) => ({ + kind: "exec_fetch", + execMsgId, + execId, + url: decodeStringField(variantBytes, ARG_FETCH_URL), + }), + [ESM_WRITE_SHELL_STDIN_ARGS]: ({ execMsgId, execId }) => ({ + kind: "exec_write_shell_stdin", + execMsgId, + execId, + }), + [ESM_MCP_ARGS]: decodeMcpExecEvent, +}; + +function decodeExecEventContext( + payload: Buffer +): (ExecEventContext & { variantField: number }) | null { + const top = findField(decodeFields(payload), ASM_EXEC_SERVER_MESSAGE); + if (top?.wireType !== WT_LEN) return null; + + const fields = decodeFields(top.bytes); + const idField = findField(fields, ESM_ID); + const variant = fields.find( + (field) => field.wireType === WT_LEN && field.fieldNumber !== ESM_EXEC_ID + ); + if (!variant || variant.wireType !== WT_LEN) return null; + return { + execMsgId: idField?.wireType === WT_VARINT ? Number(idField.varint) : 0, + execId: findLengthDelimitedField(fields, ESM_EXEC_ID)?.toString("utf8") ?? "", + variantField: variant.fieldNumber, + variantBytes: variant.bytes, + }; +} + +export function decodeExecServerEvent(payload: Buffer): ExecServerEvent | null { + const context = decodeExecEventContext(payload); + if (!context) return null; + const decoder = EXEC_EVENT_DECODERS[context.variantField]; + return decoder?.(context) ?? null; } /** @@ -1316,6 +1288,87 @@ export function jsonSchemaToProtobufValue(json: unknown): Buffer { * Handles all six Value variants: null, number (double), string, bool, * struct (object), list (array). Unknown fields are skipped. */ +type ProtobufValueDecodeResult = { value: unknown; nextPos: number }; +type ProtobufValueDecoder = ( + buf: Buffer, + pos: number, + wireType: number +) => ProtobufValueDecodeResult; + +function readLengthDelimitedPayload( + buf: Buffer, + pos: number, + wireType: number +): { payload: Buffer; nextPos: number } | null { + if (wireType !== WT_LEN) return null; + const [len, afterLength] = decodeVarint(buf, pos); + const lenN = checkedLen(len, afterLength, buf); + return { + payload: buf.subarray(afterLength, afterLength + lenN), + nextPos: afterLength + lenN, + }; +} + +function decodeNullValue(buf: Buffer, pos: number, wireType: number): ProtobufValueDecodeResult { + const nextPos = wireType === WT_VARINT ? decodeVarint(buf, pos)[1] : pos; + return { value: null, nextPos }; +} + +function decodeNumberValue(buf: Buffer, pos: number, wireType: number): ProtobufValueDecodeResult { + const valid = wireType === 1 && pos + 8 <= buf.length; + return { value: valid ? buf.readDoubleLE(pos) : 0, nextPos: valid ? pos + 8 : pos }; +} + +function decodeStringValue(buf: Buffer, pos: number, wireType: number): ProtobufValueDecodeResult { + const decoded = readLengthDelimitedPayload(buf, pos, wireType); + return { + value: decoded?.payload.toString("utf8") ?? "", + nextPos: decoded?.nextPos ?? pos, + }; +} + +function decodeBoolValue(buf: Buffer, pos: number, wireType: number): ProtobufValueDecodeResult { + if (wireType !== WT_VARINT) return { value: false, nextPos: pos }; + const [value, nextPos] = decodeVarint(buf, pos); + return { value: value !== 0n, nextPos }; +} + +function decodeStructValue(buf: Buffer, pos: number, wireType: number): ProtobufValueDecodeResult { + const decoded = readLengthDelimitedPayload(buf, pos, wireType); + return { + value: decoded ? decodeProtobufStruct(decoded.payload) : {}, + nextPos: decoded?.nextPos ?? pos, + }; +} + +function decodeListValue(buf: Buffer, pos: number, wireType: number): ProtobufValueDecodeResult { + const decoded = readLengthDelimitedPayload(buf, pos, wireType); + return { + value: decoded ? decodeProtobufList(decoded.payload) : [], + nextPos: decoded?.nextPos ?? pos, + }; +} + +const PROTOBUF_VALUE_DECODERS: Partial> = { + [VAL_NULL]: decodeNullValue, + [VAL_NUMBER]: decodeNumberValue, + [VAL_STRING]: decodeStringValue, + [VAL_BOOL]: decodeBoolValue, + [VAL_STRUCT]: decodeStructValue, + [VAL_LIST]: decodeListValue, +}; + +function skipUnknownProtobufField(buf: Buffer, pos: number, wireType: number): number { + if (wireType === WT_VARINT) return decodeVarint(buf, pos)[1]; + if (wireType === WT_LEN) { + const [len, afterLength] = decodeVarint(buf, pos); + return afterLength + checkedLen(len, afterLength, buf); + } + if (wireType === 1) return pos + 8; + if (wireType === 5) return pos + 4; + return pos; +} + export function decodeProtobufValue(buf: Buffer): unknown { let pos = 0; while (pos < buf.length) { @@ -1323,97 +1376,26 @@ export function decodeProtobufValue(buf: Buffer): unknown { pos = np; const fieldNumber = Number(t >> 3n); const wireType = Number(t & 0x7n); - switch (fieldNumber) { - case VAL_NULL: { - if (wireType === WT_VARINT) { - [, pos] = decodeVarint(buf, pos); - } - return null; - } - case VAL_NUMBER: { - if (wireType === 1 && pos + 8 <= buf.length) { - const value = buf.readDoubleLE(pos); - pos += 8; - return value; - } - return 0; - } - case VAL_STRING: { - if (wireType === WT_LEN) { - const [len, np2] = decodeVarint(buf, pos); - pos = np2; - const lenN = checkedLen(len, pos, buf); - const value = buf.subarray(pos, pos + lenN).toString("utf8"); - pos += lenN; - return value; - } - return ""; - } - case VAL_BOOL: { - if (wireType === WT_VARINT) { - const [val, np2] = decodeVarint(buf, pos); - pos = np2; - return val !== 0n; - } - return false; - } - case VAL_STRUCT: { - if (wireType === WT_LEN) { - const [len, np2] = decodeVarint(buf, pos); - pos = np2; - const lenN = checkedLen(len, pos, buf); - const inner = buf.subarray(pos, pos + lenN); - pos += lenN; - return decodeProtobufStruct(inner); - } - return {}; - } - case VAL_LIST: { - if (wireType === WT_LEN) { - const [len, np2] = decodeVarint(buf, pos); - pos = np2; - const lenN = checkedLen(len, pos, buf); - const inner = buf.subarray(pos, pos + lenN); - pos += lenN; - return decodeProtobufList(inner); - } - return []; - } - default: - // Skip unknown field - if (wireType === WT_VARINT) { - [, pos] = decodeVarint(buf, pos); - } else if (wireType === WT_LEN) { - const [len, np2] = decodeVarint(buf, pos); - pos = np2; - pos += Number(len); - } else if (wireType === 1) { - pos += 8; - } else if (wireType === 5) { - pos += 4; - } - } + const decoder = PROTOBUF_VALUE_DECODERS[fieldNumber]; + if (decoder) return decoder(buf, pos, wireType).value; + pos = skipUnknownProtobufField(buf, pos, wireType); } return null; } +function decodeProtobufStructEntry(payload: Buffer): { key: string; value: unknown } | null { + const fields = decodeFields(payload); + const key = findLengthDelimitedField(fields, MAP_KEY)?.toString("utf8") ?? ""; + const valueBytes = findLengthDelimitedField(fields, MAP_VALUE); + return key && valueBytes ? { key, value: decodeProtobufValue(valueBytes) } : null; +} + function decodeProtobufStruct(buf: Buffer): Record { const result: Record = {}; - for (const f of decodeFields(buf)) { - if (f.fieldNumber === STRUCT_FIELDS && f.wireType === 2) { - let key = ""; - let valueBytes: Buffer | null = null; - for (const entry of decodeFields(f.bytes)) { - if (entry.fieldNumber === MAP_KEY && entry.wireType === 2) { - key = entry.bytes.toString("utf8"); - } else if (entry.fieldNumber === MAP_VALUE && entry.wireType === 2) { - valueBytes = entry.bytes; - } - } - if (key && valueBytes) { - result[key] = decodeProtobufValue(valueBytes); - } - } + for (const field of decodeFields(buf)) { + if (field.fieldNumber !== STRUCT_FIELDS || field.wireType !== WT_LEN) continue; + const entry = decodeProtobufStructEntry(field.bytes); + if (entry) result[entry.key] = entry.value; } return result; } @@ -1477,6 +1459,39 @@ export type ChatMessage = { tool_call_id?: string; }; +function messageContentToText(content: ChatMessage["content"]): string { + if (typeof content === "string") return content; + if (!Array.isArray(content)) return ""; + return content + .map((part) => (typeof part?.text === "string" ? part.text : "")) + .filter(Boolean) + .join("\n"); +} + +function assistantMessageLines(message: ChatMessage, text: string): string[] { + const lines = text ? [`Assistant: ${text}`] : []; + for (const toolCall of message.tool_calls ?? []) { + const name = toolCall.function?.name ?? "(unknown)"; + const args = toolCall.function?.arguments ?? ""; + lines.push(`Assistant called tool ${name} (${toolCall.id}) with arguments: ${args}`); + } + return lines; +} + +function chatMessageLines(message: ChatMessage): string[] { + const text = messageContentToText(message.content); + if (message.role === "user") return text ? [`User: ${text}`] : []; + if (message.role === "assistant") return assistantMessageLines(message, text); + if (message.role === "tool") { + return [`Tool result (${message.tool_call_id ?? "(unknown)"}): ${text}`]; + } + return text ? [`${message.role}: ${text}`] : []; +} + +function joinSystemText(systemTexts: string[], body: string): string { + return systemTexts.length > 0 ? `${systemTexts.join("\n\n")}\n\n${body}` : body; +} + /** * Flatten an OpenAI-shaped message list down to a single user-text string * suitable for cursor's UserMessage. The agent endpoint expects ONE user @@ -1490,57 +1505,23 @@ export type ChatMessage = { export function flattenMessages(messages: ChatMessage[]): string { if (!Array.isArray(messages) || messages.length === 0) return ""; - const partsToText = (content: ChatMessage["content"]): string => { - if (typeof content === "string") return content; - if (content == null) return ""; - if (!Array.isArray(content)) return ""; - return content - .map((p) => (typeof p?.text === "string" ? p.text : "")) - .filter(Boolean) - .join("\n"); - }; - // System instructions go first as a labeled prefix. (The cursor executor // routes system messages through the KV blob channel — see Phase 7 — but // this branch is kept for non-cursor callers.) const systemTexts = messages .filter((m) => m.role === "system") - .map((m) => partsToText(m.content)) + .map((m) => messageContentToText(m.content)) .filter(Boolean); const turn = messages.filter((m) => m.role !== "system"); // Single-user-message fast path (no tool_calls, no labels). if (turn.length === 1 && turn[0].role === "user" && !turn[0].tool_calls) { - const userText = partsToText(turn[0].content); - return systemTexts.length > 0 ? `${systemTexts.join("\n\n")}\n\n${userText}` : userText; + return joinSystemText(systemTexts, messageContentToText(turn[0].content)); } // Multi-turn / tool-using format. Each message is labeled. Tool calls // and tool results get their own labeled lines. - const lines: string[] = []; - for (const m of turn) { - const text = partsToText(m.content); - if (m.role === "user") { - if (text) lines.push(`User: ${text}`); - } else if (m.role === "assistant") { - if (text) lines.push(`Assistant: ${text}`); - if (Array.isArray(m.tool_calls)) { - for (const tc of m.tool_calls) { - const args = tc.function?.arguments ?? ""; - lines.push( - `Assistant called tool ${tc.function?.name ?? "(unknown)"} ` + - `(${tc.id}) with arguments: ${args}` - ); - } - } - } else if (m.role === "tool") { - const callId = m.tool_call_id ?? "(unknown)"; - lines.push(`Tool result (${callId}): ${text}`); - } else { - if (text) lines.push(`${m.role}: ${text}`); - } - } - const labelled = lines.join("\n\n"); - return systemTexts.length > 0 ? `${systemTexts.join("\n\n")}\n\n${labelled}` : labelled; + const labelled = turn.flatMap(chatMessageLines).join("\n\n"); + return joinSystemText(systemTexts, labelled); } diff --git a/open-sse/utils/cursorAgentProtobuf/requestedModelParameters.ts b/open-sse/utils/cursorAgentProtobuf/requestedModelParameters.ts new file mode 100644 index 0000000000..da431746eb --- /dev/null +++ b/open-sse/utils/cursorAgentProtobuf/requestedModelParameters.ts @@ -0,0 +1,113 @@ +export const CURSOR_EFFORT_SUFFIXES = ["low", "medium", "high", "xhigh", "max"] as const; + +type CursorRequestedModel = { + modelId: string; + parameters: Array<{ id: string; value: string }>; +}; + +const CURSOR_ONE_MILLION_SUFFIX = "-1m"; +const CURSOR_GPT_REASONING_LEVELS = ["none", ...CURSOR_EFFORT_SUFFIXES] as const; + +const CURSOR_CLAUDE_ONE_MILLION_FAMILIES = [ + { + legacyPrefix: "claude-fable-5-1", + modelId: "claude-fable-5-1", + supportsFast: false, + trailingThinking: false, + }, + { + legacyPrefix: "claude-opus-5", + modelId: "claude-opus-5", + supportsFast: true, + trailingThinking: false, + }, + { + legacyPrefix: "claude-opus-4-8", + modelId: "claude-opus-4-8", + supportsFast: true, + trailingThinking: false, + }, + { + legacyPrefix: "claude-sonnet-5", + modelId: "claude-sonnet-5", + supportsFast: false, + trailingThinking: false, + }, + { + legacyPrefix: "claude-4.6-sonnet", + modelId: "claude-sonnet-4-6", + supportsFast: false, + trailingThinking: true, + }, +] as const; + +type CursorClaudeOneMillionFamily = (typeof CURSOR_CLAUDE_ONE_MILLION_FAMILIES)[number]; + +function isCursorEffort(value: string): value is (typeof CURSOR_EFFORT_SUFFIXES)[number] { + return CURSOR_EFFORT_SUFFIXES.some((effort) => effort === value); +} + +function resolveGptOneMillionContextModel(legacyId: string): CursorRequestedModel | null { + const match = /^(gpt-5\.6-(?:sol|terra|luna))-(none|low|medium|high|xhigh|max)$/.exec(legacyId); + if (!match) return null; + + const [, modelId, reasoning] = match; + if (!CURSOR_GPT_REASONING_LEVELS.some((level) => level === reasoning)) return null; + return { + modelId, + parameters: [ + { id: "context", value: "1m" }, + { id: "reasoning", value: reasoning }, + { id: "fast", value: "false" }, + ], + }; +} + +function resolveClaudeOneMillionVariant( + legacyId: string, + family: CursorClaudeOneMillionFamily +): CursorRequestedModel | null { + const prefix = `${family.legacyPrefix}-`; + if (!legacyId.startsWith(prefix)) return null; + + let variant = legacyId.slice(prefix.length); + const fast = variant.endsWith("-fast"); + if (fast) variant = variant.slice(0, -"-fast".length); + if (fast && !family.supportsFast) return null; + + const trailingThinking = family.trailingThinking && variant.endsWith("-thinking"); + const leadingThinking = !family.trailingThinking && variant.startsWith("thinking-"); + if (trailingThinking) variant = variant.slice(0, -"-thinking".length); + if (leadingThinking) variant = variant.slice("thinking-".length); + if (!isCursorEffort(variant)) return null; + + const parameters = [ + { id: "thinking", value: String(trailingThinking || leadingThinking) }, + { id: "context", value: "1m" }, + { id: "effort", value: variant }, + ]; + if (family.supportsFast) parameters.push({ id: "fast", value: String(fast) }); + return { modelId: family.modelId, parameters }; +} + +function resolveClaudeOneMillionContextModel(legacyId: string): CursorRequestedModel | null { + for (const family of CURSOR_CLAUDE_ONE_MILLION_FAMILIES) { + const resolved = resolveClaudeOneMillionVariant(legacyId, family); + if (resolved) return resolved; + } + return null; +} + +/** + * Cursor reuses each legacy slug for both its default and 1M context variants, + * so the public catalog adds a terminal `-1m` discriminator. Translate that + * synthetic id to the canonical wire model plus the complete parameter set + * reported by Cursor's AvailableModels metadata. + */ +export function resolveOneMillionContextModel(normalized: string): CursorRequestedModel | null { + if (!normalized.endsWith(CURSOR_ONE_MILLION_SUFFIX)) return null; + const legacyId = normalized.slice(0, -CURSOR_ONE_MILLION_SUFFIX.length); + return ( + resolveGptOneMillionContextModel(legacyId) ?? resolveClaudeOneMillionContextModel(legacyId) + ); +} diff --git a/open-sse/utils/registeredEffortVariants.ts b/open-sse/utils/registeredEffortVariants.ts index 06e2d5dfc6..2cef8d6721 100644 --- a/open-sse/utils/registeredEffortVariants.ts +++ b/open-sse/utils/registeredEffortVariants.ts @@ -14,10 +14,9 @@ export function getRegisteredProviderEffortBaseModelId( modelId: string ): string | null { const providerModels = getProviderModels(providerId); + const registeredVariant = providerModels.find((candidate) => candidate.id === modelId); - if (!providerModels.some((candidate) => candidate.id === modelId)) { - return null; - } + if (!registeredVariant) return null; for (const effort of REGISTERED_EFFORT_SUFFIXES) { const suffix = `-${effort}`; @@ -25,7 +24,15 @@ export function getRegisteredProviderEffortBaseModelId( const baseModelId = modelId.slice(0, -suffix.length); - return providerModels.some((candidate) => candidate.id === baseModelId) ? baseModelId : null; + if (providerModels.some((candidate) => candidate.id === baseModelId)) return baseModelId; + + // Curated providers may intentionally expose only useful variants while the + // authoritative live catalog exposes their unsuffixed wire model. The registry + // declaration is the proof; never infer this relationship from spelling alone. + const declaredLiveBase = registeredVariant.liveCatalogIds?.find( + (candidate) => candidate === baseModelId || !candidate.endsWith(`-${effort}`) + ); + return declaredLiveBase ?? null; } return null; diff --git a/src/lib/db/models/activeSyncedCatalog.ts b/src/lib/db/models/activeSyncedCatalog.ts index 981175219f..18ade8260b 100644 --- a/src/lib/db/models/activeSyncedCatalog.ts +++ b/src/lib/db/models/activeSyncedCatalog.ts @@ -1,5 +1,6 @@ import { providerUsesAuthoritativeLiveCatalog } from "@omniroute/open-sse/config/providerRegistry"; import { PROVIDER_ID_TO_ALIAS } from "@omniroute/open-sse/config/providerModels.ts"; +import { ensureCursorAutoCatalogEntry } from "@/lib/providerModels/cursorAutoCatalog"; import { getSyncedAvailableModels, getSyncedAvailableModelsByConnection, @@ -76,6 +77,19 @@ function collectModelsForConnections( return Array.from(models.values()); } +function enrichCursorCatalog( + providerId: string, + models: SyncedAvailableModel[] +): SyncedAvailableModel[] { + // An empty sync means discovery has not completed (or failed). Do not let the + // synthetic Cursor auto-router rows turn that empty state into an authoritative + // catalog, otherwise every built-in model is incorrectly marked unavailable. + if (models.length === 0) return models; + return providerId === "cursor" || providerId === "cursor-api" + ? ensureCursorAutoCatalogEntry(models) + : models; +} + /** * Return the unioned synced catalog belonging only to active connections. * @@ -105,7 +119,10 @@ export async function getActiveSyncedCatalog(providerId: string): Promise connection !== null) .map((connection) => connection.id); - const models = collectModelsForConnections(modelsByConnection, activeConnectionIds); + const models = enrichCursorCatalog( + storedProviderId, + collectModelsForConnections(modelsByConnection, activeConnectionIds) + ); if (models.length > 0) { return { authoritative: providerUsesAuthoritativeLiveCatalog(providerId), @@ -125,7 +142,13 @@ export async function getActiveSyncedCatalog(providerId: string): Promise { const modelsByConnection = await getSyncedAvailableModelsByConnection(providerId); - const models = collectModelsForConnections(modelsByConnection, connectionIds); + const models = enrichCursorCatalog( + providerId, + collectModelsForConnections(modelsByConnection, connectionIds) + ); if (models.length > 0) { result[providerId] = models; diff --git a/src/lib/providerModels/cursorAutoCatalog.ts b/src/lib/providerModels/cursorAutoCatalog.ts index a9c0f2dca2..ddea216560 100644 --- a/src/lib/providerModels/cursorAutoCatalog.ts +++ b/src/lib/providerModels/cursorAutoCatalog.ts @@ -8,7 +8,6 @@ export type CursorAutoCatalogEntry = { id: string; name: string; owned_by?: string; - [key: string]: unknown; }; export const CURSOR_AUTO_ROUTER_VARIANT_IDS = [ @@ -26,10 +25,56 @@ const CURSOR_AUTO_ROUTER_VARIANT_NAMES: Record< "auto-intelligence": "Auto (intelligence)", }; +const CURSOR_ONE_MILLION_CONTEXT = 1_000_000; +const CURSOR_CONTEXT_EFFORT = "(?:low|medium|high|xhigh|max)"; +const CURSOR_ONE_MILLION_MODEL_PATTERNS = [ + new RegExp(`^claude-fable-5-1-thinking-${CURSOR_CONTEXT_EFFORT}$`), + new RegExp(`^claude-opus-5-(?:thinking-)?${CURSOR_CONTEXT_EFFORT}(?:-fast)?$`), + new RegExp(`^claude-opus-4-8-(?:thinking-)?${CURSOR_CONTEXT_EFFORT}(?:-fast)?$`), + new RegExp(`^claude-sonnet-5-(?:thinking-)?${CURSOR_CONTEXT_EFFORT}$`), + new RegExp(`^claude-4\\.6-sonnet-${CURSOR_CONTEXT_EFFORT}(?:-thinking)?$`), + new RegExp(`^gpt-5\\.6-(?:sol|terra|luna)-(?:none|${CURSOR_CONTEXT_EFFORT})$`), +] as const; + +const CURSOR_CONTEXT_FAMILY_NAMES = [ + "Claude Fable 5.1", + "Claude Opus 5", + "Claude Opus 4.8", + "Claude Sonnet 5", + "Claude Sonnet 4.6", + "GPT-5.6 Sol", + "GPT-5.6 Terra", + "GPT-5.6 Luna", +] as const; + +function supportsCursorOneMillionContext(id: string): boolean { + return CURSOR_ONE_MILLION_MODEL_PATTERNS.some((pattern) => pattern.test(id)); +} + +function oneMillionDisplayName(name: string): string { + const family = CURSOR_CONTEXT_FAMILY_NAMES.find((candidate) => name.startsWith(candidate)); + return family ? `${family} 1M${name.slice(family.length)}` : `${name} 1M`; +} + /** Cursor auto-router: catalog id `auto`, wire id `default`. Always keep `auto` visible. */ export function ensureCursorAutoCatalogEntry(models: T[]): T[] { const byId = new Map(models.map((m) => [m.id, m])); - const out = [...models]; + const out: T[] = []; + + for (const model of models) { + const oneMillionId = `${model.id}-1m`; + if (supportsCursorOneMillionContext(model.id) && !byId.has(oneMillionId)) { + const oneMillionEntry = { + ...model, + id: oneMillionId, + name: oneMillionDisplayName(model.name), + contextLength: CURSOR_ONE_MILLION_CONTEXT, + } as T; + out.push(oneMillionEntry); + byId.set(oneMillionId, oneMillionEntry); + } + out.push(model); + } if (!byId.has("auto")) { const defaultEntry = byId.get("default"); diff --git a/src/lib/providerModels/cursorAvailableModels.ts b/src/lib/providerModels/cursorAvailableModels.ts index 0fdc3d1939..8c4eb7b04a 100644 --- a/src/lib/providerModels/cursorAvailableModels.ts +++ b/src/lib/providerModels/cursorAvailableModels.ts @@ -9,8 +9,11 @@ import { humanizeCursorModelId, type CursorAgentModelEntry, } from "@/lib/providerModels/cursorAgent"; +import { ensureCursorAutoCatalogEntry } from "@/lib/providerModels/cursorAutoCatalog"; import { getConsistentMachineId } from "@/shared/utils/machineId"; +export { ensureCursorAutoCatalogEntry } from "@/lib/providerModels/cursorAutoCatalog"; + export type FetchCursorAvailableModelsOptions = { accessToken: string; machineId?: string | null; @@ -40,6 +43,45 @@ function pickModelName(entry: Record, id: string): string { return humanizeCursorModelId(id); } +function collectArrays(record: Record, keys: string[]): unknown[] { + return keys.flatMap((key) => (Array.isArray(record[key]) ? record[key] : [])); +} + +function collectModelCandidates(payload: unknown): unknown[] { + const root = asRecord(payload) ?? {}; + const candidates = collectArrays(root, [ + "models", + "availableModels", + "available_models", + "model", + ]); + const nestedModels = asRecord(root.models); + if (nestedModels) candidates.push(...collectArrays(nestedModels, ["models", "items", "list"])); + if (Array.isArray(payload)) candidates.push(...payload); + return candidates; +} + +function isUnavailableModel(entry: Record): boolean { + return ( + entry.disabled === true || + entry.isDisabled === true || + entry.usable === false || + entry.isUsable === false + ); +} + +function normalizeModelCandidate(item: unknown): CursorAgentModelEntry | null { + if (typeof item === "string") { + const id = item.trim(); + return id ? { id, name: humanizeCursorModelId(id), owned_by: "cursor" } : null; + } + + const entry = asRecord(item); + if (!entry || isUnavailableModel(entry)) return null; + const id = pickModelId(entry); + return id ? { id, name: pickModelName(entry, id), owned_by: "cursor" } : null; +} + /** * Normalize AvailableModels JSON (Connect JSON or protobuf-json) into catalog rows. * Exported for unit tests. @@ -48,99 +90,18 @@ function pickModelName(entry: Record, id: string): string { * only). OmniRoute clients request `cu/auto`; resolveRequestedModel maps it to `default`. */ export function normalizeCursorAvailableModelsPayload(payload: unknown): CursorAgentModelEntry[] { - const root = asRecord(payload) ?? {}; - const candidates: unknown[] = []; - - for (const key of ["models", "availableModels", "available_models", "model"]) { - const v = root[key]; - if (Array.isArray(v)) candidates.push(...v); - } - - // Some Connect JSON responses nest under `models.models` or similar - const nestedModels = asRecord(root.models); - if (nestedModels) { - for (const key of ["models", "items", "list"]) { - const v = nestedModels[key]; - if (Array.isArray(v)) candidates.push(...v); - } - } - - if (Array.isArray(payload)) candidates.push(...payload); - const seen = new Set(); const out: CursorAgentModelEntry[] = []; - for (const item of candidates) { - if (typeof item === "string" && item.trim()) { - const id = item.trim(); - if (seen.has(id)) continue; - seen.add(id); - out.push({ id, name: humanizeCursorModelId(id), owned_by: "cursor" }); - continue; - } - const rec = asRecord(item); - if (!rec) continue; - const id = pickModelId(rec); - if (!id || seen.has(id)) continue; - // Prefer usable / non-disabled when flags exist - if (rec.disabled === true || rec.isDisabled === true) continue; - if (rec.usable === false || rec.isUsable === false) continue; - seen.add(id); - out.push({ id, name: pickModelName(rec, id), owned_by: "cursor" }); + for (const item of collectModelCandidates(payload)) { + const model = normalizeModelCandidate(item); + if (!model || seen.has(model.id)) continue; + seen.add(model.id); + out.push(model); } return ensureCursorAutoCatalogEntry(out); } -/** OpenCodex-style Cursor Router optimization modes (catalog ids). */ -export const CURSOR_AUTO_ROUTER_VARIANT_IDS = [ - "auto-cost", - "auto-balance", - "auto-intelligence", -] as const; - -const CURSOR_AUTO_ROUTER_VARIANT_NAMES: Record< - (typeof CURSOR_AUTO_ROUTER_VARIANT_IDS)[number], - string -> = { - "auto-cost": "Auto (cost)", - "auto-balance": "Auto (balance)", - "auto-intelligence": "Auto (intelligence)", -}; - -/** Cursor auto-router: catalog id `auto`, wire id `default`. Always keep `auto` visible. */ -export function ensureCursorAutoCatalogEntry( - models: CursorAgentModelEntry[] -): CursorAgentModelEntry[] { - const byId = new Map(models.map((m) => [m.id, m])); - const out = [...models]; - - if (!byId.has("auto")) { - const defaultEntry = byId.get("default"); - const autoEntry: CursorAgentModelEntry = { - id: "auto", - name: defaultEntry?.name || "Auto (current, default)", - owned_by: "cursor", - }; - // Prefer `auto` as the public id; keep `default` for wire-compat listings. - out.unshift(autoEntry); - byId.set("auto", autoEntry); - } - - // Always expose Cost/Balance/Intelligence router modes (OpenCodex CURSOR_ROUTER_MODEL_IDS). - for (const id of CURSOR_AUTO_ROUTER_VARIANT_IDS) { - if (byId.has(id)) continue; - const entry: CursorAgentModelEntry = { - id, - name: CURSOR_AUTO_ROUTER_VARIANT_NAMES[id], - owned_by: "cursor", - }; - out.push(entry); - byId.set(id, entry); - } - - return out; -} - export async function fetchCursorAvailableModels( options: FetchCursorAvailableModelsOptions ): Promise { diff --git a/src/lib/providers/staticModels.ts b/src/lib/providers/staticModels.ts index 63eebdce28..5bb6598db5 100644 --- a/src/lib/providers/staticModels.ts +++ b/src/lib/providers/staticModels.ts @@ -35,6 +35,7 @@ const STATIC_MODEL_PROVIDERS: Record Array<{ id: string; name: str ], antigravity: () => ANTIGRAVITY_PUBLIC_MODELS.map((model) => ({ ...model })), claude: () => [ + { id: "claude-fable-5-1", name: "Claude Fable 5.1" }, { id: "claude-fable-5", name: "Claude Fable 5" }, { id: "claude-opus-5", name: "Claude Opus 5" }, { id: "claude-opus-4-8", name: "Claude Opus 4.8" }, diff --git a/src/shared/constants/cliTools.ts b/src/shared/constants/cliTools.ts index 95d956356b..01401ea986 100644 --- a/src/shared/constants/cliTools.ts +++ b/src/shared/constants/cliTools.ts @@ -45,7 +45,7 @@ export const CLI_TOOLS: Record = { name: "Claude Fable", alias: "fable", envKey: "ANTHROPIC_DEFAULT_FABLE_MODEL", - defaultValue: _cc.fable ? `cc/${_cc.fable}` : "cc/claude-fable-5", + defaultValue: _cc.fable ? `cc/${_cc.fable}` : "cc/claude-fable-5-1", isTopLevel: true, }, { diff --git a/src/shared/constants/modelSpecs.ts b/src/shared/constants/modelSpecs.ts index fcc96c8e9b..c8b6faaf4b 100644 --- a/src/shared/constants/modelSpecs.ts +++ b/src/shared/constants/modelSpecs.ts @@ -24,9 +24,13 @@ export interface ModelSpec { // Model ONLY supports adaptive thinking: manual extended thinking was removed. Sending // `thinking.type:"enabled"` or any `thinking.budget_tokens` returns HTTP 400; reasoning // is steered exclusively by `output_config.effort` (low/medium/high/xhigh/max). True for - // Claude Opus 4.7 and later (Opus 4.7/4.8/5, Fable 5). Per Anthropic's migration guide, + // Claude Opus 4.7 and later (Opus 4.7/4.8/5, Fable 5/5.1). Per Anthropic's migration guide, // any request that tries to set a fixed thinking budget gets a 400 error. adaptiveThinkingOnly?: boolean; + // The model rejects tool_choice values that require a tool call. Keep tools available, + // but normalize a forced choice to the default auto behavior before dispatch. Fable 5.1 always runs + // adaptive thinking, so forced tool use cannot be combined with any valid request. + rejectsForcedToolChoice?: boolean; // Highest effort accepted while `thinking.type:"disabled"` is present. Claude Opus 5 // rejects disabled thinking with xhigh/max, while accepting it through high. maxEffortWhenThinkingDisabled?: "high"; @@ -371,6 +375,21 @@ export const MODEL_SPECS: Record = { aliases: BEDROCK_CLAUDE_ALIASES("claude-opus-4-7", "claude-opus-4.7"), }, + // ── Claude Fable 5.1 ──────────────────────────────────────────── + "claude-fable-5-1": { + maxOutputTokens: 128000, + contextWindow: 1000000, + defaultThinkingBudget: 32000, + thinkingBudgetCap: 120000, + supportsThinking: true, + supportsTools: true, + supportsVision: true, + rejectsThinkingDisabled: true, + adaptiveThinkingOnly: true, + rejectsForcedToolChoice: true, + aliases: BEDROCK_CLAUDE_ALIASES("claude-fable-5-1"), + }, + // ── Claude Fable 5 ────────────────────────────────────────────── "claude-fable-5": { maxOutputTokens: 128000, @@ -849,9 +868,39 @@ export function normalizeThinkingForModel>( getModelSpec(modelId)?.rejectsThinkingDisabled ) { const { thinking: _omitted, ...rest } = body as Record; - return rest as T; + return normalizeForcedToolChoiceForModel(rest as T, modelId); } - return body; + return normalizeForcedToolChoiceForModel(body, modelId); +} + +/** + * Normalize tool-choice constraints that a resolved model cannot accept. + * + * Claude Fable 5.1 always uses adaptive thinking and rejects tool choices that force + * either any tool or one named tool. Preserve the declared tools and every unrelated + * request field, but drop the choice to select the default `auto` behavior so routing a + * request to Fable 5.1 does not turn a recoverable preference into an upstream 400. + */ +export function normalizeForcedToolChoiceForModel>( + body: T, + modelId: string +): T { + if (!getModelSpec(modelId)?.rejectsForcedToolChoice) return body; + + const toolChoice = body.tool_choice; + const forced = + toolChoice === "required" || + toolChoice === "any" || + (toolChoice !== null && + typeof toolChoice === "object" && + !Array.isArray(toolChoice) && + ["any", "tool", "function"].includes( + String((toolChoice as Record).type || "").toLowerCase() + )); + if (!forced) return body; + + const { tool_choice: _omitted, ...rest } = body; + return rest as T; } export function capMaxOutputTokens(modelId: string, requested?: number): number | undefined { diff --git a/src/shared/constants/pricing/default-pricing.ts b/src/shared/constants/pricing/default-pricing.ts index 08b84a86e9..a36654bd06 100644 --- a/src/shared/constants/pricing/default-pricing.ts +++ b/src/shared/constants/pricing/default-pricing.ts @@ -7,10 +7,12 @@ import { DEFAULT_PRICING_OAUTH } from "./oauth-subscriptions"; import { DEFAULT_PRICING_FRONTIER } from "./frontier-labs"; import { DEFAULT_PRICING_INFERENCE } from "./inference-hosts"; import { DEFAULT_PRICING_REGIONAL } from "./regional"; +import { DEFAULT_PRICING_DEVIN } from "./devin"; export const DEFAULT_PRICING = { ...DEFAULT_PRICING_OAUTH, ...DEFAULT_PRICING_FRONTIER, ...DEFAULT_PRICING_INFERENCE, ...DEFAULT_PRICING_REGIONAL, + ...DEFAULT_PRICING_DEVIN, }; diff --git a/src/shared/constants/pricing/devin.ts b/src/shared/constants/pricing/devin.ts new file mode 100644 index 0000000000..36a0e49b3a --- /dev/null +++ b/src/shared/constants/pricing/devin.ts @@ -0,0 +1,142 @@ +type DevinTokenPricing = { + input: number; + cached: number; + output: number; +}; + +const QUALITY_EFFORTS = ["max", "xhigh", "high", "medium", "low"] as const; +const GPT_EFFORTS = ["max", "xhigh", "high", "medium", "low", "none"] as const; + +function variantIds(base: string, efforts: readonly string[]): string[] { + return efforts.map((effort) => `${base}-${effort}`); +} + +function fastVariantIds(base: string): string[] { + return QUALITY_EFFORTS.map((effort) => `${base}-${effort}-fast`); +} + +function priorityVariantIds(base: string): string[] { + return GPT_EFFORTS.map((effort) => `${base}-${effort}-priority`); +} + +function priced(ids: readonly string[], pricing: DevinTokenPricing) { + return Object.fromEntries(ids.map((id) => [id, pricing])); +} + +const CLAUDE_FABLE_5_1 = { input: 10, cached: 0.25, output: 50 }; +const CLAUDE_OPUS = { input: 5, cached: 0.5, output: 25 }; +const CLAUDE_OPUS_FAST = { input: 10, cached: 1, output: 50 }; +const CLAUDE_SONNET_5 = { input: 2, cached: 0.2, output: 10 }; +const CLAUDE_SONNET_4_6 = { input: 3, cached: 0.3, output: 15 }; +const CLAUDE_HAIKU_4_5 = { input: 1, cached: 0.1, output: 5 }; + +const GPT_5_6_SOL = { input: 4, cached: 0.4, output: 20 }; +const GPT_5_6_SOL_FAST = { input: 8, cached: 0.8, output: 40 }; +const GPT_5_6_TERRA = { input: 2, cached: 0.2, output: 12 }; +const GPT_5_6_TERRA_FAST = { input: 4, cached: 0.4, output: 24 }; +const GPT_5_6_LUNA = { input: 0.2, cached: 0.02, output: 1.2 }; +const GPT_5_6_LUNA_FAST = { input: 0.4, cached: 0.04, output: 2.4 }; + +/** + * Exact per-UID rates returned by Devin's authenticated live catalog on + * 2026-09-02. Rates are USD per one million tokens. + */ +export const DEVIN_MODEL_PRICING: Record = { + ...priced(variantIds("claude-fable-5-1", QUALITY_EFFORTS), CLAUDE_FABLE_5_1), + ...priced(variantIds("claude-opus-5", QUALITY_EFFORTS), CLAUDE_OPUS), + ...priced(fastVariantIds("claude-opus-5"), CLAUDE_OPUS_FAST), + ...priced(variantIds("claude-opus-4-8", QUALITY_EFFORTS), CLAUDE_OPUS), + ...priced(fastVariantIds("claude-opus-4-8"), CLAUDE_OPUS_FAST), + ...priced(variantIds("claude-sonnet-5", QUALITY_EFFORTS), CLAUDE_SONNET_5), + ...priced( + [ + "claude-sonnet-4-6", + "claude-sonnet-4-6-thinking", + "claude-sonnet-4-6-1m", + "claude-sonnet-4-6-thinking-1m", + ], + CLAUDE_SONNET_4_6 + ), + MODEL_PRIVATE_11: CLAUDE_HAIKU_4_5, + + ...priced(variantIds("gpt-5-6-sol", GPT_EFFORTS), GPT_5_6_SOL), + ...priced(priorityVariantIds("gpt-5-6-sol"), GPT_5_6_SOL_FAST), + ...priced(variantIds("gpt-5-6-terra", GPT_EFFORTS), GPT_5_6_TERRA), + ...priced(priorityVariantIds("gpt-5-6-terra"), GPT_5_6_TERRA_FAST), + ...priced(variantIds("gpt-5-6-luna", GPT_EFFORTS), GPT_5_6_LUNA), + ...priced(priorityVariantIds("gpt-5-6-luna"), GPT_5_6_LUNA_FAST), + + ...priced(variantIds("kimi-k3", ["max", "high", "low"]), { + input: 3, + cached: 0.3, + output: 15, + }), + "kimi-k2-7": { input: 0.95, cached: 0.19, output: 4 }, + ...priced(variantIds("glm-5-3", ["max", "high", "low"]), { + input: 1.4, + cached: 0.26, + output: 4.4, + }), + ...priced(variantIds("glm-5-3-flash", ["max", "high", "low"]), { + input: 0.15, + cached: 0.03, + output: 0.5, + }), + + ...priced(["swe-1-7", "swe-1-7-medium"], { + input: 0.5, + cached: 0.2, + output: 2.5, + }), + ...priced(["swe-1-7-lightning", "swe-1-7-lightning-medium"], { + input: 2.5, + cached: 1, + output: 12.5, + }), + adaptive: { input: 0.5, cached: 0.1, output: 2 }, + ...priced(variantIds("grok-4-6", ["xhigh", "high", "medium", "low"]), { + input: 2, + cached: 0.3, + output: 6, + }), + ...priced(variantIds("inkling", ["max", "xhigh", "high", "medium", "low", "none"]), { + input: 1.4, + cached: 0.26, + output: 4.4, + }), + ...priced(variantIds("deepseek-v4-flash", ["max", "high", "low"]), { + input: 0.14, + cached: 0.03, + output: 0.28, + }), + ...priced(variantIds("nemotron-3-ultra", ["high", "medium", "none"]), { + input: 0.6, + cached: 0.12, + output: 2.4, + }), + ...priced(variantIds("gemini-3-7-flash", ["high", "medium", "low"]), { + input: 1.5, + cached: 0.15, + output: 7.5, + }), + ...priced(variantIds("gemini-3-1-pro", ["high", "low"]), { + input: 2, + cached: 0.2, + output: 12, + }), + ...priced(variantIds("deepseek-v4-pro", ["max", "high", "low"]), { + input: 1.32, + cached: 0.04, + output: 3.96, + }), +}; + +// Each transport gets its own provider namespace. They share today's upstream +// rate snapshot, but can diverge independently if Devin changes one channel. +export const DEFAULT_PRICING_DEVIN = { + "devin-cli": { ...DEVIN_MODEL_PRICING }, + dv: { ...DEVIN_MODEL_PRICING }, + "devin-desktop": { ...DEVIN_MODEL_PRICING }, + "devin-cli-agentic": { ...DEVIN_MODEL_PRICING }, + dva: { ...DEVIN_MODEL_PRICING }, +}; diff --git a/src/shared/constants/pricing/frontier-labs.ts b/src/shared/constants/pricing/frontier-labs.ts index d20a187720..b8a80d3bf1 100644 --- a/src/shared/constants/pricing/frontier-labs.ts +++ b/src/shared/constants/pricing/frontier-labs.ts @@ -8,6 +8,7 @@ import { GPT_5_6_LUNA_PRICING, GPT_5_6_SOL_PRICING, GPT_5_6_TERRA_PRICING, + CLAUDE_FABLE_5_1_PRICING, CLAUDE_FABLE_5_PRICING, CLAUDE_OPUS_5_PRICING, CLAUDE_OPUS_4_PRICING, @@ -213,6 +214,7 @@ export const DEFAULT_PRICING_FRONTIER = { // Common model IDs (without dates) used across providers // Intentional duplicates of dot-notation variants (e.g. claude-opus-4.6) // to cover hyphen-notation IDs (claude-opus-4-6) used by some clients + "claude-fable-5-1": CLAUDE_FABLE_5_1_PRICING, "claude-fable-5": CLAUDE_FABLE_5_PRICING, "claude-opus-5": CLAUDE_OPUS_5_PRICING, "claude-sonnet-5": CLAUDE_SONNET_5_PRICING, diff --git a/src/shared/constants/pricing/oauth-subscriptions.ts b/src/shared/constants/pricing/oauth-subscriptions.ts index 87f5aa320f..9d255e0e24 100644 --- a/src/shared/constants/pricing/oauth-subscriptions.ts +++ b/src/shared/constants/pricing/oauth-subscriptions.ts @@ -3,6 +3,7 @@ * Pure data; merged by default-pricing.ts via spread (god-file decomposition; semantic split). */ import { + CLAUDE_FABLE_5_1_PRICING, CLAUDE_OPUS_5_PRICING, GEMINI_3_7_FLASH_PROMO_PRICING, GPT_5_3_CODEX_PRICING, @@ -20,6 +21,7 @@ const ANTIGRAVITY_GEMINI_3_7_PRICING = { export const DEFAULT_PRICING_OAUTH = { cc: { + "claude-fable-5-1": CLAUDE_FABLE_5_1_PRICING, "claude-fable-5": { input: 10.0, output: 50.0, diff --git a/src/shared/constants/pricing/shared-tiers.ts b/src/shared/constants/pricing/shared-tiers.ts index 542e03521a..7cb674df09 100644 --- a/src/shared/constants/pricing/shared-tiers.ts +++ b/src/shared/constants/pricing/shared-tiers.ts @@ -60,6 +60,14 @@ export const CLAUDE_FABLE_5_PRICING = { cache_creation: 15.0, }; +export const CLAUDE_FABLE_5_1_PRICING = { + input: 10.0, + output: 50.0, + cached: 0.25, + reasoning: 50.0, + cache_creation: 12.5, +}; + export const CLAUDE_OPUS_5_PRICING = { input: 5.0, output: 25.0, diff --git a/tests/unit/claude-fable-5-1.test.ts b/tests/unit/claude-fable-5-1.test.ts new file mode 100644 index 0000000000..b494db3d3a --- /dev/null +++ b/tests/unit/claude-fable-5-1.test.ts @@ -0,0 +1,172 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { + getModelTargetFormat, + getModelsByProviderId, + supportsClaudeMaxEffort, + supportsXHighEffort, +} from "../../open-sse/config/providerModels.ts"; +import { getUnsupportedParams } from "../../open-sse/config/providerRegistry.ts"; +import { modelHasNativeContext1m } from "../../open-sse/config/claudeCodeCompatibleIdentity.ts"; +import { modelSupportsContext1mBeta } from "../../open-sse/config/context1m.ts"; +import { normalizeClaudeAdaptiveThinking } from "../../open-sse/services/claudeAdaptiveThinking.ts"; +import { getNextFamilyFallback } from "../../open-sse/services/modelFamilyFallback.ts"; +import { getModelPricing } from "../../open-sse/services/providerCostData.ts"; +import { getStaticModelsForProvider } from "../../src/lib/providers/staticModels.ts"; +import { getDefaultPricing } from "../../src/shared/constants/pricing.ts"; +import { + getModelSpec, + normalizeForcedToolChoiceForModel, + normalizeThinkingForModel, +} from "../../src/shared/constants/modelSpecs.ts"; + +const MODEL_ID = "claude-fable-5-1"; +const BEDROCK_MODEL_ID = "anthropic.claude-fable-5-1"; +const EFFORTS = ["low", "medium", "high", "xhigh", "max"]; + +test("Claude Fable 5.1 is registered only on verified launch surfaces", () => { + for (const [providerId, modelId] of [ + ["anthropic", MODEL_ID], + ["claude", MODEL_ID], + ["claude-web", MODEL_ID], + ["bedrock", BEDROCK_MODEL_ID], + ["vertex", MODEL_ID], + ["vertex-partner", MODEL_ID], + ] as const) { + const model = getModelsByProviderId(providerId).find((entry) => entry.id === modelId); + assert.ok(model, `${providerId} must expose ${modelId}`); + if (providerId === "vertex" || providerId === "vertex-partner") { + assert.equal(model.targetFormat, "claude", `${providerId} wire format`); + } else { + assert.equal(model.contextLength, 1_000_000, `${providerId} context window`); + assert.equal(model.maxOutputTokens, 128_000, `${providerId} max output`); + assert.deepEqual(model.supportedThinkingEfforts, EFFORTS, `${providerId} effort levels`); + } + } + + assert.equal(getModelTargetFormat("vertex", MODEL_ID), "claude"); + assert.equal(getModelTargetFormat("vertex-partner", MODEL_ID), "claude"); + + for (const providerId of ["github", "ghe-copilot", "kiro"]) { + const ids = new Set(getModelsByProviderId(providerId).map((entry) => entry.id)); + assert.equal(ids.has(MODEL_ID), false, `${providerId} availability is not verified`); + } + + const cursorModels = new Map( + getModelsByProviderId("cursor").map((entry) => [entry.id, entry] as const) + ); + assert.equal(cursorModels.has(MODEL_ID), false, "cursor exposes only selectable variants"); + + const cursorApiIds = new Set(getModelsByProviderId("cursor-api").map((entry) => entry.id)); + assert.equal(cursorApiIds.has(MODEL_ID), false); + for (const effort of EFFORTS) { + const cursorId = `${MODEL_ID}-thinking-${effort}`; + const cursor1mId = `${cursorId}-1m`; + assert.equal(cursorModels.has(`${MODEL_ID}-${effort}`), false); + assert.equal(cursorModels.get(cursorId)?.contextLength, 300_000, cursorId); + assert.equal(cursorModels.get(cursorId)?.maxOutputTokens, 128_000, cursorId); + assert.equal(cursorApiIds.has(cursorId), true, `cursor-api must expose ${cursorId}`); + assert.equal(cursorModels.get(cursor1mId)?.contextLength, 1_000_000, cursor1mId); + assert.equal(cursorModels.get(cursor1mId)?.maxOutputTokens, 128_000, cursor1mId); + assert.equal(cursorApiIds.has(cursor1mId), true, `cursor-api must expose ${cursor1mId}`); + } + + assert.ok( + getStaticModelsForProvider("claude")?.some((entry) => entry.id === MODEL_ID), + "Claude OAuth static discovery must expose Fable 5.1" + ); + assert.equal( + getNextFamilyFallback(`claude/${MODEL_ID}`, new Set([`claude/${MODEL_ID}`])), + "claude/claude-fable-5" + ); +}); + +test("Claude Fable 5.1 has native 1M context and adaptive-only thinking", () => { + assert.equal(modelHasNativeContext1m(MODEL_ID), true); + assert.equal(modelHasNativeContext1m(BEDROCK_MODEL_ID), true); + assert.equal(modelSupportsContext1mBeta(MODEL_ID), false); + + const spec = getModelSpec(MODEL_ID); + assert.equal(spec?.contextWindow, 1_000_000); + assert.equal(spec?.maxOutputTokens, 128_000); + assert.equal(spec?.supportsThinking, true); + assert.equal(spec?.supportsTools, true); + assert.equal(spec?.supportsVision, true); + assert.equal(spec?.adaptiveThinkingOnly, true); + assert.equal(spec?.rejectsThinkingDisabled, true); + assert.equal( + (spec as typeof spec & { rejectsForcedToolChoice?: boolean })?.rejectsForcedToolChoice, + true + ); + + assert.equal(getModelSpec(`global.${BEDROCK_MODEL_ID}`), spec); + assert.equal(supportsXHighEffort("claude", MODEL_ID), true); + assert.equal(supportsClaudeMaxEffort(MODEL_ID), true); +}); + +test("Claude Fable 5.1 strips unsupported sampling parameters", () => { + for (const providerId of ["anthropic", "claude"] as const) { + const unsupported = getUnsupportedParams(providerId, MODEL_ID); + for (const param of ["temperature", "top_p", "top_k"]) { + assert.ok(unsupported.includes(param), `${providerId}/${MODEL_ID} must strip ${param}`); + } + } +}); + +test("Claude Fable 5.1 normalizes disabled and manual thinking to adaptive", () => { + const withoutDisabled = normalizeThinkingForModel( + { model: MODEL_ID, thinking: { type: "disabled" }, marker: true }, + MODEL_ID + ); + assert.equal("thinking" in withoutDisabled, false); + assert.equal(withoutDisabled.marker, true); + + const adaptive = normalizeClaudeAdaptiveThinking( + { model: MODEL_ID, thinking: { type: "enabled", budget_tokens: 64_000 } }, + MODEL_ID + ); + assert.deepEqual(adaptive.thinking, { type: "adaptive" }); +}); + +test("Claude Fable 5.1 relaxes forced tool choices without removing tools", () => { + for (const toolChoice of [ + "required", + "any", + { type: "any" }, + { type: "tool", name: "read_file" }, + { type: "function", function: { name: "read_file" } }, + ]) { + const tools = [{ name: "read_file", input_schema: { type: "object" } }]; + const result = normalizeForcedToolChoiceForModel( + { model: MODEL_ID, tools, tool_choice: toolChoice, marker: true }, + MODEL_ID + ); + assert.equal("tool_choice" in result, false); + assert.equal(result.tools, tools); + assert.equal(result.marker, true); + } + + const auto = { model: MODEL_ID, tool_choice: { type: "auto" } }; + assert.equal(normalizeForcedToolChoiceForModel(auto, MODEL_ID), auto); + + const older = { model: "claude-fable-5", tool_choice: { type: "tool", name: "read_file" } }; + assert.equal(normalizeForcedToolChoiceForModel(older, "claude-fable-5"), older); +}); + +test("Claude Fable 5.1 pricing matches Anthropic's published rates", () => { + for (const providerId of ["anthropic", "cc"] as const) { + const price = getDefaultPricing()[providerId][MODEL_ID]; + assert.equal(price.input, 10); + assert.equal(price.output, 50); + assert.equal(price.cached, 0.25); + assert.equal(price.reasoning, 50); + assert.equal(price.cache_creation, 12.5); + } + + assert.deepEqual(getModelPricing("anthropic", MODEL_ID), { + inputCostPer1M: 10, + outputCostPer1M: 50, + isFree: false, + }); +}); diff --git a/tests/unit/claude-web-sonnet5-registry-6209.test.ts b/tests/unit/claude-web-sonnet5-registry-6209.test.ts index 9930a04ae2..a2f568fc9c 100644 --- a/tests/unit/claude-web-sonnet5-registry-6209.test.ts +++ b/tests/unit/claude-web-sonnet5-registry-6209.test.ts @@ -9,6 +9,7 @@ test("claude-web registry matches the current selectable model set", () => { assert.deepEqual( ids, [ + "claude-fable-5-1", "claude-fable-5", "claude-haiku-4-5-20251001", "claude-opus-5", diff --git a/tests/unit/cursor-auto-catalog-entry.test.ts b/tests/unit/cursor-auto-catalog-entry.test.ts index a194885927..aea92b7394 100644 --- a/tests/unit/cursor-auto-catalog-entry.test.ts +++ b/tests/unit/cursor-auto-catalog-entry.test.ts @@ -28,4 +28,31 @@ describe("ensureCursorAutoCatalogEntry", () => { assert.equal(models.filter((m) => m.id === "auto").length, 1); assert.equal(models.filter((m) => m.id === "auto-cost").length, 1); }); + + it("injects supported 1M context variants immediately before their base ids", () => { + const models = ensureCursorAutoCatalogEntry([ + { id: "claude-opus-5-thinking-max-fast", name: "Claude Opus 5 Max Thinking Fast" }, + { id: "gpt-5.6-sol-max", name: "GPT-5.6 Sol Max" }, + { id: "gpt-5.6-sol-max-fast", name: "GPT-5.6 Sol Max Fast" }, + ]); + const ids = models.map((model) => model.id); + for (const baseId of ["claude-opus-5-thinking-max-fast", "gpt-5.6-sol-max"]) { + const oneMillionPosition = ids.indexOf(`${baseId}-1m`); + assert.ok(oneMillionPosition >= 0); + assert.equal(ids[oneMillionPosition + 1], baseId); + assert.equal( + (models[oneMillionPosition] as { contextLength?: number }).contextLength, + 1_000_000 + ); + } + assert.equal(ids.includes("gpt-5.6-sol-max-fast-1m"), false); + }); + + it("does not duplicate a discovered 1M context variant", () => { + const models = ensureCursorAutoCatalogEntry([ + { id: "gpt-5.6-luna-max-1m", name: "GPT-5.6 Luna 1M Max" }, + { id: "gpt-5.6-luna-max", name: "GPT-5.6 Luna Max" }, + ]); + assert.equal(models.filter((model) => model.id === "gpt-5.6-luna-max-1m").length, 1); + }); }); diff --git a/tests/unit/cursor-available-models.test.ts b/tests/unit/cursor-available-models.test.ts index bf4335b388..28e822427d 100644 --- a/tests/unit/cursor-available-models.test.ts +++ b/tests/unit/cursor-available-models.test.ts @@ -17,7 +17,9 @@ describe("normalizeCursorAvailableModelsPayload", () => { }); assert.equal(models[0].id, "auto"); assert.ok(models.some((m) => m.id === "claude-opus-5-high")); + assert.ok(models.some((m) => m.id === "claude-opus-5-high-1m")); assert.ok(models.some((m) => m.id === "gpt-5.6-sol-high")); + assert.ok(models.some((m) => m.id === "gpt-5.6-sol-high-1m")); assert.ok(models.some((m) => m.id === "auto-cost")); assert.equal(models.find((m) => m.id === "claude-opus-5-high")?.name, "Opus 5"); assert.equal(models.find((m) => m.id === "claude-opus-5-high")?.owned_by, "cursor"); diff --git a/tests/unit/cursor-catalog-combo-compat.test.ts b/tests/unit/cursor-catalog-combo-compat.test.ts index cd3a8f481a..d27613b317 100644 --- a/tests/unit/cursor-catalog-combo-compat.test.ts +++ b/tests/unit/cursor-catalog-combo-compat.test.ts @@ -33,11 +33,11 @@ const LEGACY_GROK_ALIASES = { "grok-4.5-fast-xhigh": "cursor-grok-4.5-xhigh-fast", } as const; -test("keeps legacy Cursor combo model ids in the static catalog", () => { +test("keeps legacy Cursor combo model ids out of the curated static catalog", () => { const catalogIds = new Set(cursorProvider.models.map((model) => model.id)); for (const modelId of LEGACY_CURSOR_COMBO_MODEL_IDS) { - assert.ok(catalogIds.has(modelId), `missing Cursor catalog model: ${modelId}`); + assert.equal(catalogIds.has(modelId), false, `unexpected Cursor catalog model: ${modelId}`); } }); diff --git a/tests/unit/cursor-model-effort-suffix-7289.test.ts b/tests/unit/cursor-model-effort-suffix-7289.test.ts index 10e4023e14..aeb2f9cdb1 100644 --- a/tests/unit/cursor-model-effort-suffix-7289.test.ts +++ b/tests/unit/cursor-model-effort-suffix-7289.test.ts @@ -80,3 +80,35 @@ test("resolveRequestedModel splits cursor-grok effort + fast together", () => { ], }); }); + +test("resolveRequestedModel expands Claude 1M catalog ids into complete wire parameters", () => { + assert.deepEqual(resolveRequestedModel("claude-opus-5-thinking-max-fast-1m"), { + modelId: "claude-opus-5", + parameters: [ + { id: "thinking", value: "true" }, + { id: "context", value: "1m" }, + { id: "effort", value: "max" }, + { id: "fast", value: "true" }, + ], + }); + assert.deepEqual(resolveRequestedModel("claude-4.6-sonnet-high-thinking-1m"), { + modelId: "claude-sonnet-4-6", + parameters: [ + { id: "thinking", value: "true" }, + { id: "context", value: "1m" }, + { id: "effort", value: "high" }, + ], + }); +}); + +test("resolveRequestedModel expands GPT-5.6 1M ids and keeps fast disabled", () => { + const id = "gpt-5.6-sol-xhigh-1m"; + assert.deepEqual(resolveRequestedModel(id, { liveCatalogIds: new Set([id]) }), { + modelId: "gpt-5.6-sol", + parameters: [ + { id: "context", value: "1m" }, + { id: "reasoning", value: "xhigh" }, + { id: "fast", value: "false" }, + ], + }); +}); diff --git a/tests/unit/cursor-registry-claude-families.test.ts b/tests/unit/cursor-registry-claude-families.test.ts index 46a41970a0..0236f99682 100644 --- a/tests/unit/cursor-registry-claude-families.test.ts +++ b/tests/unit/cursor-registry-claude-families.test.ts @@ -1,51 +1,196 @@ -import test from "node:test"; import assert from "node:assert/strict"; +import test from "node:test"; + import { cursorProvider } from "../../open-sse/config/providers/registry/cursor/index.ts"; -const EFFORTS = ["low", "medium", "high", "xhigh", "max"] as const; +const CURSOR_FAMILY_REPRESENTATIVES = [ + "cursor-grok-4.6-high-fast", + "composer-2.5", + "claude-fable-5-1-thinking-high", + "claude-opus-5-thinking-high", + "claude-opus-4-8-thinking-high", + "claude-sonnet-5-thinking-high", + "claude-4.6-sonnet-medium-thinking", + "claude-4.5-haiku-thinking", + "gpt-5.6-sol-medium", + "gpt-5.6-terra-medium", + "gpt-5.6-luna-medium", + "gemini-3.7-flash-high", + "gemini-3.1-pro", + "kimi-k3-max", + "kimi-k2.7-code", + "glm-5.2-high", +] as const; -function modelIds(): Set { - return new Set(cursorProvider.models.map((m) => m.id)); -} - -test("cursor registry excludes retired Gemini 3.5 Flash", () => { - assert.equal(modelIds().has("gemini-3.5-flash"), false); +test("cursor registry keeps every selected model family", () => { + const allIds = cursorProvider.models.map((model) => model.id); + const ids = new Set(allIds); + assert.equal(ids.size, allIds.length, "Cursor catalog model ids must be unique"); + for (const id of CURSOR_FAMILY_REPRESENTATIVES) { + assert.ok(ids.has(id), `missing Cursor model: ${id}`); + } }); -test("cursor registry includes Claude Opus 4.8 effort + thinking + fast variants", () => { - const ids = modelIds(); - for (const effort of EFFORTS) { - assert.ok(ids.has(`claude-opus-4-8-${effort}`), `missing claude-opus-4-8-${effort}`); - assert.ok(ids.has(`claude-opus-4-8-${effort}-fast`), `missing claude-opus-4-8-${effort}-fast`); +test("cursor registry omits redundant bare ids for parameterized models", () => { + const ids = new Set(cursorProvider.models.map((model) => model.id)); + for (const id of [ + "grok-4.6", + "claude-fable-5-1", + "claude-opus-5", + "claude-opus-4-8", + "claude-sonnet-5", + "claude-sonnet-4-6", + "claude-haiku-4-5", + "gpt-5.6-sol", + "gpt-5.6-terra", + "gpt-5.6-luna", + "gemini-3.7-flash", + "kimi-k3", + "glm-5.2", + ]) { + assert.equal(ids.has(id), false, `unexpected bare Cursor model: ${id}`); + } +}); + +test("cursor registry keeps thinking, effort/reasoning and fast variants selectable", () => { + const ids = new Set(cursorProvider.models.map((model) => model.id)); + for (const id of [ + "cursor-grok-4.6-xhigh-fast", + "composer-2.5-fast", + "claude-fable-5-1-thinking-max", + "claude-opus-5-thinking-xhigh-fast", + "claude-opus-4-8-thinking-max-fast", + "claude-sonnet-5-thinking-max", + "claude-4.6-sonnet-max-thinking", + "claude-4.5-haiku-thinking", + "gpt-5.6-sol-max-fast", + "gpt-5.6-terra-max-fast", + "gpt-5.6-luna-max-fast", + "gemini-3.7-flash-high", + "kimi-k3-max", + "glm-5.2-max", + ]) { + assert.ok(ids.has(id), `missing selectable Cursor variant: ${id}`); + } +}); + +test("cursor registry exposes every supported 1M context variant", () => { + const ids = cursorProvider.models.map((model) => model.id); + const oneMillionVariants = cursorProvider.models.filter((model) => model.id.endsWith("-1m")); + assert.equal(oneMillionVariants.length, 77); + for (const variant of oneMillionVariants) { + assert.match(variant.name, /\b1M\b/); + assert.equal(variant.contextLength, 1_000_000); + const position = ids.indexOf(variant.id); + assert.equal(ids[position + 1], variant.id.slice(0, -"-1m".length)); + } + for (const id of [ + "claude-fable-5-1-thinking-max-1m", + "claude-opus-5-thinking-max-fast-1m", + "claude-opus-4-8-thinking-max-fast-1m", + "claude-sonnet-5-thinking-max-1m", + "claude-4.6-sonnet-max-thinking-1m", + "gpt-5.6-sol-max-1m", + "gpt-5.6-terra-max-1m", + "gpt-5.6-luna-max-1m", + ]) { assert.ok( - ids.has(`claude-opus-4-8-thinking-${effort}`), - `missing claude-opus-4-8-thinking-${effort}` + oneMillionVariants.some((model) => model.id === id), + `missing 1M variant: ${id}` ); + } + assert.equal( + oneMillionVariants.some( + (model) => model.id.startsWith("gpt-5.6-") && model.id.includes("-fast") + ), + false, + "Cursor does not offer fast processing with GPT-5.6 1M context" + ); +}); + +test("cursor registry records the default context for context-selectable families", () => { + const models = new Map(cursorProvider.models.map((model) => [model.id, model])); + assert.equal(models.get("claude-fable-5-1-thinking-max")?.contextLength, 300_000); + assert.equal(models.get("claude-opus-5-thinking-max")?.contextLength, 300_000); + assert.equal(models.get("claude-opus-4-8-thinking-max")?.contextLength, 300_000); + assert.equal(models.get("claude-sonnet-5-thinking-max")?.contextLength, 300_000); + assert.equal(models.get("claude-4.6-sonnet-max-thinking")?.contextLength, 200_000); + assert.equal(models.get("gpt-5.6-sol-max")?.contextLength, 272_000); + assert.equal(models.get("gpt-5.6-terra-max")?.contextLength, 272_000); + assert.equal(models.get("gpt-5.6-luna-max")?.contextLength, 272_000); +}); + +test("cursor registry orders each model family by quality, thinking and speed", () => { + const ids = cursorProvider.models.map((model) => model.id); + for (const orderedIds of [ + ["cursor-grok-4.6-xhigh-fast", "cursor-grok-4.6-xhigh", "cursor-grok-4.6-low"], + ["composer-2.5-fast", "composer-2.5"], + ["claude-fable-5-1-thinking-max", "claude-fable-5-1-thinking-low"], + ["claude-opus-5-thinking-high-fast", "claude-opus-5-high-fast", "claude-opus-5-low"], + ["claude-opus-4-8-thinking-max-fast", "claude-opus-4-8-max-fast", "claude-opus-4-8-low"], + ["claude-sonnet-5-thinking-max", "claude-sonnet-5-max", "claude-sonnet-5-low"], + ["claude-4.6-sonnet-max-thinking", "claude-4.6-sonnet-max", "claude-4.6-sonnet-low"], + ["claude-4.5-haiku-thinking", "claude-4.5-haiku"], + ["gpt-5.6-sol-max-fast", "gpt-5.6-sol-max", "gpt-5.6-sol-none"], + ["gpt-5.6-terra-max-fast", "gpt-5.6-terra-max", "gpt-5.6-terra-none"], + ["gpt-5.6-luna-max-fast", "gpt-5.6-luna-max", "gpt-5.6-luna-none"], + ["gemini-3.7-flash-high", "gemini-3.7-flash-low"], + ["kimi-k3-max", "kimi-k3-low"], + ["glm-5.2-max", "glm-5.2-high"], + ]) { + const positions = orderedIds.map((id) => ids.indexOf(id)); assert.ok( - ids.has(`claude-opus-4-8-thinking-${effort}-fast`), - `missing claude-opus-4-8-thinking-${effort}-fast` + positions.every((position) => position >= 0), + `missing ordered ids: ${orderedIds}` + ); + assert.deepEqual( + positions, + [...positions].sort((left, right) => left - right) ); } }); -test("cursor registry includes Claude Fable 5 effort + thinking variants", () => { - const ids = modelIds(); - for (const effort of EFFORTS) { - assert.ok(ids.has(`claude-fable-5-${effort}`), `missing claude-fable-5-${effort}`); - assert.ok( - ids.has(`claude-fable-5-thinking-${effort}`), - `missing claude-fable-5-thinking-${effort}` +test("cursor registry uses compact Xhigh labels", () => { + const xhighVariants = cursorProvider.models.filter((model) => model.id.includes("xhigh")); + assert.ok(xhighVariants.length > 0); + for (const variant of xhighVariants) { + assert.match(variant.name, /\bXhigh\b/); + assert.doesNotMatch(variant.name, /Extra High/); + } +}); + +test("cursor registry excludes unrelated model families", () => { + const ids = cursorProvider.models.map((model) => model.id); + for (const fragment of [ + "grok-4.5", + "gpt-5.5", + "gpt-5.4", + "gpt-5.3", + "gpt-5.2", + "claude-fable-5-thinking", + "claude-opus-4-7", + "gemini-3.6-flash", + "gemini-3.5-flash", + "gemini-3-flash", + ]) { + assert.equal( + ids.some((id) => id.includes(fragment)), + false, + `unexpected Cursor model family: ${fragment}` ); } }); -test("cursor registry includes Claude Sonnet 5 effort + thinking variants", () => { - const ids = modelIds(); - for (const effort of EFFORTS) { - assert.ok(ids.has(`claude-sonnet-5-${effort}`), `missing claude-sonnet-5-${effort}`); - assert.ok( - ids.has(`claude-sonnet-5-thinking-${effort}`), - `missing claude-sonnet-5-thinking-${effort}` - ); +test("cursor registry keeps Fable 5.1 capability metadata on every selectable variant", () => { + const variants = cursorProvider.models.filter((model) => + model.id.startsWith("claude-fable-5-1-thinking-") + ); + assert.equal(variants.length, 10); + assert.deepEqual( + new Set(variants.map((variant) => variant.contextLength)), + new Set([300_000, 1_000_000]) + ); + for (const variant of variants) { + assert.equal(variant.maxOutputTokens, 128_000); } }); diff --git a/tests/unit/devin-cli-catalog.test.ts b/tests/unit/devin-cli-catalog.test.ts index 7cd92e15b1..330ee8ac31 100644 --- a/tests/unit/devin-cli-catalog.test.ts +++ b/tests/unit/devin-cli-catalog.test.ts @@ -2,54 +2,124 @@ import assert from "node:assert/strict"; import test from "node:test"; import { devin_cliProvider } from "../../open-sse/config/providers/registry/devin-cli/index.ts"; +import { devin_cli_agenticProvider } from "../../open-sse/config/providers/registry/devin-cli-agentic/index.ts"; import { devin_desktopProvider } from "../../open-sse/config/providers/registry/devin-desktop/index.ts"; import { DEVIN_MODEL_CATALOG } from "../../open-sse/config/providers/registry/devin/catalog.ts"; +import { DEVIN_MODEL_PRICING } from "../../src/shared/constants/pricing/devin.ts"; +import { DEFAULT_PRICING, getPricingForModel } from "../../src/shared/constants/pricing.ts"; -test("Devin CLI and Desktop use the shared catalog without duplicate model ids", () => { - const ids = DEVIN_MODEL_CATALOG.map((model) => model.id); +const catalogIds = DEVIN_MODEL_CATALOG.map((model) => model.id); +test("Devin transports expose the same curated catalog without duplicate ids", () => { assert.equal(devin_cliProvider.models, DEVIN_MODEL_CATALOG); assert.equal(devin_desktopProvider.models, DEVIN_MODEL_CATALOG); - assert.equal(new Set(ids).size, ids.length); - assert.ok(ids.every((id) => !id.toLowerCase().includes("byok"))); + assert.deepEqual( + devin_cli_agenticProvider.models.map((model) => model.id), + catalogIds + ); + assert.equal(catalogIds.length, 110); + assert.equal(new Set(catalogIds).size, catalogIds.length); + assert.ok(catalogIds.every((id) => !id.toLowerCase().includes("byok"))); }); -test("Devin CLI catalog includes the refreshed native model ids", () => { - const ids = new Set(DEVIN_MODEL_CATALOG.map((model) => model.id)); +test("Devin catalog contains only the operator-selected model families", () => { + const required = [ + "claude-fable-5-1-max", + "claude-opus-5-max-fast", + "claude-opus-4-8-max-fast", + "claude-sonnet-5-max", + "claude-sonnet-4-6-thinking-1m", + "MODEL_PRIVATE_11", + "gpt-5-6-sol-max-priority", + "gpt-5-6-terra-max-priority", + "gpt-5-6-luna-max-priority", + "kimi-k3-max", + "kimi-k2-7", + "glm-5-3-max", + "glm-5-3-flash-max", + "swe-1-7", + "swe-1-7-lightning", + "adaptive", + "grok-4-6-xhigh", + "inkling-max", + "deepseek-v4-flash-max", + "nemotron-3-ultra-high", + "gemini-3-7-flash-high", + "gemini-3-1-pro-high", + "deepseek-v4-pro-max", + ]; + + for (const id of required) { + assert.ok(catalogIds.includes(id), `expected selected Devin model id: ${id}`); + } for (const id of [ - "swe-1-7-lightning", "claude-5-fable-max", - "gpt-5-6-sol-max", + "claude-opus-4-7-max", "gpt-5-5-high", - "glm-5-2-max-1m", - "claude-opus-5-low", - "claude-opus-5-medium", - "claude-opus-5-high", - "claude-opus-5-xhigh", - "claude-opus-5-max", - "gemini-3-7-flash-minimal", - "gemini-3-7-flash-low", - "gemini-3-7-flash-medium", - "gemini-3-7-flash-high", - "kimi-k3-low", - "kimi-k3-high", - "kimi-k3-max", - "inkling-none", - "inkling-low", - "inkling-medium", - "inkling-high", - "inkling-xhigh", - "inkling-max", + "gemini-3-6-flash-high", + "grok-4-5-high", + "deepseek-v4", + "nemotron-3-ultra-nvfp4", + "swe-1-6-fast", ]) { - assert.ok(ids.has(id), `expected refreshed Devin model id: ${id}`); + assert.equal(catalogIds.includes(id), false, `unselected Devin model must stay absent: ${id}`); } }); -test("Devin CLI catalog does not expose retired dotted or review model ids", () => { - const ids = new Set(DEVIN_MODEL_CATALOG.map((model) => model.id)); +test("Devin catalog keeps higher-quality choices first", () => { + assert.deepEqual(catalogIds.slice(0, 5), [ + "claude-fable-5-1-max", + "claude-fable-5-1-xhigh", + "claude-fable-5-1-high", + "claude-fable-5-1-medium", + "claude-fable-5-1-low", + ]); + assert.deepEqual(catalogIds.slice(5, 9), [ + "claude-opus-5-max-fast", + "claude-opus-5-max", + "claude-opus-5-xhigh-fast", + "claude-opus-5-xhigh", + ]); +}); - for (const id of ["swe-1.6-fast", "swe-1.6", "claude-opus-4.7-review"]) { - assert.equal(ids.has(id), false, `retired Devin model id must stay absent: ${id}`); +test("every curated Devin model has an exact live provider price", () => { + assert.deepEqual(new Set(Object.keys(DEVIN_MODEL_PRICING)), new Set(catalogIds)); + + for (const provider of ["devin-cli", "dv", "devin-desktop", "devin-cli-agentic", "dva"]) { + for (const id of catalogIds) { + assert.ok(getPricingForModel(provider, id), `missing ${provider}/${id} pricing`); + } } }); + +test("Devin pricing remains provider-bound and preserves fast-tier rates", () => { + assert.notEqual(DEFAULT_PRICING["devin-cli"], DEFAULT_PRICING.anthropic); + assert.deepEqual(getPricingForModel("devin-cli", "claude-sonnet-5-max"), { + input: 2, + cached: 0.2, + output: 10, + }); + assert.deepEqual(getPricingForModel("anthropic", "claude-sonnet-5"), { + input: 3, + output: 15, + cached: 1.5, + reasoning: 22.5, + cache_creation: 3, + }); + assert.deepEqual(getPricingForModel("devin-cli", "gpt-5-6-sol-max-priority"), { + input: 8, + cached: 0.8, + output: 40, + }); +}); + +test("Devin catalog carries the live output limits for representative models", () => { + const models = new Map(DEVIN_MODEL_CATALOG.map((entry) => [entry.id, entry])); + + assert.equal(models.get("claude-fable-5-1-max")?.maxOutputTokens, 128_000); + assert.equal(models.get("MODEL_PRIVATE_11")?.maxOutputTokens, 64_000); + assert.equal(models.get("kimi-k2-7")?.maxOutputTokens, 16_000); + assert.equal(models.get("grok-4-6-xhigh")?.maxOutputTokens, 100_000); + assert.equal(models.get("gemini-3-7-flash-high")?.maxOutputTokens, 65_535); +}); diff --git a/tests/unit/fixtures/cursor-rewrite-failure-ids.ts b/tests/unit/fixtures/cursor-rewrite-failure-ids.ts index dd064537c6..641d383cb8 100644 --- a/tests/unit/fixtures/cursor-rewrite-failure-ids.ts +++ b/tests/unit/fixtures/cursor-rewrite-failure-ids.ts @@ -4,11 +4,16 @@ * Smoke checklist for catalog-aware pass-through. */ export const CURSOR_REWRITE_FAILURE_IDS = [ - // Claude (52) + // Claude (57) "claude-4.5-opus-high", "claude-4.6-opus-high", "claude-4.6-opus-max", "claude-4.6-sonnet-medium", + "claude-fable-5-1-thinking-low", + "claude-fable-5-1-thinking-medium", + "claude-fable-5-1-thinking-high", + "claude-fable-5-1-thinking-xhigh", + "claude-fable-5-1-thinking-max", "claude-fable-5-low", "claude-fable-5-medium", "claude-fable-5-high", diff --git a/tests/unit/guardrails/visionBridgeRouter.test.ts b/tests/unit/guardrails/visionBridgeRouter.test.ts index 3197f08e20..dc685a61ff 100644 --- a/tests/unit/guardrails/visionBridgeRouter.test.ts +++ b/tests/unit/guardrails/visionBridgeRouter.test.ts @@ -263,12 +263,12 @@ test("getFallbackModels — excludes fallbacks missing from an authoritative liv test("getFallbackModels — keeps registered effort variants backed by a live base model", async () => { const fallbacks = await getFallbackModels( - "cu/gpt-5.3-codex", + "cu/claude-fable-5-1-thinking-max", { maxFallbackAttempts: 6 }, - authoritativeCatalogDeps("cu", () => ["gpt-5.3-codex"]) + authoritativeCatalogDeps("cu", () => ["claude-fable-5-1"]) ); - assert.ok(fallbacks.includes("cu/gpt-5.3-codex-low")); + assert.ok(fallbacks.includes("cu/claude-fable-5-1-thinking-high")); }); // ── recordLatency / getLatencyStats ───────────────────────────────────────── diff --git a/tests/unit/live-model-catalog-reconciliation-8926.test.ts b/tests/unit/live-model-catalog-reconciliation-8926.test.ts index 4bd415dee3..5bc1d15c54 100644 --- a/tests/unit/live-model-catalog-reconciliation-8926.test.ts +++ b/tests/unit/live-model-catalog-reconciliation-8926.test.ts @@ -117,24 +117,24 @@ test("#8926: explicit custom model overrides live-catalog exclusion", async () = }); test("#8926: effort helper identifies only explicitly registered variants", () => { - assert.equal(isRegisteredProviderEffortVariant("cursor", "gpt-5.3-codex-high"), true); + assert.equal(isRegisteredProviderEffortVariant("cursor", "claude-fable-5-1-thinking-high"), true); assert.equal( - isRegisteredProviderEffortVariant("cursor", "gpt-5.3-codex-max"), + isRegisteredProviderEffortVariant("cursor", "claude-fable-5-1-thinking-ultra"), false, "an invented suffix must not bypass live-catalog authority" ); }); test("#8926: registered effort route survives while invented effort route is rejected", async () => { - await seedProviderCatalog("cursor", "cursor-live-8926", ["gpt-5.3-codex"]); + await seedProviderCatalog("cursor", "cursor-live-8926", ["claude-fable-5-1"]); - const registered = await getModelInfo("cursor/gpt-5.3-codex-high"); + const registered = await getModelInfo("cursor/claude-fable-5-1-thinking-high"); assert.equal(registered.provider, "cursor"); - assert.equal(registered.model, "gpt-5.3-codex-high"); + assert.equal(registered.model, "claude-fable-5-1-thinking-high"); - const invented = await getModelInfo("cursor/gpt-5.3-codex-max"); + const invented = await getModelInfo("cursor/claude-fable-5-1-thinking-ultra"); assert.equal(invented.provider, null); assert.equal(invented.errorType, "model_not_found"); @@ -170,13 +170,13 @@ test("#8926: providers without an authoritative live catalog retain static fallb test("#8926: registered effort variant is rejected when its live base is absent", async () => { await seedProviderCatalog("cursor", "cursor-live-without-base-8926", ["cursor-live-only-8926"]); - const explicit = await getModelInfo("cursor/gpt-5.3-codex-high"); + const explicit = await getModelInfo("cursor/claude-fable-5-1-thinking-high"); assert.equal(explicit.provider, null); assert.equal(explicit.errorType, "model_not_found"); assert.match(explicit.errorMessage, /active live catalog/i); - const bare = await getModelInfo("gpt-5.3-codex-high"); + const bare = await getModelInfo("claude-fable-5-1-thinking-high"); assert.equal(bare.provider, null); assert.equal(bare.errorType, "model_not_found"); diff --git a/tests/unit/pricing-constants-split.test.ts b/tests/unit/pricing-constants-split.test.ts index 28acb759c4..90970e1730 100644 --- a/tests/unit/pricing-constants-split.test.ts +++ b/tests/unit/pricing-constants-split.test.ts @@ -1,5 +1,5 @@ // Characterization of the pricing.ts split (god-file decomposition): the host became a barrel that -// re-exports DEFAULT_PRICING (now merged from 4 semantic family files that import shared tier consts) +// re-exports DEFAULT_PRICING (merged from semantic family files that import shared tier consts) // and keeps the helper functions. Pure-data move → behavior identical. Locks: public surface, the // spread-merge integrity, and that lookups/cost math resolve unchanged. import { test } from "node:test"; @@ -14,13 +14,14 @@ test("barrel still exports DEFAULT_PRICING + supported helpers", () => { assert.equal(Object.hasOwn(P, "calculateCostFromTokens"), false); }); -test("DEFAULT_PRICING merges the 4 family files; families partition all entries", async () => { +test("DEFAULT_PRICING merges every family file; families partition all entries", async () => { const merged = Object.keys((P as Record).DEFAULT_PRICING).length; const families: [string, string][] = [ ["oauth-subscriptions", "DEFAULT_PRICING_OAUTH"], ["frontier-labs", "DEFAULT_PRICING_FRONTIER"], ["inference-hosts", "DEFAULT_PRICING_INFERENCE"], ["regional", "DEFAULT_PRICING_REGIONAL"], + ["devin", "DEFAULT_PRICING_DEVIN"], ]; let famTotal = 0; const seen = new Set(); diff --git a/tests/unit/provider-cost-data.test.ts b/tests/unit/provider-cost-data.test.ts new file mode 100644 index 0000000000..157769dfbf --- /dev/null +++ b/tests/unit/provider-cost-data.test.ts @@ -0,0 +1,48 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { getModelPricing, KNOWN_MODEL_PRICING } from "../../open-sse/services/providerCostData.ts"; + +test("provider-specific pricing wins over a generic model fallback", () => { + const genericKey = "provider-price-test-model"; + const providerKey = `devin-cli/${genericKey}`; + const previousGeneric = KNOWN_MODEL_PRICING[genericKey]; + const previousProvider = KNOWN_MODEL_PRICING[providerKey]; + + KNOWN_MODEL_PRICING[genericKey] = { + inputCostPer1M: 9, + outputCostPer1M: 90, + isFree: false, + }; + KNOWN_MODEL_PRICING[providerKey] = { + inputCostPer1M: 1, + outputCostPer1M: 10, + isFree: false, + }; + + try { + assert.deepEqual(getModelPricing("devin-cli", genericKey), { + inputCostPer1M: 1, + outputCostPer1M: 10, + isFree: false, + }); + } finally { + if (previousGeneric) KNOWN_MODEL_PRICING[genericKey] = previousGeneric; + else delete KNOWN_MODEL_PRICING[genericKey]; + if (previousProvider) KNOWN_MODEL_PRICING[providerKey] = previousProvider; + else delete KNOWN_MODEL_PRICING[providerKey]; + } +}); + +test("tier pricing reads the exact Devin provider/model rate", () => { + assert.deepEqual(getModelPricing("devin-cli", "gpt-5-6-luna-max"), { + inputCostPer1M: 0.2, + outputCostPer1M: 1.2, + isFree: false, + }); + assert.deepEqual(getModelPricing("devin-cli", "gpt-5-6-luna-max-priority"), { + inputCostPer1M: 0.4, + outputCostPer1M: 2.4, + isFree: false, + }); +}); diff --git a/tests/unit/provider-models-config.test.ts b/tests/unit/provider-models-config.test.ts index 793605cc4f..51a5f62d18 100644 --- a/tests/unit/provider-models-config.test.ts +++ b/tests/unit/provider-models-config.test.ts @@ -141,16 +141,22 @@ test("GitHub Copilot registry reflects the current supported model lineup", () = assert.equal(ids.includes("gemini-3-flash-preview"), false); }); -test("Claude flagship catalogs keep Fable 5 first", () => { - for (const provider of ["anthropic", "cc", "cw", "gh", "ghe-copilot"]) { +test("verified Anthropic launch catalogs keep Fable 5.1 first", () => { + for (const provider of ["anthropic", "cc", "cw"]) { assert.equal( getProviderModels(provider)[0]?.id, - "claude-fable-5", + "claude-fable-5-1", `${provider} must list the strongest Claude model first` ); } }); +test("Copilot catalogs retain Fable 5 until their Fable 5.1 IDs are verified", () => { + for (const provider of ["gh", "ghe-copilot"]) { + assert.equal(getProviderModels(provider)[0]?.id, "claude-fable-5"); + } +}); + test("Kiro registry exposes the current CLI model lineup with context windows", () => { const kiroModels = getProviderModels("kr"); const byId = new Map(kiroModels.map((model) => [model.id, model])); From cb38bfa6ee105439f1c160aadfe54de7789ad4ea Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 08:10:49 -0300 Subject: [PATCH 020/143] feat(video): redact raw client-snapshot transcript fields in the detailed log (#12150 P2) (#12528) * feat(video): redact raw client-snapshot transcript fields in the detailed log (#12150 P2) clientRawRequest.body is captured before the guardrail chain runs and persisted verbatim by reqLogger.logClientRawRequest, so it retained the client's raw transcript/audioTranscript cue text on video parts even after P1's description redaction. Add redactVideoTranscriptFieldsForLog (new, dependency-light module) and wire it at the logClientRawRequest call site, gated on videoBridgeObserved: redacts the structured transcript fields in the LOGGED copy only, never the body sent to the provider or returned to the client. * refactor(video): extract the guarded client-snapshot log call to keep chatCore within its size budget Fast Quality Gates check:file-size flagged chatCore.ts growing past its frozen ceiling (5985 > 5976) from the P2a wiring. Move the guarded logClientRawRequest call into logClientRawRequestRedacted (new export in videoBridgeSnapshotRedaction.ts, which already owns the redaction), collapsing the inline if-block at the chatCore.ts call site to a single call. Net -4 lines vs the pre-P2a base. Behavior unchanged: non-observed still logs the exact same clientRawRequest.body reference; observed still logs the redacted clone. --- open-sse/handlers/chatCore.ts | 12 +- .../videoBridgeSnapshotRedaction.ts | 135 ++++++++++++ .../videoBridgeSnapshotRedaction.test.ts | 207 ++++++++++++++++++ tests/unit/video-bridge-log-redaction.test.ts | 78 +++++++ 4 files changed, 424 insertions(+), 8 deletions(-) create mode 100644 src/lib/guardrails/videoBridgeSnapshotRedaction.ts create mode 100644 tests/unit/guardrails/videoBridgeSnapshotRedaction.test.ts diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 8dbac0018b..d28711ce12 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -360,6 +360,7 @@ import { deleteSessionAccountAffinity } from "@/lib/db/sessionAccountAffinity"; import { getCacheControlSettings } from "@/lib/cacheControlSettings"; import { guardrailRegistry } from "@/lib/guardrails"; import type { VideoBridgeLogRedactionEntry } from "@/lib/guardrails/videoBridge"; +import { logClientRawRequestRedacted } from "@/lib/guardrails/videoBridgeSnapshotRedaction"; import { shouldPreserveCacheControl, resolveConnectionCacheOverride, @@ -1211,14 +1212,9 @@ export async function handleChatCore({ }); const pendingScope = { id: pendingRequestId, model, provider, connectionId: pendingConnId }; const providerRequestCapture = createPreparedRequestLogger(reqLogger, pendingScope); - // 0. Log client raw request (before format conversion) - if (clientRawRequest) { - reqLogger.logClientRawRequest( - clientRawRequest.endpoint, - clientRawRequest.body, - clientRawRequest.headers - ); - } + // 0. Log client raw request (before format conversion) — redacts video transcript + // cues in the logged copy only; see videoBridgeSnapshotRedaction.ts. + logClientRawRequestRedacted(reqLogger, clientRawRequest, videoBridgeObserved); const reasoningRouteDecision = body && typeof body === "object" ? (body as Record)._omnirouteReasoningRouteTrace diff --git a/src/lib/guardrails/videoBridgeSnapshotRedaction.ts b/src/lib/guardrails/videoBridgeSnapshotRedaction.ts new file mode 100644 index 0000000000..8738b47cfc --- /dev/null +++ b/src/lib/guardrails/videoBridgeSnapshotRedaction.ts @@ -0,0 +1,135 @@ +/** + * #12150 P2 surface 1 (the dominant transcript-retention leak): structured redaction of + * video transcript fields on the CLIENT-REQUEST SNAPSHOT that lands in the detailed-log + * artifact. + * + * `clientRawRequest.body` (src/sse/handlers/chat/clientRawRequest.ts::buildClientRawRequest) + * is a bounded clone of the client's ORIGINAL request, captured BEFORE the guardrail chain + * runs, and persisted verbatim by `reqLogger.logClientRawRequest` + * (open-sse/handlers/chatCore.ts). Because it predates the video-bridge guardrail's own + * description redaction (#12150 P1 — see `describeVideoPart`'s `descriptionRedacted` in + * videoBridgeHelpers.ts), it still carries the client's raw `transcript` / `audioTranscript` + * cue text on any video part. This module redacts THAT COPY ONLY: the body sent to the + * provider and the response returned to the client are never touched here. + * + * Deliberately a standalone, dependency-light module — NOT part of videoBridgeHelpers.ts, + * which pulls in the frame-extraction broker client, audio/video fusion, contact-sheet + * composition and `sharp` for real video processing. The chat request hot path statically + * imports whatever module owns the `logClientRawRequest` call site on every request + * (video or not), so keeping this redaction free of that dependency chain matters for cold + * start and blast radius. + * + * The field walk mirrors `extractVideoParts` (videoBridgeHelpers.ts): for each content part, + * the candidate objects are the part itself, its `video_url` sub-object, and its `source` + * sub-object (the same three checked there) — but this walk is deliberately WIDER: any of + * those objects carrying a `transcript`/`audioTranscript` key gets redacted regardless of + * the part's `type`/shape. Those two field names are video-cue-only in this codebase's + * request contract, so matching on field presence rather than a shape allowlist is strictly + * safer (fails closed on an unusual or future video shape instead of silently skipping it). + * Redaction is a structured field substitution, not a scan over rendered text, so it cannot + * be bypassed by adversarial cue content (see the discarded regex approach recorded in the + * #12150 design doc, `_tasks/superpowers/specs/2026-09-01-video-transcript-retention-design.md`). + */ + +// Kept as a local literal (not imported from videoBridgeHelpers.ts) for the reason in the +// file header above. Equality with the canonical `VIDEO_TRANSCRIPT_REDACTION_PLACEHOLDER` +// export is enforced by a drift test in +// tests/unit/guardrails/videoBridgeSnapshotRedaction.test.ts. +const REDACTION_PLACEHOLDER = "[redacted-video-transcript]"; + +const TRANSCRIPT_FIELD_NAMES = ["transcript", "audioTranscript"] as const; +const NESTED_SUBOBJECT_KEYS = ["video_url", "source"] as const; +const CONTAINER_KEYS = ["messages", "input"] as const; + +type UnknownRecord = Record; + +function isPlainRecord(value: unknown): value is UnknownRecord { + return Boolean(value) && typeof value === "object" && !Array.isArray(value); +} + +/** + * Overwrites transcript field VALUES in place on `part` and its `video_url`/`source` + * sub-objects. Only ever called on a part that already lives inside the function's own + * `structuredClone`, never on caller-owned data. Keys are overwritten, never deleted, so + * downstream shape/observability (e.g. "this part had a transcript") is preserved. + */ +function redactTranscriptFieldsOnPart(part: unknown): void { + if (!isPlainRecord(part)) return; + const candidates: UnknownRecord[] = [part]; + for (const key of NESTED_SUBOBJECT_KEYS) { + const nested = part[key]; + if (isPlainRecord(nested)) candidates.push(nested); + } + for (const candidate of candidates) { + for (const field of TRANSCRIPT_FIELD_NAMES) { + if (candidate[field] !== undefined) { + candidate[field] = REDACTION_PLACEHOLDER; + } + } + } +} + +function redactContentArray(content: unknown): void { + if (!Array.isArray(content)) return; + for (const part of content) { + redactTranscriptFieldsOnPart(part); + } +} + +/** `messages` (Chat Completions) or `input` (Responses API) — either container shape. */ +function redactContainer(container: unknown): void { + if (!Array.isArray(container)) return; + for (const message of container) { + if (!isPlainRecord(message)) continue; + redactContentArray(message.content); + } +} + +/** + * Returns a NEW structure with every video transcript cue field value replaced by the + * redaction placeholder. Never mutates `body` — the caller (chatCore.ts) must keep passing + * the untouched original to translation/dispatch/response. A non-object `body`, or one with + * neither `messages` nor `input`, or with video parts that carry no transcript field, is + * returned as an equivalent (cloned) structure with nothing to change. + */ +export function redactVideoTranscriptFieldsForLog(body: unknown): unknown { + if (!isPlainRecord(body)) return body; + const cloned = structuredClone(body) as UnknownRecord; + for (const key of CONTAINER_KEYS) { + redactContainer(cloned[key]); + } + return cloned; +} + +interface ClientRawRequestLike { + endpoint: unknown; + body: unknown; + headers?: unknown; +} + +interface RequestLoggerLike { + logClientRawRequest: (endpoint: unknown, body: unknown, headers?: unknown) => void; +} + +/** + * Call-site wrapper for `reqLogger.logClientRawRequest` (chatCore.ts's "0. Log client raw + * request" step): keeps the null-check and the observed/redacted guard out of chatCore.ts, + * which is a size-frozen file (`config/quality/file-size-baseline.json`) — this owns the + * redaction, so it owns the one guarded call site that applies it. Behavior is identical to + * the inline block it replaces: a non-observed request logs `clientRawRequest.body` by the + * exact same reference (no clone); an observed one logs the redacted clone. + */ +export function logClientRawRequestRedacted( + reqLogger: RequestLoggerLike, + clientRawRequest: ClientRawRequestLike | null | undefined, + videoBridgeObserved: boolean +): void { + if (!clientRawRequest) return; + reqLogger.logClientRawRequest( + clientRawRequest.endpoint, + videoBridgeObserved + ? redactVideoTranscriptFieldsForLog(clientRawRequest.body) + : clientRawRequest.body, + clientRawRequest.headers + ); +} diff --git a/tests/unit/guardrails/videoBridgeSnapshotRedaction.test.ts b/tests/unit/guardrails/videoBridgeSnapshotRedaction.test.ts new file mode 100644 index 0000000000..3f48bee7dc --- /dev/null +++ b/tests/unit/guardrails/videoBridgeSnapshotRedaction.test.ts @@ -0,0 +1,207 @@ +// #12150 P2 surface 1 (the dominant transcript-retention leak): pure-helper coverage for +// redactVideoTranscriptFieldsForLog — the structured redaction applied to the RAW +// client-request snapshot (clientRawRequest.body) before it is persisted by +// reqLogger.logClientRawRequest (open-sse/handlers/chatCore.ts). See +// src/lib/guardrails/videoBridgeSnapshotRedaction.ts for the full design rationale +// (deliberately dependency-light; field-presence match rather than a shape allowlist). +import assert from "node:assert/strict"; +import test from "node:test"; + +import { redactVideoTranscriptFieldsForLog } from "../../../src/lib/guardrails/videoBridgeSnapshotRedaction.ts"; +// Heavy import is fine here (test only, never in the production module under test) — used +// solely to prove the local placeholder literal never drifts from the canonical P1 constant. +import { VIDEO_TRANSCRIPT_REDACTION_PLACEHOLDER } from "../../../src/lib/guardrails/videoBridgeHelpers.ts"; + +type JsonRecord = Record; + +function asRecord(value: unknown): JsonRecord { + return value as JsonRecord; +} + +function contentAt( + body: unknown, + container: "messages" | "input", + messageIndex: number +): JsonRecord[] { + const messages = asRecord(body)[container] as JsonRecord[]; + return messages[messageIndex].content as JsonRecord[]; +} + +test("redacts transcript and audioTranscript directly on a video part (messages container)", () => { + const body = { + model: "gpt-x", + messages: [ + { role: "system", content: "sys" }, + { + role: "user", + content: [ + { type: "text", text: "look at this video" }, + { + type: "input_video", + video_url: "https://example.com/clip.mp4", + transcript: { cues: [{ text: "secret words", startSeconds: 0, endSeconds: 2 }] }, + audioTranscript: { cues: [{ text: "audio secret", startSeconds: 0, endSeconds: 1 }] }, + }, + ], + }, + ], + }; + + const result = redactVideoTranscriptFieldsForLog(body); + assert.notEqual(result, body, "must return a new structure, not the same reference"); + + const videoPart = contentAt(result, "messages", 1)[1]; + assert.equal(videoPart.transcript, "[redacted-video-transcript]"); + assert.equal(videoPart.audioTranscript, "[redacted-video-transcript]"); + // The video ref itself and the sibling non-video part must survive untouched. + assert.equal(videoPart.video_url, "https://example.com/clip.mp4"); + assert.equal(contentAt(result, "messages", 1)[0].text, "look at this video"); + assert.equal(asRecord(result).messages, asRecord(result).messages); // sanity: still an array + + const serialized = JSON.stringify(result); + assert.ok(!serialized.includes("secret words"), "raw video transcript must not survive"); + assert.ok(!serialized.includes("audio secret"), "raw audio transcript must not survive"); +}); + +test("redacts a transcript nested under the video_url sub-object", () => { + const body = { + messages: [ + { + role: "user", + content: [ + { + type: "video_url", + video_url: { + url: "https://example.com/nested.mp4", + transcript: { cues: [{ text: "nested secret" }] }, + }, + }, + ], + }, + ], + }; + + const result = redactVideoTranscriptFieldsForLog(body); + const part = contentAt(result, "messages", 0)[0]; + const videoUrl = part.video_url as JsonRecord; + assert.equal(videoUrl.transcript, "[redacted-video-transcript]"); + assert.equal(videoUrl.url, "https://example.com/nested.mp4"); + assert.ok(!JSON.stringify(result).includes("nested secret")); +}); + +test("redacts a transcript nested under the source sub-object (video_source shape)", () => { + const body = { + messages: [ + { + role: "user", + content: [ + { + type: "video_source", + source: { + type: "url", + url: "https://example.com/source.mp4", + audioTranscript: { cues: [{ text: "source secret" }] }, + }, + }, + ], + }, + ], + }; + + const result = redactVideoTranscriptFieldsForLog(body); + const part = contentAt(result, "messages", 0)[0]; + const source = part.source as JsonRecord; + assert.equal(source.audioTranscript, "[redacted-video-transcript]"); + assert.equal(source.url, "https://example.com/source.mp4"); + assert.ok(!JSON.stringify(result).includes("source secret")); +}); + +test("covers the input container (Responses API shape)", () => { + const body = { + model: "gpt-x", + input: [ + { + role: "user", + content: [ + { + type: "input_video", + video_url: "https://example.com/input.mp4", + transcript: { cues: [{ text: "input secret" }] }, + }, + ], + }, + ], + }; + + const result = redactVideoTranscriptFieldsForLog(body); + const part = contentAt(result, "input", 0)[0]; + assert.equal(part.transcript, "[redacted-video-transcript]"); + assert.ok(!JSON.stringify(result).includes("input secret")); +}); + +test("does not mutate the input", () => { + const body = { + messages: [ + { + role: "user", + content: [ + { + type: "input_video", + video_url: "https://example.com/clip.mp4", + transcript: { cues: [{ text: "secret words" }] }, + audioTranscript: { cues: [{ text: "audio secret" }] }, + }, + ], + }, + ], + }; + const before = JSON.parse(JSON.stringify(body)); + + redactVideoTranscriptFieldsForLog(body); + + assert.deepEqual(body, before, "input object must be byte-identical after the call"); +}); + +test("a non-video body is returned unchanged", () => { + const body = { + model: "gpt-x", + messages: [ + { role: "system", content: "sys" }, + { role: "user", content: "hello, no video here" }, + ], + }; + + const result = redactVideoTranscriptFieldsForLog(body); + assert.deepEqual(result, body); +}); + +test("a body with a video part but no transcript field is unchanged", () => { + const body = { + messages: [ + { + role: "user", + content: [{ type: "input_video", video_url: "https://example.com/no-transcript.mp4" }], + }, + ], + }; + + const result = redactVideoTranscriptFieldsForLog(body); + assert.deepEqual(result, body); +}); + +test("the redaction placeholder matches the canonical P1 constant (no drift)", () => { + const body = { + messages: [ + { + role: "user", + content: [ + { type: "input_video", video_url: "https://example.com/clip.mp4", transcript: "raw" }, + ], + }, + ], + }; + + const result = redactVideoTranscriptFieldsForLog(body); + const part = contentAt(result, "messages", 0)[0]; + assert.equal(part.transcript, VIDEO_TRANSCRIPT_REDACTION_PLACEHOLDER); +}); diff --git a/tests/unit/video-bridge-log-redaction.test.ts b/tests/unit/video-bridge-log-redaction.test.ts index d190db80c4..268102d246 100644 --- a/tests/unit/video-bridge-log-redaction.test.ts +++ b/tests/unit/video-bridge-log-redaction.test.ts @@ -26,6 +26,8 @@ import fs from "node:fs"; import os from "node:os"; import path from "node:path"; +import { logClientRawRequestRedacted } from "../../src/lib/guardrails/videoBridgeSnapshotRedaction.ts"; + const testDataDir = fs.mkdtempSync(path.join(os.tmpdir(), "omni-video-log-redaction-test-")); process.env.DATA_DIR = testDataDir; @@ -265,3 +267,79 @@ test("Scenario A (adversarial review): a message prepended AFTER the guardrail b "the prepended system message must be untouched" ); }); + +// #12150 P2 surface 1 (the dominant transcript-retention leak): the RAW client-request +// snapshot passed to reqLogger.logClientRawRequest (open-sse/handlers/chatCore.ts's +// "0. Log client raw request" step) is a DIFFERENT sink from persistAttemptLogs above — +// it is captured before the guardrail chain even runs, so it carries the client's raw +// `transcript`/`audioTranscript` FIELDS on a structured video part, not a flattened +// description string. Importing the real chatCore.ts here would pull the full +// request-pipeline dependency graph (executors, providers, combo routing, DB-backed +// settings, ...) into the test just to reach one guarded call a few hundred lines into +// a 5900+ line handler, for no additional proof beyond what's below — so this calls the +// REAL exported `logClientRawRequestRedacted` (the exact function chatCore.ts's call site +// invokes, post file-size-refactor) against a fake logClientRawRequest. The pure redaction +// helper itself has its own thorough suite in +// tests/unit/guardrails/videoBridgeSnapshotRedaction.test.ts. +function fakeReqLogger() { + const calls: unknown[] = []; + return { + calls, + logClientRawRequest(_endpoint: unknown, body: unknown, _headers?: unknown) { + calls.push(body); + }, + }; +} + +test("surface 2 (raw snapshot): the fake logClientRawRequest receives a redacted snapshot only when videoBridgeObserved is true", () => { + const rawBody = { + model: "openai/gpt-x", + messages: [ + { + role: "user", + content: [ + { type: "text", text: "look at this video" }, + { + type: "input_video", + video_url: "https://example.com/clip.mp4", + transcript: { cues: [{ text: SECRET, startSeconds: 0, endSeconds: 2 }] }, + }, + ], + }, + ], + }; + const clientRawRequest = { endpoint: "/v1/chat/completions", body: rawBody, headers: {} }; + + const observedLogger = fakeReqLogger(); + logClientRawRequestRedacted(observedLogger, clientRawRequest, true); + const observedSnapshot = observedLogger.calls[0]; + assert.ok( + !JSON.stringify(observedSnapshot).includes(SECRET), + "an observed request must not log the raw transcript" + ); + assert.notEqual( + observedSnapshot, + rawBody, + "the observed path must log a redacted CLONE, not the original reference" + ); + assert.ok( + JSON.stringify(rawBody).includes(SECRET), + "clientRawRequest.body itself must stay untouched for every other consumer (translation/dispatch)" + ); + + const nonObservedLogger = fakeReqLogger(); + logClientRawRequestRedacted(nonObservedLogger, clientRawRequest, false); + assert.equal( + nonObservedLogger.calls[0], + rawBody, + "the non-observed path must log the exact same object reference — byte-identical, no clone" + ); + + const skippedLogger = fakeReqLogger(); + logClientRawRequestRedacted(skippedLogger, null, true); + assert.equal( + skippedLogger.calls.length, + 0, + "a missing clientRawRequest must not call logClientRawRequest at all (mirrors the old if-guard)" + ); +}); From 3740839e2a6741ec9e4bdae5b4ce976fdb6a35eb Mon Sep 17 00:00:00 2001 From: opensource-elearning <159253500+opensource-elearning@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:50:58 +0530 Subject: [PATCH 021/143] fix(combo): fall back to full pool when collapsed sole survivor is context-too-small (#12278) Validado em lote sobre o tip de release/v3.8.51: boardou sem conflito, typecheck:core limpo e 17/17 em tests/unit/combo-context-window-filter.test.ts. Obrigado, @opensource-elearning. --- open-sse/services/combo/comboStructure.ts | 44 +++++++-- .../unit/combo-context-window-filter.test.ts | 92 +++++++++++++++++++ 2 files changed, 130 insertions(+), 6 deletions(-) diff --git a/open-sse/services/combo/comboStructure.ts b/open-sse/services/combo/comboStructure.ts index 137561fbaa..9065e123b2 100644 --- a/open-sse/services/combo/comboStructure.ts +++ b/open-sse/services/combo/comboStructure.ts @@ -807,16 +807,48 @@ export function filterTargetsByRequestCompatibility( return []; } + // #12273: a sole survivor whose catalog window is known-too-small is a + // guaranteed context_length_exceeded. Restore the remaining pool so combo.ts + // can still try larger-context targets. Unknown context (`null`) is advisory + // and must not resurrect hard-rejected targets (vision / output / tools). + if ( + compatible.length === 1 && + (targetReasons.get(compatible[0]) || []).includes("context_window") + ) { + // #8332: never restore a confirmed-non-vision target onto an image request. + const restored = requirements.requiresVision + ? targets.filter((target) => !isVisionIncompatibleTarget(target, requirements)) + : targets; + if (restored.length > compatible.length) { + log.warn( + "COMBO", + `${label}: single compatible target ${compatible[0].modelStr} has known context too small for ${requirements.requiredContextTokens} token request; falling back to full pool (#12273)` + ); + return restored; + } + } + log.info( "COMBO", `${label}: kept ${compatible.length}/${targets.length} targets for request requirements` ); - log.debug?.( - "COMBO", - `${label}: rejected targets ${rejected - .map((entry) => `${entry.target.modelStr}(${entry.reasons.join("+")})`) - .join(", ")}` - ); + // #12273: When pool collapses significantly, log rejection reasons at info + // level so the cause is diagnosable without enabling debug logging. + if (compatible.length <= 2 && targets.length > 4) { + log.info( + "COMBO", + `${label}: rejected targets ${rejected + .map((entry) => `${entry.target.modelStr}(${entry.reasons.join("+")})`) + .join(", ")}` + ); + } else { + log.debug?.( + "COMBO", + `${label}: rejected targets ${rejected + .map((entry) => `${entry.target.modelStr}(${entry.reasons.join("+")})`) + .join(", ")}` + ); + } return compatible; } diff --git a/tests/unit/combo-context-window-filter.test.ts b/tests/unit/combo-context-window-filter.test.ts index a78c3d5ba5..74d2646f3e 100644 --- a/tests/unit/combo-context-window-filter.test.ts +++ b/tests/unit/combo-context-window-filter.test.ts @@ -450,3 +450,95 @@ test("without an override the small-catalog target is ordered last for the large ["unit-override/big", "unit-override/capped"] ); }); + +// #12273: real Claude Code requests always carry `tools`, and the auto/coding +// pool mixes coding-capable providers with providers whose catalog marks +// toolCalling=false. Those non-coding targets are HARD-rejected (tools), so the +// compat filter can collapse the whole pool to a single too-small-context +// coding model (e.g. mimo-v2.5-free at 200k) for a much larger request — the +// larger-context model was never assembled into the candidate pool. Routing to +// that sole survivor is a guaranteed context_length_exceeded, so the filter +// must fall back to the full pool instead of silently pinning the request. +test("#12273 single known-too-small survivor falls back to the full pool", () => { + saveModelsDevCapabilities({ + "unit-collapse": { + small: capabilityEntry(200_000), + }, + "unit-noncoding": { + nocoder: { ...capabilityEntry(1_000_000), tool_call: false }, + }, + }); + const body = { + ...bigContextBody(300_000), + tools: [{ type: "function" }], // Claude Code always sends tools + }; + + const out = filterTargetsByRequestCompatibility( + [target("unit-collapse/small"), target("unit-noncoding/nocoder")], + body, + noopLog + ); + + // nocoder is hard-rejected (toolCalling=false); small (200k) is the only + // compatible survivor but is known to be too small for a 300k request, so the + // filter returns the full pool rather than dispatch to a guaranteed failure. + assert.deepEqual( + out.map((entry) => entry.modelStr), + ["unit-collapse/small", "unit-noncoding/nocoder"] + ); +}); + +// Guard against regression: when the single survivor's window DOES fit the +// request, the filter still collapses (existing behavior preserved). +test("#12273 single compatible target that fits is still collapsed", () => { + saveModelsDevCapabilities({ + "unit-collapse": { + big: capabilityEntry(1_000_000), + }, + "unit-noncoding": { + nocoder: { ...capabilityEntry(1_000_000), tool_call: false }, + }, + }); + const body = { + ...bigContextBody(300_000), + tools: [{ type: "function" }], + }; + + const out = filterTargetsByRequestCompatibility( + [target("unit-collapse/big"), target("unit-noncoding/nocoder")], + body, + noopLog + ); + + // big (1M) fits the 300k request, so the collapse is legitimate. + assert.deepEqual( + out.map((entry) => entry.modelStr), + ["unit-collapse/big"] + ); +}); + +// #12278: unknown context is advisory, not "known too small". Collapsing to a +// single survivor whose context limit is unknown must NOT restore hard-rejected +// targets (output_tokens here; vision is covered by combo-vision-aware-routing). +test("#12273 unknown-context sole survivor does not restore hard-rejected targets", () => { + saveModelsDevCapabilities({ + "unit-output": { + tiny: capabilityEntryWithLimits(128_000, 128_000, 4096), + }, + }); + const body = { + messages: [{ role: "user", content: "hello" }], + max_tokens: 32_000, + }; + + const out = filterTargetsByRequestCompatibility( + [target("unit-unknown/mystery"), target("unit-output/tiny")], + body, + noopLog + ); + + assert.deepEqual( + out.map((entry) => entry.modelStr), + ["unit-unknown/mystery"] + ); +}); From 8df944cd4670171ebca2f0b003f5955fca3d2781 Mon Sep 17 00:00:00 2001 From: opensource-elearning <159253500+opensource-elearning@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:54:32 +0530 Subject: [PATCH 022/143] feat(browser): adopt Obscura as primary headless browser engine with Chromium fallback (#12286) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado sobre o tip de release/v3.8.51 após reconciliar o base-drift. O pool de browsers ganhou um caminho **headed** (`headedBrowser`/`headedLaunching`, `resolvePlainBrowserLaunchOptions`, estado de launch por modo) depois que esta branch forkou. O PR reescrevia `launchBrowser()` no modelo de browser único contra o qual foi escrito, o que teria **removido o suporte headed**. Em vez disso, reapliquei a preferência pelo Obscura dentro do ramo headless de `launchBrowserInstance()`, à frente do cloakbrowser e do Chromium puro — um browser headed precisa ser um Chromium com janela real, então a preferência de engine é só do caminho headless. `state.engine` alimenta `isStealth` e `getBrowserPoolStatus()`, e o shutdown zera o engine sem matar o servidor Obscura compartilhado (dono: `./obscura.ts`). typecheck:core limpo e 3/3 em tests/unit/obscura-integration.test.ts. Obrigado, @opensource-elearning. --- .env.example | 9 + docs/reference/ENVIRONMENT.md | 3 + open-sse/executors/cloudflare-playground.ts | 23 ++- open-sse/services/browserPool.ts | 31 +++- open-sse/services/obscura.ts | 163 ++++++++++++++++++ tests/unit/obscura-integration.test.ts | 155 +++++++++++++++++ .../webpack-create-require-warning.test.ts | 4 + 7 files changed, 379 insertions(+), 9 deletions(-) create mode 100644 open-sse/services/obscura.ts create mode 100644 tests/unit/obscura-integration.test.ts diff --git a/.env.example b/.env.example index 4e6438b593..81b03d5b11 100644 --- a/.env.example +++ b/.env.example @@ -1524,6 +1524,15 @@ CURSOR_USER_AGENT="Cursor/3.4" # request into the browser-backed path. # OMNIROUTE_BROWSER_POOL=on # WEB_COOKIE_USE_BROWSER=0 +# Obscura (https://github.com/h4ckf0r0day/obscura) is the primary headless +# engine: a lightweight CDP server the pool and cloudflare-playground connect +# to before falling back to Chromium. Unset OBSCURA_BIN to auto-detect from +# PATH; set OBSCURA_CDP_ENDPOINT to reuse an already-running Obscura instead +# of spawning one; set OBSCURA_PORT to pin the spawned serve port. +# Used by: open-sse/services/obscura.ts +# OBSCURA_BIN= +# OBSCURA_CDP_ENDPOINT= +# OBSCURA_PORT= # ── Kimi Web (international kimi.ai Connect-RPC) ── # Used by: open-sse/executors/kimi-web.ts. Override the base/chat URLs only if diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index ace16cf6c9..9dd2fff9f0 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -776,6 +776,9 @@ REQUEST_TIMEOUT_MS (global override) | `OMNIROUTE_NOTION_TLS_TIMEOUT_MS` | `30000` | Native wreq-js request timeout (`notionTlsClient.ts`); `notion-web` raises it per request to `180000` for long generations. | | `OMNIROUTE_NOTION_TLS_GRACE_MS` | `10000` | Absolute JS hard-deadline grace added on top of the native timeout. | | `OMNIROUTE_BROWSER_POOL` | `on` | Shared Playwright browser pool for browser-backed web-cookie chat (`browserPool.ts`); set `off` to disable. | +| `OBSCURA_BIN` | `auto-detect` | Path to the `obscura` binary used as the primary engine by the browser pool and Cloudflare Playground executor (`open-sse/services/obscura.ts`); auto-detected from the system PATH when unset. | +| `OBSCURA_CDP_ENDPOINT` | _(unset)_ | Point at an already-running Obscura (`http://host:port`) instead of spawning one; the module does not own that process (`open-sse/services/obscura.ts`). | +| `OBSCURA_PORT` | `random free port` | Explicit port for the spawned `obscura serve`; a free port is chosen automatically when unset (`open-sse/services/obscura.ts`). | | `WEB_COOKIE_USE_BROWSER` | `0` | Opt a web-cookie chat request into the browser-backed path (`browserBackedChat.ts`); `1` to enable. | | `KIMI_WEB_BASE_URL` | `https://www.kimi.ai` | Base URL for the Kimi Web (international kimi.ai Connect-RPC) executor (`kimi-web.ts`); override only for mirror/proxy endpoints. | | `KIMI_WEB_CHAT_URL` | `/apiv2/kimi.gateway.chat.v1.ChatService/Chat` | Full chat endpoint for the Kimi Web executor (`kimi-web.ts`). | diff --git a/open-sse/executors/cloudflare-playground.ts b/open-sse/executors/cloudflare-playground.ts index ba309f1eed..249e069da3 100644 --- a/open-sse/executors/cloudflare-playground.ts +++ b/open-sse/executors/cloudflare-playground.ts @@ -36,6 +36,7 @@ import { randomUUID } from "crypto"; import { BaseExecutor, type ExecuteInput } from "./base.ts"; import { makeExecutorErrorResult as makeErrorResult } from "../utils/error.ts"; +import { connectObscuraBrowser } from "../services/obscura.ts"; import type { Browser, Page } from "playwright"; export const PLAYGROUND_URL = "https://playground.ai.cloudflare.com/"; @@ -296,14 +297,22 @@ export class PlaywrightCfTransport implements CfTransport { config: CfTransportConfig ): Promise<{ ok: true } | { ok: false; status: number; message: string }> { try { + // #12274: prefer the shared Obscura browser (browser-grade TLS fingerprint + // on the WS upgrade, ~30MB) over a full Chromium per request; fall back to + // a direct Chromium launch when Obscura is unavailable. + const obscura = await connectObscuraBrowser(); const playwright = await importPlaywright(); - const executablePath = - this.chromeExecutablePath ?? process.env.CLOUDFLARE_PLAYGROUND_CHROME_PATH; - this.browser = await playwright.chromium.launch({ - ...(executablePath ? { executablePath } : {}), - headless: true, - args: BROWSER_ARGS, - }); + if (obscura) { + this.browser = obscura.browser; + } else { + const executablePath = + this.chromeExecutablePath ?? process.env.CLOUDFLARE_PLAYGROUND_CHROME_PATH; + this.browser = await playwright.chromium.launch({ + ...(executablePath ? { executablePath } : {}), + headless: true, + args: BROWSER_ARGS, + }); + } const context = await this.browser.newContext({ userAgent: PLAYGROUND_UA }); const page = await context.newPage(); this.page = page; diff --git a/open-sse/services/browserPool.ts b/open-sse/services/browserPool.ts index bcab2dc174..51a1abb44b 100644 --- a/open-sse/services/browserPool.ts +++ b/open-sse/services/browserPool.ts @@ -26,6 +26,8 @@ import { Buffer } from "node:buffer"; +import { connectObscuraBrowser } from "./obscura.ts"; + type Browser = import("playwright").Browser; type BrowserContext = import("playwright").BrowserContext; type Page = import("playwright").Page; @@ -86,8 +88,12 @@ function createBrowserPoolMetrics(): BrowserPoolMetrics { }; } +type PoolEngine = "obscura" | "cloakbrowser" | "chromium"; + interface PoolState { browser: Browser | null; + /** Engine backing the headless browser, for metrics and stealth detection. */ + engine: PoolEngine | null; headedBrowser: Browser | null; contexts: Map; pendingContexts: Map>; @@ -110,6 +116,7 @@ const DEFAULT_USER_AGENT = const state: PoolState = { browser: null, + engine: null, headedBrowser: null, contexts: new Map(), pendingContexts: new Map(), @@ -288,13 +295,26 @@ async function launchBrowserInstance( options: BrowserPoolContextOptions, headless: boolean ): Promise { + // A headed browser must be a real windowed Chromium, so the engine + // preference below applies to the headless path only. if (!headless) { const { chromium } = await import("playwright"); return chromium.launch(resolvePlainBrowserLaunchOptions(options)); } + // #12274: prefer Obscura (lightweight, browser-grade CDP) over a full + // Chromium; fall back to cloakbrowser, then plain Chromium. Obscura's + // lifecycle (one shared `obscura serve` per process) lives in ./obscura.ts, + // so executors like cloudflare-playground reuse the same server. + const obscura = await connectObscuraBrowser(); + if (obscura) { + state.engine = "obscura"; + return obscura.browser; + } + const cloakLaunch = await resolveCloakLaunch(); if (cloakLaunch) { + state.engine = "cloakbrowser"; return cloakLaunch({ headless: true, args: ["--no-sandbox", "--disable-dev-shm-usage"], @@ -303,6 +323,7 @@ async function launchBrowserInstance( // Fallback: plain Playwright. Works for Claude web (cookie-only auth) but // DDG's VQD challenge will detect this Chromium build. + state.engine = "chromium"; const { chromium } = await import("playwright"); return chromium.launch(resolvePlainBrowserLaunchOptions(options)); } @@ -471,7 +492,7 @@ export async function acquireBrowserContext( launchBrowser(options), resolveBrowserContextProxy(key, options), ]); - const isStealth = headless && state.cloakLaunch !== null; + const isStealth = headless && (state.engine === "obscura" || state.cloakLaunch !== null); const context = await browser.newContext({ userAgent: options.userAgent || DEFAULT_USER_AGENT, locale: options.locale || "en-US", @@ -580,6 +601,10 @@ export async function shutdownPool(reason: string): Promise { } state.launching = null; state.headedLaunching = null; + // #12274: the shared Obscura server is owned by ./obscura.ts and reused by + // executors (cloudflare-playground), so closing the pool's CDP connection is + // enough — never kill the server here. + state.engine = null; state.lastActivity = Date.now(); // Avoid unused-parameter lint: log reason via debug if anyone hooks // process.on('exit') and prints state. @@ -590,6 +615,7 @@ export function getBrowserPoolStatus(): { enabled: boolean; contexts: number; browserRunning: boolean; + engine: PoolEngine | null; stealthAvailable: boolean; lastActivityAgoMs: number; } { @@ -597,7 +623,8 @@ export function getBrowserPoolStatus(): { enabled: isPoolEnabled(), contexts: state.contexts.size, browserRunning: state.browser !== null || state.headedBrowser !== null, - stealthAvailable: state.cloakLaunch !== null, + engine: state.engine, + stealthAvailable: state.engine === "obscura" || state.cloakLaunch !== null, lastActivityAgoMs: state.lastActivity === 0 ? -1 : Date.now() - state.lastActivity, }; } diff --git a/open-sse/services/obscura.ts b/open-sse/services/obscura.ts new file mode 100644 index 0000000000..23c4917214 --- /dev/null +++ b/open-sse/services/obscura.ts @@ -0,0 +1,163 @@ +/** + * obscura.ts — Shared Obscura browser engine (#12274). + * + * Obscura (https://github.com/h4ckf0r0day/obscura) is a lightweight Rust + * headless browser (~30MB resident) that speaks the Chrome DevTools Protocol. + * Playwright's `chromium.connectOverCDP` drives it like a real Chrome, so the + * browser pool and the cloudflare-playground executor can both use it without + * holding a 150-400MB Chromium process. + * + * Lifecycle: one Obscura `serve` process is spawned lazily on first use and + * shared for the server's lifetime. Callers receive a fresh CDP connection on + * demand; closing the connection does not stop the shared server. Set + * OBSCURA_CDP_ENDPOINT to point at an already-running Obscura instead of + * spawning one here (the process is then not owned by this module). The + * module is also disabled entirely when OMNIROUTE_BROWSER_POOL=off. + */ + +import { spawn, type ChildProcess } from "node:child_process"; +import { createServer } from "node:net"; + +export interface ObscuraConnection { + /** Playwright Browser connected over CDP to the shared Obscura server. */ + browser: import("playwright").Browser; + /** The spawned `obscura serve` process, or null when an external endpoint is used. */ + child: ChildProcess | null; +} + +let shared: { child: ChildProcess | null; endpoint: string } | null = null; +let starting: Promise<{ child: ChildProcess | null; endpoint: string } | null> | null = null; + +export function isObscuraUsable(): boolean { + const flag = process.env.OMNIROUTE_BROWSER_POOL; + if (flag === undefined) return true; + return flag !== "off" && flag !== "0" && flag !== "false"; +} + +function findFreePort(): Promise { + return new Promise((resolve, reject) => { + const srv = createServer(); + srv.once("error", reject); + srv.listen(0, "127.0.0.1", () => { + const address = srv.address(); + srv.close(() => { + if (address && typeof address === "object") resolve(address.port); + else reject(new Error("obscura: could not allocate a free port")); + }); + }); + }); +} + +async function obscuraBinaryPath(): Promise { + const bin = process.env.OBSCURA_BIN; + if (bin) return bin; + const { resolve } = await import("node:path"); + const { existsSync, accessSync, constants } = await import("node:fs"); + const dirs = (process.env.PATH || "").split(":"); + for (const dir of dirs) { + const candidate = resolve(dir, "obscura"); + try { + accessSync(candidate, constants.X_OK); + if (existsSync(candidate)) return candidate; + } catch { + /* not executable here — keep looking */ + } + } + return null; +} + +async function waitForCdpEndpoint(endpoint: string, timeoutMs: number): Promise { + const deadline = Date.now() + timeoutMs; + while (Date.now() < deadline) { + try { + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), 1500); + // Probe /json/version, not the base URL: Obscura's HTTP server answers + // the CDP info route, while a bare GET to "/" never completes a response. + const probe = endpoint.replace(/^ws/, "http").replace(/\/$/, "") + "/json/version"; + const res = await fetch(probe, { signal: controller.signal }); + clearTimeout(timer); + if (res.ok) return true; + } catch { + /* not up yet */ + } + await new Promise((r) => setTimeout(r, 250)); + } + return false; +} + +/** Ensure the shared Obscura server is up; returns its endpoint or null. */ +export async function ensureObscuraServer(): Promise<{ + child: ChildProcess | null; + endpoint: string; +} | null> { + if (!isObscuraUsable()) return null; + if (shared) return shared; + if (starting) return starting; + starting = (async () => { + const endpoint = process.env.OBSCURA_CDP_ENDPOINT; + if (endpoint) { + shared = { child: null, endpoint }; + return shared; + } + const bin = await obscuraBinaryPath(); + if (!bin) return null; + const port = Number(process.env.OBSCURA_PORT) || (await findFreePort()); + const child = spawn(bin, ["serve", "--port", String(port), "--host", "127.0.0.1"], { + stdio: ["ignore", "ignore", "pipe"], + }); + child.stderr?.on("data", () => {}); // obscura logs verbosely — swallow + const endpointForServer = `http://127.0.0.1:${port}`; + // A bad binary path (or a binary that cannot serve) must not hold the + // readiness wait for the full timeout: bail as soon as the child exits + // (or fails to spawn at all — 'exit' alone misses an ENOENT 'error'). + const died = new Promise((resolve) => { + child.once("exit", () => resolve(true)); + child.once("error", () => resolve(true)); + }); + const ready = await Promise.race([ + waitForCdpEndpoint(endpointForServer, 30_000), + died.then(() => false as const), + ]); + if (ready !== true) { + child.kill("SIGKILL"); + return null; + } + shared = { child, endpoint: endpointForServer }; + return shared; + })(); + try { + return await starting; + } finally { + starting = null; + } +} + +/** + * Connect Playwright to the shared Obscura server. Returns null when Obscura + * is disabled, not installed, or the server could not start (callers fall + * back to their previous Chromium strategy). + */ +export async function connectObscuraBrowser(): Promise { + const server = await ensureObscuraServer(); + if (!server) return null; + try { + const { chromium } = await import("playwright"); + const browser = await chromium.connectOverCDP(server.endpoint); + return { browser, child: server.child }; + } catch { + return null; + } +} + +/** Caution: this terminates the shared `obscura serve` process (process-lifetime anyway). */ +export function killSharedObscuraServer(): void { + if (shared?.child) { + try { + shared.child.kill("SIGKILL"); + } catch { + /* ignore */ + } + } + shared = null; +} diff --git a/tests/unit/obscura-integration.test.ts b/tests/unit/obscura-integration.test.ts new file mode 100644 index 0000000000..67a3006d79 --- /dev/null +++ b/tests/unit/obscura-integration.test.ts @@ -0,0 +1,155 @@ +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { spawn, spawnSync } from "node:child_process"; +import { createServer } from "node:net"; + +import { + connectObscuraBrowser, + ensureObscuraServer, + killSharedObscuraServer, + isObscuraUsable, +} from "../../open-sse/services/obscura.ts"; + +// #12274 — Obscura-first browser engine. The shared server is process-lifetime; +// each test resets it so suites run independently. When `obscura` is not +// installed the live tests skip; the null-return path is still covered. + +const BIN_RESULT = spawnSync("which", ["obscura"], { encoding: "utf8" }); +const HAS_OBSCURA = BIN_RESULT.status === 0 && BIN_RESULT.stdout.trim().length > 0; + +function freePort(): Promise { + return new Promise((resolve, reject) => { + const srv = createServer(); + srv.once("error", reject); + srv.listen(0, "127.0.0.1", () => { + const address = srv.address(); + srv.close(() => { + if (address && typeof address === "object") resolve(address.port); + else reject(new Error("no free port")); + }); + }); + }); +} + +async function waitForCdp(endpoint: string, timeoutMs = 30_000): Promise { + const deadline = Date.now() + timeoutMs; + while (Date.now() < deadline) { + try { + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), 1500); + // Probe /json/version (Obscura's bare "/" never completes a response). + const probe = endpoint.replace(/^ws/, "http").replace(/\/$/, "") + "/json/version"; + const res = await fetch(probe, { signal: controller.signal }); + clearTimeout(timer); + if (res.ok) return true; + } catch { + /* not up yet */ + } + await new Promise((r) => setTimeout(r, 250)); + } + return false; +} + +describe("obscura engine", () => { + it("respects OMNIROUTE_BROWSER_POOL=off", () => { + const original = process.env.OMNIROUTE_BROWSER_POOL; + process.env.OMNIROUTE_BROWSER_POOL = "off"; + try { + assert.equal(isObscuraUsable(), false); + } finally { + if (original === undefined) delete process.env.OMNIROUTE_BROWSER_POOL; + else process.env.OMNIROUTE_BROWSER_POOL = original; + } + }); + + it("is enabled by default (no env var)", () => { + const original = process.env.OMNIROUTE_BROWSER_POOL; + delete process.env.OMNIROUTE_BROWSER_POOL; + try { + assert.equal(isObscuraUsable(), true); + } finally { + if (original !== undefined) process.env.OMNIROUTE_BROWSER_POOL = original; + } + }); + + it("returns null when the binary is absent or cannot start", async () => { + killSharedObscuraServer(); + const originalBin = process.env.OBSCURA_BIN; + const originalEndpoint = process.env.OBSCURA_CDP_ENDPOINT; + process.env.OBSCURA_BIN = "/nonexistent/obscura"; + delete process.env.OBSCURA_CDP_ENDPOINT; + try { + const server = await ensureObscuraServer(); + assert.equal(server, null); + } finally { + if (originalBin === undefined) delete process.env.OBSCURA_BIN; + else process.env.OBSCURA_BIN = originalBin; + if (originalEndpoint === undefined) delete process.env.OBSCURA_CDP_ENDPOINT; + else process.env.OBSCURA_CDP_ENDPOINT = originalEndpoint; + killSharedObscuraServer(); + } + }); + + it("round-trips a page through Obscura when installed", async (t) => { + killSharedObscuraServer(); + if (!HAS_OBSCURA) { + t.skip("obscura binary not installed"); + return; + } + try { + const connection = await connectObscuraBrowser(); + assert.ok(connection, "expected a live Obscura connection"); + const { browser } = connection; + const context = await browser.newContext({ userAgent: "obscura-integration-test" }); + const page = await context.newPage(); + await page.goto("https://example.com", { waitUntil: "domcontentloaded", timeout: 30000 }); + const title = await page.title(); + assert.equal(title, "Example Domain"); + await context.close(); + await browser.close(); + } finally { + killSharedObscuraServer(); + } + }); + + it("connects to an external endpoint without owning its process", async (t) => { + void t; + killSharedObscuraServer(); + if (!HAS_OBSCURA) { + t.skip("obscura binary not installed"); + return; + } + // Standalone server we own outside the module, referenced as "external". + const port = await freePort(); + const child = spawn( + process.env.OBSCURA_BIN ?? "obscura", + ["serve", "--port", String(port), "--host", "127.0.0.1"], + { stdio: ["ignore", "ignore", "pipe"] } + ); + const endpoint = `http://127.0.0.1:${port}`; + const ready = await waitForCdp(endpoint); + if (!ready) { + child.kill("SIGKILL"); + t.skip("external obscura server did not come up"); + return; + } + const originalBin = process.env.OBSCURA_BIN; + const originalEndpoint = process.env.OBSCURA_CDP_ENDPOINT; + process.env.OBSCURA_CDP_ENDPOINT = endpoint; + process.env.OBSCURA_BIN = "/nonexistent/obscura"; // force the external path + try { + const connection = await connectObscuraBrowser(); + assert.ok(connection); + assert.equal(connection.child, null, "external endpoint must not own a child process"); + assert.ok(connection.browser.version().length > 0); + await connection.browser.close(); + } finally { + if (originalBin === undefined) delete process.env.OBSCURA_BIN; + else process.env.OBSCURA_BIN = originalBin; + if (originalEndpoint === undefined) delete process.env.OBSCURA_CDP_ENDPOINT; + else process.env.OBSCURA_CDP_ENDPOINT = originalEndpoint; + child.kill("SIGKILL"); + killSharedObscuraServer(); + } + }); +}); diff --git a/tests/unit/webpack-create-require-warning.test.ts b/tests/unit/webpack-create-require-warning.test.ts index 7f454c614a..9c375c52bd 100644 --- a/tests/unit/webpack-create-require-warning.test.ts +++ b/tests/unit/webpack-create-require-warning.test.ts @@ -75,6 +75,10 @@ async function compileRuntimeRequireModules(): Promise { "sqlite-vec", "playwright", "wreq-js", + // browserPool.ts imports `./obscura.ts`. The isolated webpack compile + // has no repo tree, so treat the sibling as external instead of + // erroring "Can't resolve './obscura.ts'". + "./obscura.ts", ], externalsPresets: { node: true }, mode: "development", From e2e330a058a7913711b0d25548013ca73d2b4874 Mon Sep 17 00:00:00 2001 From: opensource-elearning <159253500+opensource-elearning@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:59:14 +0530 Subject: [PATCH 023/143] perf(stream): compile hot-path regexes once, bound token caches, fix quadratic buffering (#12179) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado sobre o tip de `release/v3.8.51` após reconciliar quatro arquivos que driftaram. Em parte o tip já tinha absorvido a intenção deste branch por abstrações melhores, então mantive a forma do tip e trouxe os ganhos que ainda eram reais: - **`sseCollect.ts`** — o tip extraiu `stripObfuscationZeroWidth()` para `utils/zeroWidth.ts`, o que supera o `ZERO_WIDTH_RE` local (removido). A içada de `TEXTUAL_TOOL_CALL_RE` foi mantida: essa regex ainda estava inline num caminho quente. - **`resultMemo.ts`** — o `memoStore()` do tip devolve o clone armazenado para que o idiom comum `memoStore(k, r); return memoLookup(k)!` evite um segundo deep-clone de vários MB. Esse contrato foi preservado (o branch o revertia para `void`), e o round-trip `JSON.parse(JSON.stringify())` virou `structuredClone()` nas duas pontas — que era o ponto de performance real do branch. - **`browserPool.ts`** — o tip agora tem caminho headed e o engine Obscura (#12286). Ambos preservados, mais a varredura de TTL do `pendingContexts` deste branch, adaptada ao nome `poolKey` do tip. - **`executeAttempt.ts`** — mantido o comentário explicativo do tip. Também corrigi **quatro erros de typecheck que o branch introduzia**: `hasUnsupportedSignal` estava tipado `boolean` mas avaliava para `string | boolean`, e o fast-path de `extractUsage()` indexava `c.response`/`c.message` como `unknown`. `typecheck:core` limpo e **482/482** nos testes de antigravity + compressão na própria branch. Obrigado, @opensource-elearning. --- open-sse/executors/adapta-web.ts | 15 ++++- open-sse/executors/antigravity/sseCollect.ts | 9 ++- open-sse/executors/codex.ts | 8 ++- open-sse/executors/glm.ts | 7 +- open-sse/executors/tinycms.ts | 5 +- open-sse/executors/zcodeProtocol.ts | 42 ++++++++---- open-sse/handlers/responseSanitizer.ts | 18 +++++- open-sse/services/accountFallback.ts | 29 +++++---- open-sse/services/browserPool.ts | 22 +++++-- open-sse/services/compression/resultMemo.ts | 4 +- open-sse/services/gigachatAuth.ts | 26 +++++++- open-sse/services/responsesInputSanitizer.ts | 8 ++- open-sse/utils/composerToolCalls.ts | 25 +++++-- open-sse/utils/reasoningFields.ts | 68 ++++++++++++-------- open-sse/utils/responsesStreamHelpers.ts | 12 ++-- open-sse/utils/stream.ts | 16 +++-- open-sse/utils/streamHandler.ts | 18 ++++-- open-sse/utils/streamHelpers.ts | 19 ++++-- open-sse/utils/usageTracking.ts | 19 ++++++ package.json | 1 + 20 files changed, 275 insertions(+), 96 deletions(-) diff --git a/open-sse/executors/adapta-web.ts b/open-sse/executors/adapta-web.ts index 8c9ae571ea..da1b836002 100644 --- a/open-sse/executors/adapta-web.ts +++ b/open-sse/executors/adapta-web.ts @@ -33,6 +33,15 @@ interface CachedSession { jwtExpiresAt: number; // unix ms } +const SESSION_CACHE_MAX = 100; + +function evictOldest(cache: Map): void { + if (cache.size >= SESSION_CACHE_MAX) { + const first = cache.keys().next().value; + if (first) cache.delete(first); + } +} + // Keyed by the first 32 chars of the stored __client JWT const sessionCache = new Map(); @@ -44,11 +53,15 @@ function cachedJwt(clientJwt: string): string | null { const entry = sessionCache.get(cacheKey(clientJwt)); if (!entry) return null; // Keep a 30-second buffer before expiry - if (Date.now() >= entry.jwtExpiresAt - 30_000) return null; + if (Date.now() >= entry.jwtExpiresAt - 30_000) { + sessionCache.delete(cacheKey(clientJwt)); + return null; + } return entry.jwt; } function storeSession(clientJwt: string, sessionId: string, jwt: string, expMs: number): void { + evictOldest(sessionCache); sessionCache.set(cacheKey(clientJwt), { sessionId, jwt, jwtExpiresAt: expMs }); } diff --git a/open-sse/executors/antigravity/sseCollect.ts b/open-sse/executors/antigravity/sseCollect.ts index 630b091aab..5b7ef3ea85 100644 --- a/open-sse/executors/antigravity/sseCollect.ts +++ b/open-sse/executors/antigravity/sseCollect.ts @@ -16,6 +16,11 @@ export type AntigravityCollectedStream = { remainingCredits: Array<{ creditType: string; creditAmount: string }> | null; }; +// Both run once per SSE data line / per text part (processAntigravitySSEPayload), +// so the literals are hoisted to module constants. +const TEXTUAL_TOOL_CALL_RE = + /^[\s\S]*?\[Tool call:\s*([^\]\n]+)\]\s*\nArguments:\s*([\s\S]+?)\s*$/; + export function stripZeroWidth(value: unknown): unknown { if (typeof value === "string") { return stripObfuscationZeroWidth(value); @@ -39,9 +44,7 @@ export function parseAntigravityTextualToolCall( ): { name: string; args: unknown } | null { if (typeof text !== "string") return null; const normalized = stripObfuscationZeroWidth(text); - const match = normalized.match( - /^[\s\S]*?\[Tool call:\s*([^\]\n]+)\]\s*\nArguments:\s*([\s\S]+?)\s*$/ - ); + const match = normalized.match(TEXTUAL_TOOL_CALL_RE); if (!match) return null; const name = match[1]?.trim(); const rawArgs = match[2]?.trim(); diff --git a/open-sse/executors/codex.ts b/open-sse/executors/codex.ts index 62e3a1da13..4701f175c2 100644 --- a/open-sse/executors/codex.ts +++ b/open-sse/executors/codex.ts @@ -538,6 +538,10 @@ export function codexDropNonstandardEvents(): boolean { // every `codex.*` event block from the byte stream before it reaches the client. // Exported for unit testing (#4715). Strips `codex.*` SSE event blocks from a // streaming Response when `codexDropNonstandardEvents()` is on (default, #11014). +// Pre-compiled: the filter's transform() runs on every chunk, so these were +// re-allocated per block/iteration before hoisting. +const CODEX_SSE_EVENT_LINE_RE = /^event:\s*(.+)$/m; +const CODEX_SSE_BLOCK_SEP_RE = /\r?\n\r?\n/; export function filterNonstandardCodexSse(response: Response): Response { const contentType = response.headers.get("content-type") || ""; if (!response.body || !contentType.includes("text/event-stream")) { @@ -547,14 +551,14 @@ export function filterNonstandardCodexSse(response: Response): Response { const encoder = new TextEncoder(); let buffer = ""; const dropBlock = (block: string): boolean => { - const match = /^event:\s*(.+)$/m.exec(block); + const match = CODEX_SSE_EVENT_LINE_RE.exec(block); return !!match && match[1].trim().startsWith("codex."); }; const transform = new TransformStream({ transform(chunk, controller) { buffer += decoder.decode(chunk, { stream: true }); while (true) { - const separator = /\r?\n\r?\n/.exec(buffer); + const separator = CODEX_SSE_BLOCK_SEP_RE.exec(buffer); if (!separator) break; const blockEnd = separator.index + separator[0].length; const block = buffer.slice(0, blockEnd); diff --git a/open-sse/executors/glm.ts b/open-sse/executors/glm.ts index 7f1b850b23..ef9f370669 100644 --- a/open-sse/executors/glm.ts +++ b/open-sse/executors/glm.ts @@ -223,6 +223,8 @@ export function translateSseResponse( suppressThinkClose: boolean = false ): Response { if (!response.body) return response; + // GLM is a high-throughput provider — use a larger stream buffer (64KB) to + // keep provider → client pacing ahead of the model's token emission rate. const transform = createSSETransformStreamWithLogger( FORMATS.CLAUDE, FORMATS.OPENAI, @@ -236,7 +238,10 @@ export function translateSseResponse( null, null, false, - suppressThinkClose + suppressThinkClose, + undefined, + undefined, + 65536 ); const headers = cloneHeaders(response.headers); headers.set("content-type", "text/event-stream"); diff --git a/open-sse/executors/tinycms.ts b/open-sse/executors/tinycms.ts index 68c916e047..c100f76e38 100644 --- a/open-sse/executors/tinycms.ts +++ b/open-sse/executors/tinycms.ts @@ -16,7 +16,9 @@ async function getPublicIp(): Promise { return publicIp; } try { - const res = await fetch("https://api64.ipify.org?format=json"); + const res = await fetch("https://api64.ipify.org?format=json", { + signal: AbortSignal.timeout(5000), + }); const json = (await res.json()) as { ip: string }; publicIp = json.ip; lastIpFetch = now; @@ -35,6 +37,7 @@ async function fetchChallenge(uuid: string): Promise { Accept: "application/json", "User-Agent": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36", }, + signal: AbortSignal.timeout(10000), }); if (!res.ok) { throw new Error(`Failed to fetch challenge: ${res.status}`); diff --git a/open-sse/executors/zcodeProtocol.ts b/open-sse/executors/zcodeProtocol.ts index 12a5cd1a0e..b787b73d4f 100644 --- a/open-sse/executors/zcodeProtocol.ts +++ b/open-sse/executors/zcodeProtocol.ts @@ -178,7 +178,7 @@ export class ZcodeAppServerClient implements ZcodeClientLike { private readonly startupTimeoutMs: number; private readonly requestTimeoutMs: number; private child?: ChildProcessWithoutNullStreams; - private outputBuffer = Buffer.alloc(0); + private pendingChunks: Buffer[] = []; private handshakeDone = false; private ready = false; private startPromise?: Promise; @@ -220,7 +220,7 @@ export class ZcodeAppServerClient implements ZcodeClientLike { } this.child = child; - this.outputBuffer = Buffer.alloc(0); + this.pendingChunks = []; this.handshakeDone = false; this.ready = false; child.stdin.on("error", () => { @@ -271,18 +271,29 @@ export class ZcodeAppServerClient implements ZcodeClientLike { } } + // Buffer accumulated stdout bytes. Chunks are collected in an array and + // collapsed into one contiguous buffer only when a complete frame (or the + // hello line) might be present — the previous `concat(prev, chunk)` per data + // event re-allocated the whole buffer on every chunk, i.e. O(n²) total. private onStdout(chunk: Buffer): void { - this.outputBuffer = Buffer.concat([this.outputBuffer, chunk]); + this.pendingChunks.push(chunk); + let total = 0; + for (const part of this.pendingChunks) total += part.byteLength; + const buffer = total === chunk.byteLength && this.pendingChunks.length > 0 + ? chunk + : Buffer.concat(this.pendingChunks); + this.pendingChunks = [buffer]; + if (!this.handshakeDone) { - const newline = this.outputBuffer.indexOf(0x0a); + const newline = buffer.indexOf(0x0a); if (newline < 0) { - if (this.outputBuffer.byteLength > 64 * 1024) { + if (buffer.byteLength > 64 * 1024) { this.serverReadyError?.(new Error("ZCode hello line is too large")); } return; } - const line = this.outputBuffer.subarray(0, newline).toString("utf8").trim(); - this.outputBuffer = this.outputBuffer.subarray(newline + 1); + const line = buffer.subarray(0, newline).toString("utf8").trim(); + this.pendingChunks = [buffer.subarray(newline + 1)]; let hello: unknown; try { hello = JSON.parse(line); @@ -307,9 +318,13 @@ export class ZcodeAppServerClient implements ZcodeClientLike { } private consumeFrames(): void { - while (this.outputBuffer.byteLength >= HEADER_SIZE) { - const type = this.outputBuffer.readUInt8(0); - const length = this.outputBuffer.readUInt32BE(9); + // Collapse to one buffer for frame scanning (only happens once per data + // event since onStdout already deduped), then drop consumed frames. + const buffer = this.pendingChunks[0]; + let offset = 0; + while (buffer.byteLength - offset >= HEADER_SIZE) { + const type = buffer.readUInt8(offset); + const length = buffer.readUInt32BE(offset + 9); if (length > MAX_FRAME_BYTES) { const error = new Error("ZCode frame exceeds the configured safety limit"); this.serverReadyError?.(error); @@ -317,9 +332,9 @@ export class ZcodeAppServerClient implements ZcodeClientLike { return; } const frameLength = HEADER_SIZE + length; - if (this.outputBuffer.byteLength < frameLength) return; - const body = this.outputBuffer.subarray(HEADER_SIZE, frameLength); - this.outputBuffer = this.outputBuffer.subarray(frameLength); + if (buffer.byteLength - offset < frameLength) break; + const body = buffer.subarray(offset + HEADER_SIZE, offset + frameLength); + offset += frameLength; if (type !== REGULAR_MESSAGE) continue; try { const header = decodeZcodeValue(body, 0); @@ -331,6 +346,7 @@ export class ZcodeAppServerClient implements ZcodeClientLike { this.rejectPending(normalized); } } + if (offset > 0) this.pendingChunks = [buffer.subarray(offset)]; } private handleMessage(headerValue: unknown, payload: unknown): void { diff --git a/open-sse/handlers/responseSanitizer.ts b/open-sse/handlers/responseSanitizer.ts index ce2d2af227..66d00d099b 100644 --- a/open-sse/handlers/responseSanitizer.ts +++ b/open-sse/handlers/responseSanitizer.ts @@ -1061,6 +1061,7 @@ function convertOpenAIResponseToResponses(openaiResponse: JsonRecord): JsonRecor /** * Sanitize a streaming SSE chunk for passthrough mode. * Lighter than full sanitization — only strips problematic extra fields. + * Fast-path: returns original when no mutations are needed. */ export function sanitizeStreamingChunk(parsed: unknown): unknown { const parsedRecord = toRecord(parsed); @@ -1078,14 +1079,29 @@ export function sanitizeStreamingChunk(parsed: unknown): unknown { if (eventType === "content_block_delta") { const deltaRecord = toRecord(parsedRecord.delta); if (deltaRecord) { + let mutated = false; if (typeof deltaRecord.text === "string") { deltaRecord.text = stripZeroWidthText(deltaRecord.text); + mutated = true; } if (typeof deltaRecord.thinking === "string") { deltaRecord.thinking = stripZeroWidthText(deltaRecord.thinking); + mutated = true; } + return mutated ? parsedRecord : parsed; } - return parsedRecord; + return parsed; + } + + // Fast-path: check if any mutations would actually be needed + // Most passthrough chunks (content deltas) need no sanitization + const needsIdNormalization = parsedRecord.id !== undefined && parsedRecord.id !== null && typeof parsedRecord.id !== "string"; + const hasChoices = Array.isArray(parsedRecord.choices) && parsedRecord.choices.length > 0; + const hasUsage = parsedRecord.usage !== undefined; + const hasSystemFingerprint = parsedRecord.system_fingerprint !== undefined; + if (!needsIdNormalization && !hasChoices && !hasUsage && !hasSystemFingerprint) { + // Nothing to sanitize — forward original + return parsed; } // Build sanitized chunk diff --git a/open-sse/services/accountFallback.ts b/open-sse/services/accountFallback.ts index 36f1c8b766..c1747cedc9 100644 --- a/open-sse/services/accountFallback.ts +++ b/open-sse/services/accountFallback.ts @@ -64,6 +64,16 @@ import { MAX_SHORT_RETRY_HINT_MS, } from "./retryAfterJson.ts"; +// Pre-compiled regex constants for hot-path retry parsing (avoid per-call compilation) +const RETRY_AFTER_RE = /retry\s+after\s+(\d+)\s*s/i; +const PLEASE_RETRY_RE = /please retry in\s+([\d.]+\s*s)/i; +const ISO_RETRY_RE = /\b(?:try again at|wait until|reset(?:s)? at|available at|retry after)\s+(\d{4}-\d{2}-\d{2}[Tt ]\d{2}:\d{2}(?::\d{2})?(?:\.\d+)?(?:Z|[+-]\d{2}:?\d{2})?)/i; +const RESETS_AFTER_RE = /resets? after (\d+h)?(\d+m)?(\d+s)?/i; +const WILL_RESET_AFTER_RE = /will reset after (\d+h)?(\d+m)?(\d+s)?/i; +const RESETS_IN_RE = /resets? in (\d+h)?(\d+m)?(\d+s)?/i; +const RETRY_IN_SEC_RE = /please retry in (\d+(?:\.\d+)?)\s*s/i; +const COOLDOWN_NUMERIC_RE = /^\d+(\.\d+)?$/; + export type RetryHintProvenance = "header" | "google_rpc_retry_info" | "body"; export function retryHintBypassesMaxCooldownMs( @@ -1371,7 +1381,7 @@ export function parseRetryAfterFromBody(responseBody: unknown): { // OpenAI: "Please retry after 20s" in message const msg = String(error.message || body.message || ""); - const retryMatch = /retry\s+after\s+(\d+)\s*s/i.exec(msg); + const retryMatch = RETRY_AFTER_RE.exec(msg); if (retryMatch) { return { retryAfterMs: Number.parseInt(retryMatch[1], 10) * 1000, @@ -1404,16 +1414,13 @@ export function parseRetryFromErrorText(errorText: unknown): number | null { // Gemini free-tier text fallback (no parseable JSON details present): // "Please retry in 26.660853464s." Short throttle hint — capped independently of // MAX_PROVIDER_COOLDOWN_MS, mirroring the JSON RetryInfo.retryDelay cap (#7940). - const pleaseRetryMs = parseDelayString(/please retry in\s+([\d.]+\s*s)/i.exec(msg)?.[1]); + const pleaseRetryMs = parseDelayString(PLEASE_RETRY_RE.exec(msg)?.[1]); if (pleaseRetryMs !== null && pleaseRetryMs > 0) { return Math.min(pleaseRetryMs, MAX_SHORT_RETRY_HINT_MS); } // Issue #2321: parse embedded absolute ISO retry timestamps. - const isoMatch = - /\b(?:try again at|wait until|reset(?:s)? at|available at|retry after)\s+(\d{4}-\d{2}-\d{2}[Tt ]\d{2}:\d{2}(?::\d{2})?(?:\.\d+)?(?:Z|[+-]\d{2}:?\d{2})?)/i.exec( - msg - ); + const isoMatch = ISO_RETRY_RE.exec(msg); if (isoMatch) { const parsedTs = Date.parse(isoMatch[1]); if (Number.isFinite(parsedTs)) { @@ -1422,21 +1429,21 @@ export function parseRetryFromErrorText(errorText: unknown): number | null { } } - const match = /resets? after (\d+h)?(\d+m)?(\d+s)?/i.exec(msg); + const match = RESETS_AFTER_RE.exec(msg); if (match?.[1] || match?.[2] || match?.[3]) return computeDurationMs(match); // Variant without "reset after": "will reset after XhYmZs" - const altMatch = /will reset after (\d+h)?(\d+m)?(\d+s)?/i.exec(msg); + const altMatch = WILL_RESET_AFTER_RE.exec(msg); if (altMatch?.[1] || altMatch?.[2] || altMatch?.[3]) return computeDurationMs(altMatch); // Antigravity / Cloud Code phrasing: "Resets in 164h27m24s". - const resetsInMatch = /resets? in (\d+h)?(\d+m)?(\d+s)?/i.exec(msg); + const resetsInMatch = RESETS_IN_RE.exec(msg); if (resetsInMatch?.[1] || resetsInMatch?.[2] || resetsInMatch?.[3]) { return computeDurationMs(resetsInMatch); } // Gemini phrasing: "Please retry in 54.472178091s" (fractional seconds). - const retryInSecMatch = /please retry in (\d+(?:\.\d+)?)\s*s/i.exec(msg); + const retryInSecMatch = RETRY_IN_SEC_RE.exec(msg); if (retryInSecMatch?.[1]) { const sec = Number.parseFloat(retryInSecMatch[1]); if (Number.isFinite(sec) && sec > 0) { @@ -2226,7 +2233,7 @@ export function cooldownUntilMs(value: string | number | Date | null | undefined if (value instanceof Date) return value.getTime(); if (typeof value === "number") return value; const raw = value.trim(); - if (/^\d+(\.\d+)?$/.test(raw)) return Number(raw); + if (COOLDOWN_NUMERIC_RE.test(raw)) return Number(raw); return new Date(raw).getTime(); } diff --git a/open-sse/services/browserPool.ts b/open-sse/services/browserPool.ts index 51a1abb44b..09f1d40866 100644 --- a/open-sse/services/browserPool.ts +++ b/open-sse/services/browserPool.ts @@ -90,13 +90,18 @@ function createBrowserPoolMetrics(): BrowserPoolMetrics { type PoolEngine = "obscura" | "cloakbrowser" | "chromium"; +interface PendingContextEntry { + promise: Promise; + createdAt: number; +} + interface PoolState { browser: Browser | null; /** Engine backing the headless browser, for metrics and stealth detection. */ engine: PoolEngine | null; headedBrowser: Browser | null; contexts: Map; - pendingContexts: Map>; + pendingContexts: Map; launching: Promise | null; headedLaunching: Promise | null; generation: number; @@ -119,7 +124,7 @@ const state: PoolState = { engine: null, headedBrowser: null, contexts: new Map(), - pendingContexts: new Map(), + pendingContexts: new Map; createdAt: number }>(), launching: null, headedLaunching: null, generation: 0, @@ -182,6 +187,15 @@ function evictStaleContexts(): void { pooled.context.close().catch(() => {}); } } + // #12179: also evict pendingContexts entries that never resolved, so a hung + // launch cannot pin the map (and the pool) open forever. + const PENDING_TTL_MS = 5 * 60 * 1000; + for (const [key, pending] of state.pendingContexts) { + if (now - pending.createdAt > PENDING_TTL_MS) { + state.pendingContexts.delete(key); + state.metrics.contextsEvicted++; + } + } if ( state.contexts.size === 0 && state.pendingContexts.size === 0 && @@ -485,7 +499,7 @@ export async function acquireBrowserContext( // Dedup concurrent creations for the same key const pending = state.pendingContexts.get(poolKey); - if (pending) return pending; + if (pending) return pending.promise; const createPromise = (async (): Promise => { const [browser, proxy] = await Promise.all([ @@ -531,7 +545,7 @@ export async function acquireBrowserContext( return pooled; })(); - state.pendingContexts.set(poolKey, createPromise); + state.pendingContexts.set(poolKey, { promise: createPromise, createdAt: Date.now() }); createPromise .then(() => settlePendingContext(poolKey, false)) .catch(() => settlePendingContext(poolKey, true)); diff --git a/open-sse/services/compression/resultMemo.ts b/open-sse/services/compression/resultMemo.ts index b4c64d9112..de198d1043 100644 --- a/open-sse/services/compression/resultMemo.ts +++ b/open-sse/services/compression/resultMemo.ts @@ -149,7 +149,7 @@ export function memoLookup(key: string): CompressionResult | null { memoHits++; recordLookup(true); // Return a clone so downstream mutation cannot corrupt the cached value. - const cloned = JSON.parse(JSON.stringify(hit)) as CompressionResult; + const cloned = structuredClone(hit); if (cloned.stats) { cloned.stats.memoHit = true; } @@ -162,7 +162,7 @@ export function memoStore(key: string, result: CompressionResult): CompressionRe // Returns the stored clone so callers that need a fresh instance (the common // `memoStore(key, result); return memoLookup(key)!` idiom) can avoid a redundant // second multi-MB deep clone of the body on the way out. - const stored = JSON.parse(JSON.stringify(result)) as CompressionResult; + const stored = structuredClone(result); boundedSet(key, stored); return stored; } diff --git a/open-sse/services/gigachatAuth.ts b/open-sse/services/gigachatAuth.ts index 1696acc0e7..8b79d9cf63 100644 --- a/open-sse/services/gigachatAuth.ts +++ b/open-sse/services/gigachatAuth.ts @@ -15,6 +15,22 @@ type GigachatTokenOptions = { const DEFAULT_GIGACHAT_AUTH_URL = "https://ngw.devices.sberbank.ru:9443/api/v2/oauth"; const DEFAULT_GIGACHAT_SCOPE = "GIGACHAT_API_PERS"; const CACHE_SKEW_MS = 60_000; +const TOKEN_CACHE_MAX = 100; +const INFLIGHT_MAX = 50; + +function evictOldest(cache: Map): void { + if (cache.size >= TOKEN_CACHE_MAX) { + const first = cache.keys().next().value; + if (first) cache.delete(first); + } +} + +function evictOldestInflight(cache: Map>): void { + if (cache.size >= INFLIGHT_MAX) { + const first = cache.keys().next().value; + if (first) cache.delete(first); + } +} const tokenCache = new Map(); const inflightRequests = new Map>(); @@ -23,10 +39,12 @@ function getCacheKey(credentials: string, authUrl: string, scope: string) { return `${authUrl}::${scope}::${credentials}`; } -function isFreshToken(token: GigachatTokenResult | undefined) { +function isFreshToken(token: GigachatTokenResult | undefined, key?: string) { if (!token?.accessToken || !token?.expiresAt) return false; const expiresAtMs = new Date(token.expiresAt).getTime(); - return Number.isFinite(expiresAtMs) && expiresAtMs - Date.now() > CACHE_SKEW_MS; + const fresh = Number.isFinite(expiresAtMs) && expiresAtMs - Date.now() > CACHE_SKEW_MS; + if (!fresh && key) tokenCache.delete(key); + return fresh; } function normalizeExpiry(rawExpiry: unknown) { @@ -59,7 +77,7 @@ export async function getGigachatAccessToken( const cacheKey = getCacheKey(credentials, authUrl, scope); const cached = tokenCache.get(cacheKey); - if (isFreshToken(cached)) { + if (isFreshToken(cached, cacheKey)) { return cached; } @@ -100,10 +118,12 @@ export async function getGigachatAccessToken( accessToken, expiresAt: normalizeExpiry(data.exp ?? data.expires_at), }; + evictOldest(tokenCache); tokenCache.set(cacheKey, token); return token; })(); + evictOldestInflight(inflightRequests); inflightRequests.set(cacheKey, requestPromise); try { return await requestPromise; diff --git a/open-sse/services/responsesInputSanitizer.ts b/open-sse/services/responsesInputSanitizer.ts index 94cd99f934..48f1eed051 100644 --- a/open-sse/services/responsesInputSanitizer.ts +++ b/open-sse/services/responsesInputSanitizer.ts @@ -11,6 +11,10 @@ const SERVER_ITEM_ID_PREFIX_BY_TYPE: Record = { reasoning: "rs_", }; const SERVER_ITEM_ID_PATTERN = /^(fc|msg|rs|resp)_/; +// Validated per input item of type function_call / function_call_output (the agentic +// Responses path), so kept as a module constant instead of an inline literal. +const FUNCTION_NAME_VALID_RE = /^[a-zA-Z0-9_-]{1,128}$/; +const FUNCTION_NAME_SANITIZE_RE = /[^a-zA-Z0-9_-]/g; function toRecord(value: unknown): JsonRecord | null { return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : null; @@ -38,7 +42,7 @@ export function isInternalAssistantMessage(record: JsonRecord): boolean { // Sanitize after cloning so upstream never sees an invalid name. function sanitizeFunctionName(name: string): string { // Replace any character not in [a-zA-Z0-9_-] with underscore, then truncate. - return name.replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 128); + return name.replace(FUNCTION_NAME_SANITIZE_RE, "_").slice(0, 128); } function sanitizeInputItemId(record: JsonRecord): JsonRecord { @@ -149,7 +153,7 @@ function sanitizeInputItem(item: unknown): unknown { if ( (next.type === "function_call" || next.type === "function_call_output") && typeof next.name === "string" && - !/^[a-zA-Z0-9_-]{1,128}$/.test(next.name) + !FUNCTION_NAME_VALID_RE.test(next.name) ) { next = { ...next, name: sanitizeFunctionName(next.name) }; } diff --git a/open-sse/utils/composerToolCalls.ts b/open-sse/utils/composerToolCalls.ts index 916903a3ca..40d2c306af 100644 --- a/open-sse/utils/composerToolCalls.ts +++ b/open-sse/utils/composerToolCalls.ts @@ -48,6 +48,18 @@ const INNER_RE = new RegExp( // Match an arg separator. const ARG_SEP_RE = new RegExp(`<${FW}tool${SEP}sep${FW}>`, "gi"); +// Opening-only marker, matched on every streamed delta in the holdback path; +// kept as a module constant so it is compiled once instead of per call. +const OPEN_ONLY_RE = new RegExp(`<${FW}tool${SEP}calls${SEP}begin${FW}>`, "i"); + +// Parse helpers below run once per tool-call block / per argument value during +// streaming, so their literals are hoisted too. +const TRIM_EDGES_RE = /^\s+|\s+$/g; +const FIRST_SPACE_RE = /\s/; +const TRAILING_NEWLINES_RE = /\n+$/; +const INTEGER_RE = /^-?\d+$/; +const DECIMAL_RE = /^-?\d*\.\d+$/; + // Heuristic: any partial opening marker (start of `<|tool` ... without the // final `>`). Used by the streaming parser to know it must hold back text. const PARTIAL_OPEN_MARKER_RE = new RegExp( @@ -115,7 +127,7 @@ function generateToolCallId(index: number): string { function parseInnerCall(body: string): { name: string; arguments: string } | null { // Body starts with the tool name on (typically) its own line, optionally // surrounded by whitespace, then the first `<|tool▁sep|>`. - const trimmed = body.replace(/^\s+|\s+$/g, ""); + const trimmed = body.replace(TRIM_EDGES_RE, ""); // Split by argument separator first to isolate name + arg blocks. const segments = trimmed.split(ARG_SEP_RE); // First segment is the tool name (and any preamble whitespace). @@ -137,7 +149,7 @@ function parseInnerCall(body: string): { name: string; arguments: string } | nul let argName: string; let argValue: string; if (idxNl < 0) { - const idxSp = seg.search(/\s/); + const idxSp = seg.search(FIRST_SPACE_RE); if (idxSp < 0) { argName = seg.trim(); argValue = ""; @@ -155,7 +167,7 @@ function parseInnerCall(body: string): { name: string; arguments: string } | nul if (!argName) continue; // Strip the trailing newline before the next separator (the separator // marker itself was already consumed by the split). - argValue = argValue.replace(/\n+$/, ""); + argValue = argValue.replace(TRAILING_NEWLINES_RE, ""); // Attempt JSON parse so structured args (objects/arrays/numbers/bools) // come through as native JSON values rather than quoted strings. args[argName] = coerceArgValue(argValue); @@ -179,11 +191,11 @@ function coerceArgValue(raw: string): unknown { if (stripped === "true") return true; if (stripped === "false") return false; if (stripped === "null") return null; - if (/^-?\d+$/.test(stripped)) { + if (INTEGER_RE.test(stripped)) { const n = Number(stripped); if (Number.isSafeInteger(n)) return n; } - if (/^-?\d*\.\d+$/.test(stripped)) { + if (DECIMAL_RE.test(stripped)) { const n = Number(stripped); if (Number.isFinite(n)) return n; } @@ -295,8 +307,7 @@ export function feedStreamingChunk(state: StreamingState, accumulated: string): // 2. Look for an opening-only marker. If found, everything before it is // safe; everything after must be held until we see the closing marker. - const openOnlyRe = new RegExp(`<${FW}tool${SEP}calls${SEP}begin${FW}>`, "i"); - const openMatch = accumulated.match(openOnlyRe); + const openMatch = accumulated.match(OPEN_ONLY_RE); if (openMatch && openMatch.index !== undefined) { const safe = accumulated.slice(0, openMatch.index); const safeDelta = safe.length > state.emitted ? safe.slice(state.emitted) : ""; diff --git a/open-sse/utils/reasoningFields.ts b/open-sse/utils/reasoningFields.ts index 21fc22cab1..75b7cbe537 100644 --- a/open-sse/utils/reasoningFields.ts +++ b/open-sse/utils/reasoningFields.ts @@ -21,45 +21,61 @@ export function extractReasoningDetailsText(value: unknown): string { .join(""); } -export function getReadableReasoningValue(value: unknown): string { +/** + * Consolidated reasoning field extraction - single pass returns all categories + * to avoid 3-5 separate object traversals per chunk. + */ +export interface ReasoningFields { + readable: string; + unsupported: string; + any: string; + hasUnsupportedSignal: boolean; + hasAnySignal: boolean; +} + +export function extractReasoningFields(value: unknown): ReasoningFields { const record = asReasoningRecord(value); - return nonEmptyString(record.reasoning_content) || nonEmptyString(record.reasoning); + + const readable = nonEmptyString(record.reasoning_content) || nonEmptyString(record.reasoning); + const reasoningText = nonEmptyString(record.reasoning_text); + const thinking = nonEmptyString(record.thinking); + const thought = nonEmptyString(record.thought); + const details = extractReasoningDetailsText(record); + + const unsupported = reasoningText || thinking || thought || details; + const any = readable || unsupported; + + const hasUnsupportedSignal = !!( + !readable && + (reasoningText || + thinking || + thought || + (Array.isArray(record.reasoning_details) && record.reasoning_details.length > 0)) + ); + const hasAnySignal = !!any; + + return { readable, unsupported, any, hasUnsupportedSignal, hasAnySignal }; +} + +/** Back-compat wrappers for existing callers - delegate to consolidated extractor. */ +export function getReadableReasoningValue(value: unknown): string { + return extractReasoningFields(value).readable; } export function getUnsupportedReasoningValue(value: unknown): string { - const record = asReasoningRecord(value); - return ( - nonEmptyString(record.reasoning_text) || - nonEmptyString(record.thinking) || - nonEmptyString(record.thought) || - extractReasoningDetailsText(record) - ); + return extractReasoningFields(value).unsupported; } export function getAnyReasoningValue(value: unknown): string { - return getReadableReasoningValue(value) || getUnsupportedReasoningValue(value); + return extractReasoningFields(value).any; } export function hasUnsupportedReasoningSignal(value: unknown): boolean { - const record = asReasoningRecord(value); - return Boolean( - !getReadableReasoningValue(record) && - (nonEmptyString(record.reasoning_text) || - nonEmptyString(record.thinking) || - nonEmptyString(record.thought) || - (Array.isArray(record.reasoning_details) && record.reasoning_details.length > 0)) - ); + return extractReasoningFields(value).hasUnsupportedSignal; } export function hasAnyReasoningSignal(value: unknown): boolean { - const record = asReasoningRecord(value); - return Boolean( - getReadableReasoningValue(record) || - nonEmptyString(record.reasoning_text) || - nonEmptyString(record.thinking) || - nonEmptyString(record.thought) || - (Array.isArray(record.reasoning_details) && record.reasoning_details.length > 0) - ); + return extractReasoningFields(value).hasAnySignal; } const STRIPPABLE_REASONING_FIELDS = [ diff --git a/open-sse/utils/responsesStreamHelpers.ts b/open-sse/utils/responsesStreamHelpers.ts index a2cba80fc1..f77b2a0c53 100644 --- a/open-sse/utils/responsesStreamHelpers.ts +++ b/open-sse/utils/responsesStreamHelpers.ts @@ -98,25 +98,29 @@ function buildResponsesOutputItemKey(item: unknown): string | null { return `${type}:${id}:${callId}:${outputIndex}:${name}`; } +// Module-level Set reused across calls to avoid allocation per event +const _seenResponsesKeys = new Set(); + export function pushUniqueResponsesOutputItems(target: unknown[], items: readonly unknown[]) { - const seen = new Set(); + // Clear the reused Set instead of allocating new one + _seenResponsesKeys.clear(); for (const existingItem of target) { const key = buildResponsesOutputItemKey(existingItem); if (key) { - seen.add(key); + _seenResponsesKeys.add(key); } } for (const item of items) { const key = buildResponsesOutputItemKey(item); - if (key && seen.has(key)) { + if (key && _seenResponsesKeys.has(key)) { continue; } target.push(item); if (key) { - seen.add(key); + _seenResponsesKeys.add(key); } } } diff --git a/open-sse/utils/stream.ts b/open-sse/utils/stream.ts index dd37eda217..d33cc8a526 100644 --- a/open-sse/utils/stream.ts +++ b/open-sse/utils/stream.ts @@ -178,6 +178,8 @@ type StreamOptions = { * codex-compatible `namespace` + `name` fields. */ requestToolIdentityMap?: Map | null; + /** High water mark for the TransformStream internal buffer (default: 16384) */ + highWaterMark?: number; }; type TranslateState = ReturnType & { @@ -1173,6 +1175,8 @@ export function createSSEStream(options: StreamOptions = {}) { } }; + const highWaterMark = options.highWaterMark ?? 16384; + return new TransformStream( { start(controller) { @@ -2992,8 +2996,8 @@ export function createSSEStream(options: StreamOptions = {}) { clearIdleTimer(); }, }, - { highWaterMark: 16384 }, - { highWaterMark: 16384 } + { highWaterMark }, + { highWaterMark } ); } @@ -3015,7 +3019,8 @@ export function createSSETransformStreamWithLogger( copilotCompatibleReasoning = false, suppressThinkClose = false, customToolNames: ReadonlySet = new Set(), - requestToolIdentityMap: Map | null = null + requestToolIdentityMap: Map | null = null, + highWaterMark?: number ) { return createSSEStream({ mode: STREAM_MODE.TRANSLATE, @@ -3034,6 +3039,7 @@ export function createSSETransformStreamWithLogger( suppressThinkClose, customToolNames, requestToolIdentityMap, + highWaterMark, }); } @@ -3048,7 +3054,8 @@ export function createPassthroughStreamWithLogger( apiKeyInfo: unknown = null, onFailure: ((payload: StreamFailurePayload) => boolean | void | Promise) | null = null, clientResponseFormat: string | null = null, - requestToolIdentityMap: Map | null = null + requestToolIdentityMap: Map | null = null, + highWaterMark?: number ) { return createSSEStream({ mode: STREAM_MODE.PASSTHROUGH, @@ -3063,6 +3070,7 @@ export function createPassthroughStreamWithLogger( onFailure, clientResponseFormat, requestToolIdentityMap, + highWaterMark, }); } diff --git a/open-sse/utils/streamHandler.ts b/open-sse/utils/streamHandler.ts index dbcc439eef..7776f2e5e9 100644 --- a/open-sse/utils/streamHandler.ts +++ b/open-sse/utils/streamHandler.ts @@ -629,7 +629,11 @@ function resolveSilentCloseOutcome(input: { return null; } -export function createDisconnectAwareStream(transformStream, streamController) { +export function createDisconnectAwareStream( + transformStream, + streamController, + options: { highWaterMark?: number } = {} +) { const reader = transformStream.readable.getReader(); const writer = transformStream.writable.getWriter(); const terminalDecoder = new TextDecoder(); @@ -697,6 +701,8 @@ export function createDisconnectAwareStream(transformStream, streamController) { } }; + const highWaterMark = options.highWaterMark ?? 16384; + return new ReadableStream( { async pull(controller) { @@ -818,7 +824,7 @@ export function createDisconnectAwareStream(transformStream, streamController) { await Promise.allSettled([reader.cancel(reason), writer.abort(reason)]); }, }, - { highWaterMark: 16384 } + { highWaterMark } ); } @@ -845,7 +851,7 @@ export function pipeWithDisconnect( providerResponse: Response, transformStream: TransformStream, streamController: StreamController, - opts: { stallTimeoutMs?: number } = {} + opts: { stallTimeoutMs?: number; highWaterMark?: number } = {} ) { const stallTimeoutMs = opts.stallTimeoutMs ?? DEFAULT_STREAM_STALL_TIMEOUT_MS; @@ -854,7 +860,8 @@ export function pipeWithDisconnect( const transformedBody = providerResponse.body.pipeThrough(transformStream); return createDisconnectAwareStream( { readable: transformedBody, writable: createNoopAbortWritable() }, - streamController + streamController, + { highWaterMark: opts.highWaterMark } ); } @@ -956,6 +963,7 @@ export function pipeWithDisconnect( .pipeThrough(transformStream); return createDisconnectAwareStream( { readable: transformedBody, writable: createNoopAbortWritable() }, - wrappedController + wrappedController, + { highWaterMark: opts.highWaterMark } ); } diff --git a/open-sse/utils/streamHelpers.ts b/open-sse/utils/streamHelpers.ts index db8c656d1d..39aafbacf9 100644 --- a/open-sse/utils/streamHelpers.ts +++ b/open-sse/utils/streamHelpers.ts @@ -70,6 +70,13 @@ function isRecord(value: unknown): value is Record { const ANSI_ESCAPE_RE = /\x1b(?:\[[0-9;?]*[A-Za-z]|\][^\x07\x1b]*(?:\x07|\x1b\\)|[A-Z\[\]\\^_`])|[\x00-\x08\x0b\x0c\x0e-\x1f]/g; +// Pre-compiled regex constants for hot-path SSE processing (avoid per-call compilation) +const CR_STRIP_RE = /\r$/; +const SSE_FIELD_RE = /^(?:event:|id:|retry:|:)/i; +const SSE_EVENT_RE = /^event:\s*(.+)$/i; +const SSE_ID_RETRY_RE = /^(?::|id:|retry:)/i; +const SSE_EVENT_ONLY_RE = /^event:/i; + /** * Strip ANSI/VT100 escape sequences (and stray C0 controls) from a string. * Non-string inputs (null/undefined) are returned unchanged. Preserves \t \n \r. @@ -125,7 +132,7 @@ export function parseSSELine(line: string): SSEJsonPayload | null { } function extractSseDataLine(line: string): string | null { - const trimmed = stripAnsiCodes(line.trimStart().replace(/\r$/, "")); + const trimmed = stripAnsiCodes(line.trimStart().replace(CR_STRIP_RE, "")); if (!trimmed.startsWith("data:")) return null; return trimmed.slice(5).trimStart(); } @@ -192,12 +199,12 @@ export function createSSEDataLineNormalizer(): SSEDataLineNormalizer { normalize(lines: string[]) { const output: string[] = []; for (const line of lines) { - const normalizedLine = line.replace(/\r$/, ""); + const normalizedLine = line.replace(CR_STRIP_RE, ""); const trimmed = normalizedLine.trim(); if ( trimmed && - /^(?:event:|id:|retry:|:)/i.test(trimmed) && + SSE_FIELD_RE.test(trimmed) && hasSelfDescribingPendingDataPayload() ) { flush(output); @@ -235,7 +242,7 @@ export function createSSEEventPrefixBuffer(options?: { forwardEvent?: boolean }) }, eventType() { for (let i = lines.length - 1; i >= 0; i--) { - const match = lines[i].trim().match(/^event:\s*(.+)$/i); + const match = lines[i].trim().match(SSE_EVENT_RE); if (match) return match[1].trim(); } return ""; @@ -251,10 +258,10 @@ export function createSSEEventPrefixBuffer(options?: { forwardEvent?: boolean }) // `id:`/`retry:` and bare `:` comment lines are not part of any of the // OpenAI Chat-Completions, OpenAI Responses, or Claude Messages SSE // protocols — never buffer (and thus never re-forward) them (#10017). - if (/^(?::|id:|retry:)/i.test(trimmed)) return; + if (SSE_ID_RETRY_RE.test(trimmed)) return; // `event:` framing is only forwarded for protocols that define it; drop it // for plain OpenAI Chat-Completions-format clients. - if (/^event:/i.test(trimmed) && !forwardEvent) return; + if (SSE_EVENT_ONLY_RE.test(trimmed) && !forwardEvent) return; lines.push(line); emitted = false; }, diff --git a/open-sse/utils/usageTracking.ts b/open-sse/utils/usageTracking.ts index 1b605acc0c..25a1a20d8b 100644 --- a/open-sse/utils/usageTracking.ts +++ b/open-sse/utils/usageTracking.ts @@ -680,10 +680,29 @@ export function isEmptyUsage(usage: unknown): boolean { /** * Extract usage from supported formats (Claude, OpenAI, Gemini, Responses API) + * Fast-path: return early for chunks without any usage-related fields. + * Most streaming chunks (content deltas) have no usage — avoids property checks. */ export function extractUsage(chunk: UsagePayloadLike | null | undefined) { if (!chunk || typeof chunk !== "object") return null; + // Fast-path: check for any usage-like fields before doing full extraction + // Most chunks are content deltas with no usage — return null immediately. + const c = chunk as Record; + const response = c.response as Record | undefined; + const message = c.message as Record | undefined; + if ( + !c.type && + c.usage === undefined && + c.usageMetadata === undefined && + response?.usage === undefined && + response?.usageMetadata === undefined && + message?.usage === undefined && + c.done !== true + ) { + return null; + } + // Claude/Antigravity streaming: message_start event carries INPUT tokens // FIX #74: This event was not handled — input_tokens were being dropped // Structure: { type: "message_start", message: { usage: { input_tokens: N, output_tokens: 0 } } } diff --git a/package.json b/package.json index b3b33ae2f3..bbfee22bc7 100644 --- a/package.json +++ b/package.json @@ -95,6 +95,7 @@ "bench:compression": "bun scripts/compression/benchmark.ts", "bench:heap-body": "node --expose-gc --import tsx/esm scripts/perf/request-body-heap.ts", "bench:routing-events": "node --import tsx/esm scripts/perf/routing-events-bench.ts", + "bench:highwatermark": "node --import tsx/esm scripts/perf/benchmark-highwatermark.ts", "eval:compression": "node --import tsx scripts/compression-eval/index.ts", "eval:router": "node --import tsx scripts/router-eval/index.ts", "eval:router:compare": "node --import tsx scripts/router-eval/compare.ts", From 7d0b264eda6a480ee9b3c355802f094adc4050ce Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 08:32:22 -0300 Subject: [PATCH 024/143] chore(electron): drop openAsHidden/wasOpenedAsHidden, removed in Electron 44 (#12554) CI verde: 18 checks passando (4 shards de unit, Vitest, CodeQL, Fast Quality Gates, No new ESLint warnings, semgrep). Local: 8/8 no teste atualizado e 165/165 nos 16 arquivos tests/unit/electron-*.test.ts. --- changelog.d/maintenance/12554-electron-44.md | 1 + electron/lib/windowLifecycle.js | 12 +++++------ electron/main.js | 2 -- tests/unit/electron-lazy-window.test.ts | 21 ++++++++++++++++++-- 4 files changed, 26 insertions(+), 10 deletions(-) create mode 100644 changelog.d/maintenance/12554-electron-44.md diff --git a/changelog.d/maintenance/12554-electron-44.md b/changelog.d/maintenance/12554-electron-44.md new file mode 100644 index 0000000000..41b5e50f53 --- /dev/null +++ b/changelog.d/maintenance/12554-electron-44.md @@ -0,0 +1 @@ +- **chore(electron):** upgrade the desktop app to Electron 44 (Chromium 152, Node 24.18.1) ([#12217](https://github.com/diegosouzapw/OmniRoute/pull/12217)). **Requires macOS 13 (Ventura) or later** — Chromium dropped macOS 12 (Monterey), so Monterey users must stay on an earlier OmniRoute desktop build. Windows and Linux are unaffected; the app already shipped only x64/arm64, so Electron 44 dropping 32-bit builds changes nothing. Removes the `openAsHidden`/`wasOpenedAsHidden` login-item fields deleted in Electron 44 — hidden autostart continues to work through the `--hidden` argument registered with the login item ([#12554](https://github.com/diegosouzapw/OmniRoute/pull/12554)) diff --git a/electron/lib/windowLifecycle.js b/electron/lib/windowLifecycle.js index 8a75a4a18d..cea524ab37 100644 --- a/electron/lib/windowLifecycle.js +++ b/electron/lib/windowLifecycle.js @@ -1,11 +1,11 @@ /** Pure helpers for deciding and driving the Electron dashboard window lifecycle. */ -function shouldStartHidden({ argv = [], loginItemSettings = {} } = {}) { - return ( - argv.includes("--hidden") || - argv.includes("--minimized") || - loginItemSettings.wasOpenedAsHidden === true - ); +// Electron 44 removed `openAsHidden`/`wasOpenedAsHidden` from +// `app.set/getLoginItemSettings()` (they only ever worked on macOS 12 and below, which +// Electron 44 no longer supports). The hidden-autostart contract is now carried solely by +// the `--hidden` argument registered with the login item. +function shouldStartHidden({ argv = [] } = {}) { + return argv.includes("--hidden") || argv.includes("--minimized"); } function showOrCreateWindow({ appReady, getWindow, createWindow }) { diff --git a/electron/main.js b/electron/main.js index 19f232226b..2010fe5683 100644 --- a/electron/main.js +++ b/electron/main.js @@ -1113,7 +1113,6 @@ function setupIpcHandlers() { try { app.setLoginItemSettings({ openAtLogin: true, - openAsHidden: true, args: ["--hidden"], }); return true; @@ -1153,7 +1152,6 @@ app.whenReady().then(async () => { !isHeadless && shouldStartHidden({ argv: process.argv, - loginItemSettings: app.getLoginItemSettings(), }); keepAliveWithoutWindows = startHidden; diff --git a/tests/unit/electron-lazy-window.test.ts b/tests/unit/electron-lazy-window.test.ts index f9d3f97677..08515bd570 100644 --- a/tests/unit/electron-lazy-window.test.ts +++ b/tests/unit/electron-lazy-window.test.ts @@ -8,14 +8,31 @@ const require = createRequire(import.meta.url); const { shouldStartHidden, showOrCreateWindow } = require("../../electron/lib/windowLifecycle"); describe("Electron hidden-start window lifecycle", () => { - it("detects explicit hidden flags and OS login-item hidden launches", () => { + it("detects explicit hidden flags", () => { assert.equal(shouldStartHidden({ argv: ["electron", "--hidden"] }), true); assert.equal(shouldStartHidden({ argv: ["electron", "--minimized"] }), true); + assert.equal(shouldStartHidden({ argv: ["electron"] }), false); + assert.equal(shouldStartHidden(), false); + }); + + // Electron 44 removed `wasOpenedAsHidden` from `app.getLoginItemSettings()`, so a hidden + // autostart is signalled ONLY by the `--hidden` argument the login item registers. Guards + // against re-introducing a dependency on the removed field. + it("ignores login-item settings entirely", () => { assert.equal( shouldStartHidden({ argv: ["electron"], loginItemSettings: { wasOpenedAsHidden: true } }), + false + ); + assert.equal( + shouldStartHidden({ argv: ["electron", "--hidden"], loginItemSettings: {} }), true ); - assert.equal(shouldStartHidden({ argv: ["electron"], loginItemSettings: {} }), false); + }); + + it("keeps the --hidden argument registered with the login item", () => { + const mainJs = readFileSync(join(import.meta.dirname, "../../electron/main.js"), "utf8"); + assert.match(mainJs, /openAtLogin: true,\s*\n\s*args: \["--hidden"\],/); + assert.doesNotMatch(mainJs, /openAsHidden/); }); it("creates the dashboard only when an explicit open action has no live window", () => { From c41d8755db0c964471c228d7d641ba563fbb6211 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 08:39:46 -0300 Subject: [PATCH 025/143] fix(ci): document eloqnt MIT exceptions and isolate A2A vitest (#12595) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado numa worktree sobre o tip de `release/v3.8.51`. Verifiquei as premissas em vez de aceitar a justificativa: **Exceções de licença** — todas as afirmações batem. Os três pacotes estão instalados exatamente nas versões travadas citadas (`@eloqnt/config@0.0.2`, `@eloqnt/format-json@0.0.3`, `@eloqnt/format-po@0.0.3`), os três **omitem `package.json#license` e não têm arquivo LICENSE**, e `npm ls` confirma que são transitivos de `next-intl@4.14.1` (MIT). No registry, o SPDX das três é MIT e o `@eloqnt/config@0.1.0` existe, como o texto diz. Uso de `exceptions` (com `risk`/`reviewAt`) em vez de `allowed: UNKNOWN` é o mecanismo certo. `check:licenses` verde: **0 violações de política**, 948 permitidos, as três novas entradas aparecendo como exceções sinalizadas não bloqueantes ao lado das duas já existentes. Teste-guarda `tests/unit/build/check-licenses.test.ts`: **36/36**. **Vitest do A2A lifecycle** — as duas seams usadas já existiam antes deste PR: `constructor(ttlMinutes = 5, persistence: A2APersistence = defaultPersistence)` e o 4º parâmetro opcional `deps?: MemoryHitsDeps` de `executeA2ATaskWithState`. O padrão é idêntico ao que `tests/unit/a2a-task-persistence.test.ts` já fazia. Nenhuma asserção foi removida ou enfraquecida — e como o comportamento de persistência tem suíte dedicada, injetar no-ops aqui tira acoplamento incidental, não cobertura. Ganho medido nos dois lados: corpo dos testes **784ms no tip → 233ms com o PR**, sem nenhuma linha `[DB] SQLite database ready`. Registro honesto: **no tip o arquivo passa localmente** (4/4) — o red era do CI, sob o thread pool com as 167 migrações; localmente dá para comprovar o mecanismo e a aceleração, não a falha em si. Suíte vitest completa: **51/51 arquivos, 465/465 testes**. --- .../12581-basereds-licenses-a2a-lifecycle.md | 1 + config/quality/.license-allowlist.json | 18 +++++++++++++ .../mcp-server/__tests__/a2aLifecycle.test.ts | 25 +++++++++++++++---- tests/unit/build/check-licenses.test.ts | 16 ++++++++++++ 4 files changed, 55 insertions(+), 5 deletions(-) create mode 100644 changelog.d/fixes/12581-basereds-licenses-a2a-lifecycle.md diff --git a/changelog.d/fixes/12581-basereds-licenses-a2a-lifecycle.md b/changelog.d/fixes/12581-basereds-licenses-a2a-lifecycle.md new file mode 100644 index 0000000000..8dbb345f96 --- /dev/null +++ b/changelog.d/fixes/12581-basereds-licenses-a2a-lifecycle.md @@ -0,0 +1 @@ +- **fix(ci):** document MIT exceptions for `@eloqnt/{config,format-json,format-po}` (next-intl transitive; locked tarballs omit `license`) and keep the A2A lifecycle vitest off the real SQLite persistence seam ([#12581](https://github.com/diegosouzapw/OmniRoute/issues/12581)) diff --git a/config/quality/.license-allowlist.json b/config/quality/.license-allowlist.json index f2cca50374..c74ee066e4 100644 --- a/config/quality/.license-allowlist.json +++ b/config/quality/.license-allowlist.json @@ -74,6 +74,24 @@ "justification": "CC-BY-4.0 applies to the caniuse browser-support data (a dataset, not code). The Creative Commons Attribution license requires attribution when distributing — OmniRoute does not distribute caniuse-lite data directly to end users; it is consumed by browserslist/PostCSS at build time to generate CSS compatibility info. This is a widely accepted pattern in the Node.js ecosystem (caniuse-lite is in millions of projects). Attribution is satisfied by keeping the package in node_modules with its original license file.", "risk": "low", "reviewAt": "v4.0.0" + }, + "@eloqnt/config": { + "license": "MIT", + "justification": "Transitive of next-intl (MIT). npm registry SPDX for the @eloqnt scope is MIT; @eloqnt/config@0.1.0 republished with license: MIT. The locked 0.0.2 tarball (next-intl's ^0.0.2 range, which is 0.0.x only) omits both package.json#license and a LICENSE file, so license-checker reports UNKNOWN. Same author (Jan Amann / amannn). OmniRoute does not modify the package. Re-review when next-intl bumps the range to a release that ships the license field.", + "risk": "low", + "reviewAt": "v4.0.0" + }, + "@eloqnt/format-json": { + "license": "MIT", + "justification": "Same as @eloqnt/config: next-intl transitive, registry SPDX MIT, locked 0.0.3 tarball omits license field and LICENSE file so the checker reports UNKNOWN. Re-review with the next-intl range bump.", + "risk": "low", + "reviewAt": "v4.0.0" + }, + "@eloqnt/format-po": { + "license": "MIT", + "justification": "Same as @eloqnt/config: next-intl transitive, registry SPDX MIT, locked 0.0.3 tarball omits license field and LICENSE file so the checker reports UNKNOWN. Re-review with the next-intl range bump.", + "risk": "low", + "reviewAt": "v4.0.0" } } } diff --git a/open-sse/mcp-server/__tests__/a2aLifecycle.test.ts b/open-sse/mcp-server/__tests__/a2aLifecycle.test.ts index 470dc1f2bd..b660006a0f 100644 --- a/open-sse/mcp-server/__tests__/a2aLifecycle.test.ts +++ b/open-sse/mcp-server/__tests__/a2aLifecycle.test.ts @@ -1,11 +1,21 @@ import { afterEach, describe, expect, it } from "vitest"; -import { A2ATaskManager } from "../../../src/lib/a2a/taskManager.ts"; +import { A2ATaskManager, type A2APersistence } from "../../../src/lib/a2a/taskManager.ts"; import { executeA2ATaskWithState } from "../../../src/lib/a2a/taskExecution.ts"; const managers: A2ATaskManager[] = []; +// Default persistence opens SQLite (167 migrations) inside the vitest thread pool. +// Tests inject a no-op so they never touch the DB (same seam as a2a-task-persistence.test.ts). +function noopPersistence(): A2APersistence { + return { + upsert: (() => {}) as A2APersistence["upsert"], + appendEvent: (() => {}) as A2APersistence["appendEvent"], + purge: ((): number => 0) as A2APersistence["purge"], + }; +} + function createManager(ttlMinutes = 5) { - const manager = new A2ATaskManager(ttlMinutes); + const manager = new A2ATaskManager(ttlMinutes, noopPersistence()); managers.push(manager); return manager; } @@ -44,9 +54,14 @@ describe("A2A task lifecycle regressions", () => { tm.updateTask(task.id, "working"); await expect( - executeA2ATaskWithState(tm, task, async () => { - throw new Error("upstream failure"); - }) + executeA2ATaskWithState( + tm, + task, + async () => { + throw new Error("upstream failure"); + }, + { search: async () => [], appendEvent: () => {} } + ) ).rejects.toThrow("upstream failure"); const loaded = tm.getTask(task.id); diff --git a/tests/unit/build/check-licenses.test.ts b/tests/unit/build/check-licenses.test.ts index 746f9cc9dc..8fb104ef9a 100644 --- a/tests/unit/build/check-licenses.test.ts +++ b/tests/unit/build/check-licenses.test.ts @@ -330,3 +330,19 @@ test("integration: classifyLicense denies AGPL-3.0 against real allowlist", () = const result = classifyLicense("hypothetical-agpl@1.0.0", "AGPL-3.0", allowlist); assert.equal(result.status, "denied"); }); + +test("integration: @eloqnt/* UNKNOWN licenses are documented exceptions (next-intl transitive)", () => { + const allowlist = loadAllowlist(); + for (const pkg of [ + "@eloqnt/config@0.0.2", + "@eloqnt/format-json@0.0.3", + "@eloqnt/format-po@0.0.3", + ]) { + const result = classifyLicense(pkg, "UNKNOWN", allowlist); + assert.equal( + result.status, + "exception", + `${pkg} ships no license field; must be a documented exception, not allowed/denied` + ); + } +}); From e1cf5423785820d073e42b4f1536c48083a184b2 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 09:07:15 -0300 Subject: [PATCH 026/143] chore(deps): bump fast-uri to 3.1.7 in the electron lockfile (#12601) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Dependabot alerts #196–#199 — four HIGH advisories on fast-uri (GHSA-jqff-g426-hqxp, GHSA-fph4-wmhf-6fwf, GHSA-f65p-4m7j-42xc, GHSA-5jgf-p345-68v8), all patched in 3.1.6. The root package-lock.json was already on a patched fast-uri (3.1.7) — those alerts close on their own with the next scan. `electron/package-lock.json` is a second lockfile and was still pinning 3.1.5, which is what these four alerts are actually reporting. Transitive, one copy, pulled by ajv (`^3.0.1`), so a package-lock-only update lifts it without touching any manifest. The diff is three lines: version, resolved and integrity for that single entry. check:lockfile and check:tracked-artifacts pass. Not fixed here: extract-zip (#191, HIGH, <= 2.0.1) has no published patch. It comes in through @openai/codex-security and is dev-scope; it needs either an upstream release or a decision to drop/replace the dependency, neither of which belongs in a lockfile bump. --- electron/package-lock.json | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/electron/package-lock.json b/electron/package-lock.json index 7dbc37a13c..8eb9f2634a 100644 --- a/electron/package-lock.json +++ b/electron/package-lock.json @@ -1575,9 +1575,9 @@ "license": "MIT" }, "node_modules/fast-uri": { - "version": "3.1.5", - "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.5.tgz", - "integrity": "sha512-gHwA1O9LDIcKunMKhObS/HimwtehO1nPUECKAu5TpKgaO19fcWEl4bliWe1jWxVFvIXztJjjQ4L8XQ1EU9f7Jw==", + "version": "3.1.7", + "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.7.tgz", + "integrity": "sha512-dOvZVzjdZdz7phd9v6jCbwxrBW3fK6n8Rc0CtdmM4bumzMnxywBYhuph6J819RRw/ku+rLbelwfMunktuzVVHg==", "dev": true, "funding": [ { From 3f3d27e264ecdcd8479e6f40dd598ddea88ec7e3 Mon Sep 17 00:00:00 2001 From: Giorgos Giakoumettis Date: Thu, 3 Sep 2026 15:10:45 +0300 Subject: [PATCH 027/143] fix(ci): openapi-security-tiers checker must honor routeGuard patterns + imported prefixes (#12350) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado numa worktree sobre o tip de `release/v3.8.51`, medindo o gate dos dois lados: **red no tip** (dezenas de rotas `volcengine-plan`/`vnc-session` reportadas como "has x-loopback-only but is NOT covered") e **PASS com este PR**, exit 0. Como é um gate de segurança, confirmei que o fix torna o checker *preciso* e não *frouxo*. A afirmação central do PR — que uma rota é coberta se casar com um prefixo resolvido **ou** com um pattern — bate exatamente com o runtime (`src/server/authz/routeGuard.ts:252-255`): ```ts return ( LOCAL_ONLY_API_PREFIXES.some((p) => path === p || path.startsWith(p)) || LOCAL_ONLY_API_PATTERNS.some((re) => re.test(path)) ); ``` O checker antigo enxergava só o primeiro braço, e nem isso por completo: a captura `[^\]]+` quebrava no `]` dentro de classes de regex, então `LOCAL_ONLY_API_PATTERNS` não era parseado, e `VNC_ROUTE_PREFIX` (const importada, não literal) não era resolvido. Resultado: rotas efetivamente protegidas em runtime apareciam como desprotegidas. Nenhum achado real foi silenciado — as 95 linhas de `WARN — missing x-loopback-only annotation` continuam saindo, são explicitamente não-fatais e pré-existentes. Fecha um dos HARDs do base-red #12335. Obrigado, @ggiak. --- .../check/check-openapi-security-tiers.mjs | 174 +++++++++++++----- 1 file changed, 127 insertions(+), 47 deletions(-) diff --git a/scripts/check/check-openapi-security-tiers.mjs b/scripts/check/check-openapi-security-tiers.mjs index 812e4a0c1b..89d17c0482 100644 --- a/scripts/check/check-openapi-security-tiers.mjs +++ b/scripts/check/check-openapi-security-tiers.mjs @@ -1,9 +1,23 @@ #!/usr/bin/env node /** * Cross-references openapi.yaml x-loopback-only / x-always-protected annotations - * against the compile-time constants in src/server/authz/routeGuard.ts. + * against the compile-time route-classification constants in + * src/server/authz/routeGuard.ts. * - * Fails if any YAML annotation disagrees with the routeGuard.ts constants. + * routeGuard classifies a loopback-only route through TWO mechanisms, and this + * checker must honor BOTH or it reports false positives (regression #12335): + * + * 1. LOCAL_ONLY_API_PREFIXES — flat string prefixes. One entry + * (VNC_ROUTE_PREFIX) is an imported const rather than a string literal, so + * it is resolved from its source module. + * 2. LOCAL_ONLY_API_PATTERNS — RegExp entries for spawn-capable routes whose + * dynamic path parameter sits BEFORE the gated segment (e.g. + * /api/providers/{id}/login), which a flat prefix cannot target without + * over-broadening the whole /api/providers/ subtree. + * + * A route is "covered" iff it matches a resolved prefix OR a pattern — exactly + * the `isLocalOnlyPath()` runtime contract. Fails if any YAML annotation + * disagrees with the routeGuard.ts constants. */ import fs from "node:fs"; @@ -13,34 +27,116 @@ import * as yaml from "js-yaml"; const ROOT = process.cwd(); const OPENAPI_PATH = path.join(ROOT, "docs", "openapi.yaml"); const ROUTE_GUARD_PATH = path.join(ROOT, "src", "server", "authz", "routeGuard.ts"); +const guardSrc = fs.readFileSync(ROUTE_GUARD_PATH, "utf-8"); -function parseStringArray(match) { - if (!match) return []; - // Strip line comments before splitting — array entries in routeGuard.ts often - // carry inline `// T-XX:` annotations that would otherwise pollute the parsed tokens. - return match[1] - .replace(/\/\/[^\n]*/g, "") - .split(",") - .map((s) => s.trim().replace(/^["']|["']$/g, "")) - .filter(Boolean); +// Capture an exported array's body up to its closing `\n];`. Unlike a `[^\]]+` +// capture, this is immune to `]` characters inside comments or regex character +// classes (e.g. `[^/]`) — the exact footgun documented at routeGuard.ts's +// /api/oauth/cursor/auto-import entry, and the reason regex patterns could not +// be parsed at all before. +function extractArrayBody(name) { + const m = guardSrc.match( + new RegExp(`export const ${name}\\b[\\s\\S]*?=\\s*\\[([\\s\\S]*?)\\n\\];`) + ); + return m ? m[1] : null; } -const guardSrc = fs.readFileSync(ROUTE_GUARD_PATH, "utf-8"); -const LOCAL_ONLY_PREFIXES = parseStringArray( - guardSrc.match(/export const LOCAL_ONLY_API_PREFIXES.*?=\s*\[([^\]]+)\]/s) -); -const ALWAYS_PROTECTED_PATHS = parseStringArray( - guardSrc.match(/export const ALWAYS_PROTECTED_API_PATHS.*?=\s*\[([^\]]+)\]/s) -); +const stripLineComments = (s) => s.replace(/\/\/[^\n]*/g, ""); -if (LOCAL_ONLY_PREFIXES.length === 0 || ALWAYS_PROTECTED_PATHS.length === 0) { - console.error("[openapi-security-tiers] FAIL — could not parse routeGuard.ts constants"); +function resolveModule(spec) { + let base; + if (spec.startsWith("@/")) base = path.join(ROOT, "src", spec.slice(2)); + else if (spec.startsWith(".")) base = path.resolve(path.dirname(ROUTE_GUARD_PATH), spec); + else throw new Error(`openapi-security-tiers: unsupported import specifier '${spec}'`); + for (const cand of [base, `${base}.ts`, `${base}.mts`, path.join(base, "index.ts")]) { + if (fs.existsSync(cand) && fs.statSync(cand).isFile()) return cand; + } + throw new Error(`openapi-security-tiers: cannot resolve module '${spec}' (from ${base})`); +} + +// Resolve a bare identifier used inside a prefix array (e.g. VNC_ROUTE_PREFIX) +// to its string-literal value by following its import in routeGuard.ts. +function resolveIdentifier(ident) { + const imp = guardSrc.match( + new RegExp(`import\\s*(?:type\\s*)?\\{[^}]*\\b${ident}\\b[^}]*\\}\\s*from\\s*["']([^"']+)["']`) + ); + if (!imp) + throw new Error( + `openapi-security-tiers: '${ident}' used in a prefix array has no import in routeGuard.ts` + ); + const modSrc = fs.readFileSync(resolveModule(imp[1]), "utf-8"); + const lit = modSrc.match(new RegExp(`export const ${ident}\\s*=\\s*["']([^"']+)["']`)); + if (!lit) + throw new Error(`openapi-security-tiers: cannot resolve '${ident}' to a string literal`); + return lit[1]; +} + +// String prefixes: quoted entries pass through; bare identifiers are resolved. +function parsePrefixes(name) { + const body = extractArrayBody(name); + if (body == null) + throw new Error(`openapi-security-tiers: could not locate ${name} in routeGuard.ts`); + return stripLineComments(body) + .split(",") + .map((s) => s.trim()) + .filter(Boolean) + .map((tok) => { + const unquoted = tok.replace(/^["']|["']$/g, ""); + return unquoted !== tok ? unquoted : resolveIdentifier(tok); + }); +} + +// RegExp patterns: one `/.../ ` literal per line. +function parsePatterns(name) { + const body = extractArrayBody(name); + if (body == null) + throw new Error(`openapi-security-tiers: could not locate ${name} in routeGuard.ts`); + const out = []; + for (const raw of body.split("\n")) { + const t = raw + .replace(/\/\/.*$/, "") + .trim() + .replace(/,\s*$/, "") + .trim(); + if (t.length > 2 && t.startsWith("/") && t.endsWith("/")) out.push(new RegExp(t.slice(1, -1))); + } + return out; +} + +const LOCAL_ONLY_PREFIXES = parsePrefixes("LOCAL_ONLY_API_PREFIXES"); +const LOCAL_ONLY_PATTERNS = parsePatterns("LOCAL_ONLY_API_PATTERNS"); +const ALWAYS_PROTECTED_PATHS = parsePrefixes("ALWAYS_PROTECTED_API_PATHS"); + +if ( + LOCAL_ONLY_PREFIXES.length === 0 || + LOCAL_ONLY_PATTERNS.length === 0 || + ALWAYS_PROTECTED_PATHS.length === 0 +) { + console.error( + `[openapi-security-tiers] FAIL — could not parse routeGuard.ts constants ` + + `(prefixes=${LOCAL_ONLY_PREFIXES.length}, patterns=${LOCAL_ONLY_PATTERNS.length}, ` + + `alwaysProtected=${ALWAYS_PROTECTED_PATHS.length})` + ); process.exit(1); } +// OpenAPI template params ({id}, {sessionId}, …) → a concrete single non-slash +// segment, so pattern regexes written against resolved paths (`[^/]+`) match. +const concretize = (p) => p.replace(/\{[^}]+\}/g, "x"); + +const matchesPrefix = (concrete) => + LOCAL_ONLY_PREFIXES.some((prefix) => { + const norm = prefix.endsWith("/") ? prefix.slice(0, -1) : prefix; + return concrete === norm || concrete.startsWith(`${norm}/`); + }); + +function coveredByLocalOnly(pathStr) { + const concrete = concretize(pathStr); + return matchesPrefix(concrete) || LOCAL_ONLY_PATTERNS.some((re) => re.test(concrete)); +} + const raw = yaml.load(fs.readFileSync(OPENAPI_PATH, "utf-8")); const paths = raw.paths || {}; - const errors = []; for (const [pathStr, methods] of Object.entries(paths)) { @@ -48,17 +144,11 @@ for (const [pathStr, methods] of Object.entries(paths)) { for (const [method, spec] of Object.entries(methods)) { if (!["get", "post", "put", "patch", "delete"].includes(method) || !spec) continue; - if (spec["x-loopback-only"] === true) { - const matchesPrefix = LOCAL_ONLY_PREFIXES.some((prefix) => { - const norm = prefix.endsWith("/") ? prefix.slice(0, -1) : prefix; - return pathStr === norm || pathStr.startsWith(norm + "/"); - }); - if (!matchesPrefix) { - errors.push( - `${method.toUpperCase()} ${pathStr}: has x-loopback-only but is NOT covered by ` + - `LOCAL_ONLY_API_PREFIXES [${LOCAL_ONLY_PREFIXES.join(", ")}]` - ); - } + if (spec["x-loopback-only"] === true && !coveredByLocalOnly(pathStr)) { + errors.push( + `${method.toUpperCase()} ${pathStr}: has x-loopback-only but is NOT covered by ` + + `LOCAL_ONLY_API_PREFIXES or LOCAL_ONLY_API_PATTERNS` + ); } if (spec["x-always-protected"] === true) { @@ -75,23 +165,13 @@ for (const [pathStr, methods] of Object.entries(paths)) { } } -// Reverse pass: every YAML path that falls under a LOCAL_ONLY prefix should -// carry `x-loopback-only: true` on every method, otherwise external API -// consumers have no signal that the route is loopback-restricted. Closes the -// "new spawn-capable route added without annotation" regression class. -// -// Currently reported as warnings (non-fatal) because the v3.8.4 release ships -// with a known annotation gap on /api/services/* and /api/cli-tools/runtime/* -// that will be patched in a follow-up doc-only PR. Promote to errors once the -// backlog is cleared. +// Reverse pass (non-fatal): every YAML path that falls under a LOCAL_ONLY prefix +// should carry `x-loopback-only`. Pattern-only routes are intentionally excluded +// — they are not "under" a broad prefix. Known annotation gaps stay warnings. const reverseWarnings = []; for (const [pathStr, methods] of Object.entries(paths)) { if (!methods || typeof methods !== "object") continue; - const fallsUnderLocalOnly = LOCAL_ONLY_PREFIXES.some((prefix) => { - const norm = prefix.endsWith("/") ? prefix.slice(0, -1) : prefix; - return pathStr === norm || pathStr.startsWith(norm + "/"); - }); - if (!fallsUnderLocalOnly) continue; + if (!matchesPrefix(concretize(pathStr))) continue; for (const [method, spec] of Object.entries(methods)) { if (!["get", "post", "put", "patch", "delete"].includes(method) || !spec) continue; if (spec["x-loopback-only"] !== true) { @@ -105,7 +185,7 @@ for (const [pathStr, methods] of Object.entries(paths)) { if (reverseWarnings.length > 0) { console.warn( - `[openapi-security-tiers] WARN — ${reverseWarnings.length} LOCAL_ONLY paths missing x-loopback-only annotation (non-fatal, follow-up doc PR):` + `[openapi-security-tiers] WARN — ${reverseWarnings.length} LOCAL_ONLY paths missing x-loopback-only annotation (non-fatal):` ); reverseWarnings.forEach((w) => console.warn(` - ${w}`)); } From 49c4a620ca651227b97029852df31d8ab2db1595 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 09:19:44 -0300 Subject: [PATCH 028/143] fix(authz): hard-gate every credential export and CLI-config write (GHSA-5926-2w35-7h4q) (#12600) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(authz): hard-gate every credential export and CLI-config write GHSA-5926-2w35-7h4q: `POST /api/providers/{id}/claude-auth/export` and `.../codex-auth/export` gate on `requireManagementAuth(request)` with no `alwaysRequireAuth`, and neither path was in ALWAYS_PROTECTED_API_PATHS. Under `requireLogin=false` — the local-first default — both fail open, so anyone who knows a connection id downloads the operator's raw Claude/Codex OAuth access_token / refresh_token (plus the Codex id_token). This is the third recurrence of one class. GHSA-mghq-58h3-qcqj added /api/db-backups; GHSA-v7g9-7f55-5g46 added the /api/settings/*-json siblings mghq had missed; these two are the siblings both missed. So the fix is written against the class, not the two reported routes. Sweeping every route that hands out stored credentials, dumps captured traffic, or writes the operator's CLI config turned up four more on the fail-open tier: - GET /api/logs/export — dumps call_logs (prompts and responses) and proxy_logs for up to 168h. - /api/cli-tools/codex-profiles — GET leaks the operator's account label; PUT writes attacker-supplied auth.json and config.toml straight into the host's Codex CLI config. Its only guard is ensureCliConfigWriteAllowed() with no targetPath, which checks CLI_ALLOW_CONFIG_WRITES — default true. Paired with the POST that stores an arbitrary profile, that is: save a profile holding the attacker's auth.json, apply it, and the operator's CLI now runs on attacker credentials (or, via config.toml, an attacker base URL). - {claude,codex}-auth/apply-local and providers/agy-auth/apply-local — write a stored credential into ~/.codex/auth.json and ~/.gemini/antigravity-cli/antigravity-oauth-token. The traffic-inspector HAR exports were already covered by LOCAL_ONLY. Routes with a dynamic segment cannot be expressed in the exact/prefix list — a `/api/providers/` prefix would hard-gate the whole provider surface and break every keyless install — so this adds ALWAYS_PROTECTED_API_PATTERNS, mirroring the existing LOCAL_ONLY_API_PATTERNS, and `isAlwaysProtectedPath` consults both. The apply-local routes get ALWAYS_PROTECTED rather than LOCAL_ONLY on purpose: it closes the anonymous hole without breaking an operator driving the dashboard through a tunnel. Deliberately NOT adding `{ alwaysRequireAuth: true }` at the handlers. Tier 2 is the architecture's designated mechanism and the guard runs before the handler; a second copy of the same decision inside each route is exactly the kind of duplicate that drifts out of sync (cf. the dashboardCsrf prefix scan that had to be unified in #11417). tests/unit/authz/credential-export-always-protected.test.ts — 5 tests, red before the fix. Written as an inventory of the whole class rather than two more assertions, plus negative cases: the neighbouring provider routes must stay on MANAGEMENT, and a connection id containing a slash must not slip past `[^/]+`. openapi.yaml marks the seven newly-gated operations `x-always-protected`, and openapi-security-tiers.test.ts now resolves `{param}` placeholders so it can validate the pattern entries too. Reported by @skeletonsec. Closes GHSA-5926-2w35-7h4q * chore(quality): register the credential-export authz test in stryker tap.testFiles The new tests/unit/authz/credential-export-always-protected.test.ts covers src/server/authz/routeGuard.ts, so check:mutation-test-coverage --strict fails until it is listed — its mutant kills would not count otherwise. Inserted in place (no re-serialization: a JSON round-trip on this file reorders ~10 curated entries that are already out of alphabetical order, cf. #11438). --- docs/openapi.yaml | 10 ++ src/server/authz/routeGuard.ts | 41 +++++- stryker.conf.json | 1 + ...credential-export-always-protected.test.ts | 132 ++++++++++++++++++ tests/unit/openapi-security-tiers.test.ts | 26 +++- 5 files changed, 203 insertions(+), 7 deletions(-) create mode 100644 tests/unit/authz/credential-export-always-protected.test.ts diff --git a/docs/openapi.yaml b/docs/openapi.yaml index 2f83d88080..16a39eef8d 100644 --- a/docs/openapi.yaml +++ b/docs/openapi.yaml @@ -2297,6 +2297,7 @@ paths: post: tags: [Providers] summary: Auto-detect and import the local Antigravity CLI (agy) login from disk + x-always-protected: true responses: "200": description: Created or updated provider connection @@ -4050,12 +4051,14 @@ paths: get: tags: [CLI Tools] summary: Get Codex profiles + x-always-protected: true responses: "200": description: Codex profile list post: tags: [CLI Tools] summary: Create Codex profile + x-always-protected: true requestBody: required: true content: @@ -4068,6 +4071,7 @@ paths: put: tags: [CLI Tools] summary: Update Codex profile + x-always-protected: true requestBody: required: true content: @@ -4080,6 +4084,7 @@ paths: delete: tags: [CLI Tools] summary: Delete Codex profile + x-always-protected: true responses: "200": description: Profile deleted @@ -9751,6 +9756,7 @@ paths: tags: - Logs summary: "GET logs › export" + x-always-protected: true responses: "200": description: OK @@ -10230,6 +10236,7 @@ paths: tags: - Providers summary: "POST providers › › claude auth › apply local" + x-always-protected: true responses: "200": description: OK @@ -10238,6 +10245,7 @@ paths: tags: - Providers summary: "POST providers › › claude auth › export" + x-always-protected: true responses: "200": description: OK @@ -10246,6 +10254,7 @@ paths: tags: - Providers summary: "POST providers › › codex auth › apply local" + x-always-protected: true responses: "200": description: OK @@ -10254,6 +10263,7 @@ paths: tags: - Providers summary: "POST providers › › codex auth › export" + x-always-protected: true responses: "200": description: OK diff --git a/src/server/authz/routeGuard.ts b/src/server/authz/routeGuard.ts index 60a8d8e15a..b987f4fe49 100644 --- a/src/server/authz/routeGuard.ts +++ b/src/server/authz/routeGuard.ts @@ -140,6 +140,42 @@ export const ALWAYS_PROTECTED_API_PATHS: ReadonlyArray = [ // which is false under requireLogin=false. (GHSA-v7g9-7f55-5g46) "/api/settings/export-json", "/api/settings/import-json", + // Bulk log export: call_logs carries prompts and responses, proxy_logs carries + // client/public IPs, and the handler only calls requireManagementAuth() with no + // alwaysRequireAuth. Found sweeping the GHSA-5926-2w35-7h4q class. + "/api/logs/export", + // Codex CLI profile store. GET leaks the operator's account label; PUT writes + // attacker-supplied auth.json + config.toml straight into the operator's Codex + // CLI config (ensureCliConfigWriteAllowed() only checks CLI_ALLOW_CONFIG_WRITES, + // which defaults to true), so a POST+PUT pair repoints the CLI at attacker + // credentials or an attacker base URL. Found sweeping the same class. + "/api/cli-tools/codex-profiles", + // Writes into ~/.gemini/antigravity-cli/antigravity-oauth-token. Same family + // as the {claude,codex}-auth/apply-local pattern below; a plain path because + // it carries no dynamic segment. + "/api/providers/agy-auth/apply-local", +]; + +/** + * ALWAYS_PROTECTED routes whose path carries a dynamic segment, so the plain + * exact/prefix list above cannot express them: a `/api/providers/` prefix would + * hard-gate the entire provider surface and break every keyless local-first + * install. Mirrors LOCAL_ONLY_API_PATTERNS. + * + * The Claude/Codex OAuth export routes return the connection's raw + * access_token / refresh_token (and the Codex id_token) and gate only on + * `requireManagementAuth(request)` with no `alwaysRequireAuth`, which fails open + * under requireLogin=false (GHSA-5926-2w35-7h4q). They are the siblings that + * both GHSA-mghq-58h3-qcqj and GHSA-v7g9-7f55-5g46 missed. + */ +export const ALWAYS_PROTECTED_API_PATTERNS: ReadonlyArray = [ + // `export` hands the caller the raw token; `apply-local` writes it into the + // host's CLI config (~/.codex/auth.json and the Claude equivalent). The second + // does not disclose the credential, but "anonymous" is still the wrong + // audience for it. ALWAYS_PROTECTED rather than LOCAL_ONLY on purpose: it + // closes the anonymous hole without breaking an operator driving the dashboard + // through a tunnel. + /^\/api\/providers\/[^/]+\/(claude|codex)-auth\/(export|apply-local)\/?$/, ]; export function isLoopbackHost(hostHeader: string | null): boolean { @@ -295,5 +331,8 @@ export function isLocalOnlyBypassableByManageScope(path: string): boolean { } export function isAlwaysProtectedPath(path: string): boolean { - return ALWAYS_PROTECTED_API_PATHS.some((p) => path === p || path.startsWith(p)); + return ( + ALWAYS_PROTECTED_API_PATHS.some((p) => path === p || path.startsWith(p)) || + ALWAYS_PROTECTED_API_PATTERNS.some((re) => re.test(path)) + ); } diff --git a/stryker.conf.json b/stryker.conf.json index d13bf2c1d4..23dd4fb2d5 100644 --- a/stryker.conf.json +++ b/stryker.conf.json @@ -90,6 +90,7 @@ "tests/unit/auth-opencode-zen-noauth-fallback.test.ts", "tests/unit/auth-passthrough-per-model-402-12242.test.ts", "tests/unit/auth-terminal-status.test.ts", + "tests/unit/authz/credential-export-always-protected.test.ts", "tests/unit/authz/discovery-routes-local-only.test.ts", "tests/unit/authz/oauth-autoimport-local-only.test.ts", "tests/unit/authz/route-guard-local-prefix.test.ts", diff --git a/tests/unit/authz/credential-export-always-protected.test.ts b/tests/unit/authz/credential-export-always-protected.test.ts new file mode 100644 index 0000000000..c297bad003 --- /dev/null +++ b/tests/unit/authz/credential-export-always-protected.test.ts @@ -0,0 +1,132 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { + isAlwaysProtectedPath, + isLocalOnlyPath, + ALWAYS_PROTECTED_API_PATHS, +} from "../../../src/server/authz/routeGuard.ts"; + +// GHSA-5926-2w35-7h4q — the Claude/Codex OAuth export routes gate on +// `requireManagementAuth(request)` with no `alwaysRequireAuth`, which fails open +// under requireLogin=false, and neither path was in ALWAYS_PROTECTED_API_PATHS. +// An unauthenticated caller who knows a connection id could download the +// operator's raw access_token / refresh_token / id_token. +// +// This is the THIRD recurrence of one class: GHSA-mghq-58h3-qcqj added +// /api/db-backups, GHSA-v7g9-7f55-5g46 added the /api/settings/*-json siblings +// it had missed, and this one is the siblings BOTH missed. So the test is +// written as an inventory of the whole class rather than two more assertions: +// a route that hands out stored credentials, dumps captured traffic, or writes +// the operator's CLI config must be hard-gated (ALWAYS_PROTECTED or +// LOCAL_ONLY), never left on the fail-open MANAGEMENT tier. + +const HARD_GATED_INVENTORY: ReadonlyArray<{ path: string; why: string }> = [ + // ── Reported in GHSA-5926-2w35-7h4q ────────────────────────────────────── + { + path: "/api/providers/6f3c1b7e-0000-4000-8000-000000000000/claude-auth/export", + why: "returns the connection's raw Claude OAuth access_token/refresh_token", + }, + { + path: "/api/providers/6f3c1b7e-0000-4000-8000-000000000000/codex-auth/export", + why: "returns the connection's raw Codex access_token/refresh_token/id_token", + }, + // ── Found sweeping the class while fixing the above ────────────────────── + { + path: "/api/logs/export", + why: "dumps call_logs (prompts and responses) and proxy_logs for up to 168h", + }, + { + path: "/api/cli-tools/codex-profiles", + why: "PUT writes attacker-supplied auth.json and config.toml into the operator's Codex CLI config", + }, + // ── Same family: WRITE the operator's credentials into host CLI files ─── + // These do not hand the credential to the caller, so they are a step below + // the export routes — but anonymous is still the wrong audience for "write + // this connection's token into ~/.codex/auth.json". ALWAYS_PROTECTED rather + // than LOCAL_ONLY on purpose: it closes the anonymous hole without breaking + // an operator driving the dashboard through a tunnel. + { + path: "/api/providers/6f3c1b7e-0000-4000-8000-000000000000/codex-auth/apply-local", + why: "writes the connection's credential into the host's ~/.codex/auth.json", + }, + { + path: "/api/providers/6f3c1b7e-0000-4000-8000-000000000000/claude-auth/apply-local", + why: "writes the connection's credential into the host's Claude CLI config", + }, + { + path: "/api/providers/agy-auth/apply-local", + why: "writes into ~/.gemini/antigravity-cli/antigravity-oauth-token", + }, + // ── Already fixed; pinned so a refactor cannot silently drop them ──────── + { path: "/api/db-backups/export", why: "GHSA-mghq-58h3-qcqj" }, + { path: "/api/db-backups/exportAll", why: "GHSA-mghq-58h3-qcqj" }, + { path: "/api/settings/export-json", why: "GHSA-v7g9-7f55-5g46" }, + { path: "/api/settings/import-json", why: "GHSA-v7g9-7f55-5g46" }, + { path: "/api/settings/database", why: "irreversible database replace" }, + { path: "/api/shutdown", why: "stops the server" }, + // ── Hard-gated by the LOCAL_ONLY tier instead ──────────────────────────── + { + path: "/api/tools/traffic-inspector/export.har", + why: "captured traffic can contain Authorization headers (LOCAL_ONLY)", + }, + { + path: "/api/tools/traffic-inspector/sessions/abc/export.har", + why: "same, per session (LOCAL_ONLY)", + }, +]; + +test("every credential/traffic export and CLI-config write is hard-gated", () => { + for (const { path, why } of HARD_GATED_INVENTORY) { + const gated = isAlwaysProtectedPath(path) || isLocalOnlyPath(path); + assert.ok( + gated, + `${path} is on the fail-open MANAGEMENT tier — anonymous under requireLogin=false. ${why}` + ); + } +}); + +test("the trailing-slash spelling is gated too", () => { + for (const path of [ + "/api/providers/abc/claude-auth/export/", + "/api/providers/abc/codex-auth/export/", + "/api/logs/export/", + "/api/cli-tools/codex-profiles/", + ]) { + assert.ok(isAlwaysProtectedPath(path) || isLocalOnlyPath(path), path); + } +}); + +test("the new patterns do not over-protect their neighbours", () => { + // The dynamic-segment entries must not swallow the rest of /api/providers/, + // which is ordinary MANAGEMENT and has to keep working under requireLogin=false. + for (const path of [ + "/api/providers", + "/api/providers/abc", + "/api/providers/abc/models", + "/api/providers/abc/claude-auth", + "/api/providers/abc/codex-auth", + "/api/providers/abc/claude-auth/apply", + "/api/providers/agy-auth", + "/api/logs", + "/api/cli-tools", + ]) { + assert.equal( + isAlwaysProtectedPath(path), + false, + `${path} must stay on the MANAGEMENT tier — hard-gating it breaks keyless local-first installs` + ); + } +}); + +test("a connection id cannot escape the pattern with a slash", () => { + // `[^/]+` is deliberate: a traversal-ish id must not match and silently drop + // back to the fail-open tier by looking like a different route. + assert.equal(isAlwaysProtectedPath("/api/providers/a/b/claude-auth/export"), false); +}); + +test("the plain-path allowlist keeps its existing entries", () => { + for (const p of ["/api/shutdown", "/api/settings/database", "/api/db-backups"]) { + assert.ok(ALWAYS_PROTECTED_API_PATHS.includes(p), p); + } +}); diff --git a/tests/unit/openapi-security-tiers.test.ts b/tests/unit/openapi-security-tiers.test.ts index 8203ab6f2d..d1380a5c08 100644 --- a/tests/unit/openapi-security-tiers.test.ts +++ b/tests/unit/openapi-security-tiers.test.ts @@ -7,8 +7,12 @@ import * as yaml from "js-yaml"; const ROOT = process.cwd(); const OPENAPI_PATH = path.join(ROOT, "docs", "openapi.yaml"); -const { LOCAL_ONLY_API_PREFIXES, LOCAL_ONLY_API_PATTERNS, ALWAYS_PROTECTED_API_PATHS } = - await import("../../src/server/authz/routeGuard.ts"); +const { + LOCAL_ONLY_API_PREFIXES, + LOCAL_ONLY_API_PATTERNS, + ALWAYS_PROTECTED_API_PATHS, + ALWAYS_PROTECTED_API_PATTERNS, +} = await import("../../src/server/authz/routeGuard.ts"); const raw: any = yaml.load(fs.readFileSync(OPENAPI_PATH, "utf-8")); const paths: Record = raw.paths || {}; @@ -131,12 +135,22 @@ test("every x-always-protected path matches ALWAYS_PROTECTED_API_PATHS in routeG for (const [method, spec] of Object.entries(methods as Record)) { if (!["get", "post", "put", "patch", "delete"].includes(method)) continue; if (spec?.["x-always-protected"] !== true) continue; - const matchesPath = (ALWAYS_PROTECTED_API_PATHS as ReadonlyArray).some( - (p: string) => pathStr === p || pathStr.startsWith(`${p}/`) - ); + // Routes with a dynamic segment cannot be expressed in the plain + // exact/prefix list, so routeGuard also carries ALWAYS_PROTECTED_API_PATTERNS + // (GHSA-5926-2w35-7h4q). Substitute a concrete value for the OpenAPI + // `{param}` placeholders before testing those. + const concretePath = pathStr.replace(/\{[^}]+\}/g, "sample-id"); + const matchesPath = + (ALWAYS_PROTECTED_API_PATHS as ReadonlyArray).some( + (p: string) => pathStr === p || pathStr.startsWith(`${p}/`) + ) || + (ALWAYS_PROTECTED_API_PATTERNS as ReadonlyArray).some((re) => + re.test(concretePath) + ); assert.ok( matchesPath, - `YAML path "${pathStr}" ${method.toUpperCase()} has x-always-protected but is NOT in ALWAYS_PROTECTED_API_PATHS. ` + + `YAML path "${pathStr}" ${method.toUpperCase()} has x-always-protected but is NOT in ALWAYS_PROTECTED_API_PATHS ` + + `nor matched by ALWAYS_PROTECTED_API_PATTERNS. ` + `Entries: ${(ALWAYS_PROTECTED_API_PATHS as ReadonlyArray).join(", ")}` ); } From 2c4ad3e557a74a20e6728f6e2625ebc09742768a Mon Sep 17 00:00:00 2001 From: Giorgos Giakoumettis Date: Thu, 3 Sep 2026 15:21:00 +0300 Subject: [PATCH 029/143] =?UTF-8?q?docs(readme):=20introduce=20OmniRouteTr?= =?UTF-8?q?ay=20=E2=80=94=20the=20macOS=20menu-bar=20companion=20(#12276)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Aprovado pelo operador. Adiciona o OmniRouteTray (@zoispag) ao README — app de menu-bar para macOS, rotulado com honestidade como projeto da comunidade e não release oficial. Mudança só de markdown, sem tocar nada executável; inclui também dois ajustes de alinhamento na tabela de contatos. Obrigado, @ggiak. --- README.md | 39 ++++++++++++++++++++++++++++++++++++++- 1 file changed, 38 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index a026f9c9c3..055f71a88b 100644 --- a/README.md +++ b/README.md @@ -724,6 +724,7 @@ of your shell history. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md)
📦
npm (global)npm install -g omnirouteOne command, any OS 🐳 Dockerdocker run … diegosouzapw/omnirouteMulti-arch AMD64 + ARM64 🖥️ Desktop (Electron)npm run electron:buildNative window + system tray — Windows / macOS / Linux + 🎩 Menu-bar (OmniRouteTray)brew install --cask zoispag/tap/omniroute-traySupervises & auto-updates the server — macOS 💪 ARMnative arm64Raspberry Pi, ARM servers, Apple Silicon 📱 Android (Termux)pkg install nodejs && npx -y omnirouteRuns on your phone, 24/7, no root 📲 PWA"Add to Home Screen"Fullscreen, offline, installable from browser @@ -732,7 +733,7 @@ of your shell history. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md) 🛠️ From sourcenpm install && npm run devHack on it, contribute -📖 [Docker Guide](docs/guides/DOCKER_GUIDE.md) · [Desktop](electron/README.md) · [Termux](docs/guides/TERMUX_GUIDE.md) · [PWA](docs/guides/PWA_GUIDE.md) · [OpenCode](docs/frameworks/OPENCODE.md) +📖 [Docker Guide](docs/guides/DOCKER_GUIDE.md) · [Desktop](electron/README.md) · [Menu-bar tray](https://github.com/zoispag/omniroute-tray) · [Termux](docs/guides/TERMUX_GUIDE.md) · [PWA](docs/guides/PWA_GUIDE.md) · [OpenCode](docs/frameworks/OPENCODE.md)
@@ -767,6 +768,42 @@ From inside the editor: open the **Extensions** view, search **"OmniRoute"**, cl
+### 🎩 New: OmniRouteTray — your gateway, living in the menu bar + +
+ +> `omniroute serve` is happiest when it's always on. **[OmniRouteTray](https://github.com/zoispag/omniroute-tray)** +> turns that into a set-and-forget menu-bar app for macOS: it starts the server, keeps it alive +> across reboots, updates it in place, and puts your live token budget one click away — **no +> terminal window left open, no `npm install -g omniroute` to babysit.** + +Built with [Tauri v2](https://v2.tauri.app/) (a Rust core the size of a rounding error), it ships +its own signed Node 24 runtime and manages an app-owned OmniRoute install, so it never fights your +global `node`/`bun`. It **shares your existing `~/.omniroute/` config and database** — so it's the +same OmniRoute you already run, just with a hat on. 🎩 + + + + + + + + +
What it doesHow
🟢 Supervises the serverSpawns omniroute serve, adopts an already-running instance instead of duplicating it
📊 Live usage at a glanceProvider quota bars, Claude session/weekly limits with reset countdowns, 30-day cost breakdown
🔄 Auto-updates in placeStaged install, atomic swap, rollback on failure — always on the newest release
🚀 Start on loginOptional launch at login; tray-only, no dock icon
🩺 Doctor & logsOne-click diagnostics and server log access
+ +```sh +brew install --cask zoispag/tap/omniroute-tray +``` + +Prefer a download? Grab the latest .dmg from +Releases. Source, issues and build +docs live at zoispag/omniroute-tray. +
💛 A community project by @zoispag — not an official OmniRoute release.
+ +
+ +
+ ## 🔒 Private & Local-First
From 9af3ec5112f7315b0404b337dfd34909a63abb9e Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 10:49:01 -0300 Subject: [PATCH 030/143] feat(video): redact transcript in the in-memory pending-request snapshot (#12430 item 6) (#12596) * feat(video): redact transcript fields in the in-memory pending-request snapshot (#12430 item 6) trackPendingRequest (open-sse/handlers/chatCore.ts) stored the raw client body (with video transcript/audioTranscript cues) under `clientRequest`, live-exposed via /api/usage/call-logs (pendingDetails), /api/logs/[id] and /api/conversations while a request is in-flight. P2a redacted the persisted detailed-log snapshot but not this in-memory copy. Add redactPendingBody() to videoBridgeSnapshotRedaction.ts (sibling to logClientRawRequestRedacted from P2a): when videoBridgeObserved, returns the redacted clone from redactVideoTranscriptFieldsForLog; otherwise returns the exact same reference. Wire it into the trackPendingRequest call site (chatCore.ts:934), keeping the file within its frozen 5976-line budget (5971 -> 5974). * feat(video): substring-redact transcript in derived-prompt dispatch logs (#12430 item 4) Extend applyVideoBridgeLogRedaction with a string-content branch: pipeline-strategy stages, smart-auto-pipeline, and context-handoff summaries embed the transcript as a substring of a rendered prompt string rather than an exact array part, so the existing exact part-array match silently skipped them. Adds a mutually-exclusive string branch (Array.isArray vs typeof === "string") that does a replaceAll of the trusted fullText literal against a lazily cloned message, reusing the existing rootClone/clonedContainers/clonedMessages clone-on-write pattern so siblings keep original references and the input is never mutated. --- open-sse/handlers/chatCore.ts | 7 +- open-sse/handlers/chatCore/attemptLogging.ts | 51 +++++ .../videoBridgeSnapshotRedaction.ts | 14 ++ .../videoBridgeSnapshotRedaction.test.ts | 59 +++++- ...eo-bridge-derived-prompt-redaction.test.ts | 184 ++++++++++++++++++ 5 files changed, 312 insertions(+), 3 deletions(-) create mode 100644 tests/unit/video-bridge-derived-prompt-redaction.test.ts diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index d28711ce12..a50df1069b 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -360,7 +360,10 @@ import { deleteSessionAccountAffinity } from "@/lib/db/sessionAccountAffinity"; import { getCacheControlSettings } from "@/lib/cacheControlSettings"; import { guardrailRegistry } from "@/lib/guardrails"; import type { VideoBridgeLogRedactionEntry } from "@/lib/guardrails/videoBridge"; -import { logClientRawRequestRedacted } from "@/lib/guardrails/videoBridgeSnapshotRedaction"; +import { + logClientRawRequestRedacted, + redactPendingBody, +} from "@/lib/guardrails/videoBridgeSnapshotRedaction"; import { shouldPreserveCacheControl, resolveConnectionCacheOverride, @@ -928,7 +931,7 @@ export async function handleChatCore({ const pendingRequestId = trackPendingRequest(model, provider, pendingConnId, true, { clientEndpoint: clientRawRequest?.endpoint || "/v1/chat/completions", - clientRequest: clientRawRequest?.body ?? body, + clientRequest: redactPendingBody(clientRawRequest?.body ?? body, videoBridgeObserved), providerRequest: initialProviderRequest, stage: "registered", correlationId, diff --git a/open-sse/handlers/chatCore/attemptLogging.ts b/open-sse/handlers/chatCore/attemptLogging.ts index 2a583896a7..65a492a541 100644 --- a/open-sse/handlers/chatCore/attemptLogging.ts +++ b/open-sse/handlers/chatCore/attemptLogging.ts @@ -50,6 +50,16 @@ import { attachLogMeta } from "./cacheUsageMeta.ts"; * never touches a part whose text differs — see * `tests/unit/video-bridge-log-redaction.test.ts`'s "Scenario A" test for the * reproduction this fixes. + * + * #12430 item 4 (P2c): a message's `content` can also be a plain STRING that + * embeds `fullText` as a SUBSTRING rather than an exact array part — derived + * dispatches (pipeline-strategy stages, smart-auto-pipeline, context-handoff + * summaries) all interpolate the transcript blob into a larger rendered + * prompt string before calling `handleSingleModel`. That string branch is + * mutually exclusive with the array branch (a message's `content` is one or + * the other, never both) and uses `String.prototype.replaceAll` against the + * trusted `fullText` literal to swap every occurrence — see + * `tests/unit/video-bridge-derived-prompt-redaction.test.ts`. */ export function applyVideoBridgeLogRedaction( body: unknown, @@ -78,6 +88,47 @@ export function applyVideoBridgeLogRedaction( const originalMessage = originalContainer[messageIndex]; if (!originalMessage || typeof originalMessage !== "object") continue; const originalContent = (originalMessage as Record).content; + + // Derived-prompt dispatches (pipeline-strategy stages, smart-auto-pipeline, + // context-handoff summaries — #12430 item 4) embed the transcript as a + // SUBSTRING of a plain string `content`, e.g. a rendered stage prompt or a + // `{HISTORY}`-interpolated handoff summary, never as an exact array part. + // Mutually exclusive with the array branch below: a message's `content` + // is either a string or an array, never both, so this and the + // `Array.isArray` check never both match the same message. + if (typeof originalContent === "string") { + if (!originalContent.includes(fullText)) continue; + + // Same lazy clone-on-write as the array branch: root -> container + // array -> this message. Siblings keep referencing the originals. + if (!rootClone) rootClone = { ...source }; + let containerClone = clonedContainers.get(container); + if (!containerClone) { + containerClone = [...originalContainer]; + clonedContainers.set(container, containerClone); + rootClone[container] = containerClone; + } + + const messageKey = `${container}:${messageIndex}`; + let messageClone = clonedMessages.get(messageKey); + if (!messageClone) { + messageClone = { ...(originalMessage as Record) }; + clonedMessages.set(messageKey, messageClone); + containerClone[messageIndex] = messageClone; + } + + // Re-read from the (possibly already-cloned) message so a second + // redaction entry matching the same string content composes with the + // first instead of clobbering it. `fullText` is a trusted literal + // (the `[Video description:...]` blob), so replaceAll(string, string) + // needs no regex and is safe. replaceAll (not replace): a stage/summary + // prompt can quote the transcript back more than once. + const currentText = + typeof messageClone.content === "string" ? messageClone.content : originalContent; + messageClone.content = currentText.replaceAll(fullText, redactedText); + redacted = true; + continue; + } if (!Array.isArray(originalContent)) continue; for (let partIndex = 0; partIndex < originalContent.length; partIndex++) { diff --git a/src/lib/guardrails/videoBridgeSnapshotRedaction.ts b/src/lib/guardrails/videoBridgeSnapshotRedaction.ts index 8738b47cfc..8324b4be90 100644 --- a/src/lib/guardrails/videoBridgeSnapshotRedaction.ts +++ b/src/lib/guardrails/videoBridgeSnapshotRedaction.ts @@ -133,3 +133,17 @@ export function logClientRawRequestRedacted( clientRawRequest.headers ); } + +/** + * Call-site wrapper for the `clientRequest` field stored by `trackPendingRequest` + * (open-sse/handlers/chatCore.ts): the sibling in-memory leak to + * `logClientRawRequestRedacted` above — same raw body, but live-exposed via + * /api/usage/call-logs (pendingDetails), /api/logs/[id] and /api/conversations + * while the request is in-flight, not just in the persisted detailed-log + * snapshot. Identical observed/non-observed branching: a non-observed request + * keeps the exact same reference (no clone); an observed one gets the redacted + * clone. + */ +export function redactPendingBody(clientRequest: unknown, videoBridgeObserved: boolean): unknown { + return videoBridgeObserved ? redactVideoTranscriptFieldsForLog(clientRequest) : clientRequest; +} diff --git a/tests/unit/guardrails/videoBridgeSnapshotRedaction.test.ts b/tests/unit/guardrails/videoBridgeSnapshotRedaction.test.ts index 3f48bee7dc..8ca1dbab9f 100644 --- a/tests/unit/guardrails/videoBridgeSnapshotRedaction.test.ts +++ b/tests/unit/guardrails/videoBridgeSnapshotRedaction.test.ts @@ -7,7 +7,10 @@ import assert from "node:assert/strict"; import test from "node:test"; -import { redactVideoTranscriptFieldsForLog } from "../../../src/lib/guardrails/videoBridgeSnapshotRedaction.ts"; +import { + redactVideoTranscriptFieldsForLog, + redactPendingBody, +} from "../../../src/lib/guardrails/videoBridgeSnapshotRedaction.ts"; // Heavy import is fine here (test only, never in the production module under test) — used // solely to prove the local placeholder literal never drifts from the canonical P1 constant. import { VIDEO_TRANSCRIPT_REDACTION_PLACEHOLDER } from "../../../src/lib/guardrails/videoBridgeHelpers.ts"; @@ -205,3 +208,57 @@ test("the redaction placeholder matches the canonical P1 constant (no drift)", ( const part = contentAt(result, "messages", 0)[0]; assert.equal(part.transcript, VIDEO_TRANSCRIPT_REDACTION_PLACEHOLDER); }); + +// #12430 item 6 (P2c): the sibling in-memory leak. `trackPendingRequest` +// (open-sse/handlers/chatCore.ts) stores the same raw client body under +// `clientRequest`, live-exposed via /api/usage/call-logs (pendingDetails), +// /api/logs/[id] and /api/conversations while the request is in-flight. This +// helper is the guarded call-site wrapper chatCore.ts uses, mirroring +// logClientRawRequestRedacted's observed/non-observed branching. +test("redactPendingBody: observed=true delegates to redactVideoTranscriptFieldsForLog", () => { + const body = { + messages: [ + { + role: "user", + content: [ + { + type: "input_video", + video_url: "https://example.com/clip.mp4", + transcript: { cues: [{ text: "pending secret" }] }, + }, + ], + }, + ], + }; + + const result = redactPendingBody(body, true); + assert.notEqual( + result, + body, + "observed path must return a new structure, not the same reference" + ); + const part = contentAt(result, "messages", 0)[0]; + assert.equal(part.transcript, VIDEO_TRANSCRIPT_REDACTION_PLACEHOLDER); + assert.ok(!JSON.stringify(result).includes("pending secret")); + assert.deepEqual(result, redactVideoTranscriptFieldsForLog(body)); +}); + +test("redactPendingBody: observed=false returns the SAME reference unchanged", () => { + const body = { + messages: [ + { + role: "user", + content: [ + { + type: "input_video", + video_url: "https://example.com/clip.mp4", + transcript: { cues: [{ text: "not observed" }] }, + }, + ], + }, + ], + }; + + const result = redactPendingBody(body, false); + assert.equal(result, body, "non-observed path must return the exact same reference"); +}); diff --git a/tests/unit/video-bridge-derived-prompt-redaction.test.ts b/tests/unit/video-bridge-derived-prompt-redaction.test.ts new file mode 100644 index 0000000000..1e31e75ea1 --- /dev/null +++ b/tests/unit/video-bridge-derived-prompt-redaction.test.ts @@ -0,0 +1,184 @@ +// tests/unit/video-bridge-derived-prompt-redaction.test.ts +// P2c of #12150/#12430 (Video Bridge transcript retention — derived-prompt +// dispatch logs, item 4). +// +// Seam trace finding (decisive): videoBridgeLog is ALREADY threaded end-to-end +// to every nested handleChatCore — pipeline-strategy stages +// (src/domain/pipeline.ts::executeStage), smart-auto-pipeline, and +// context-handoff summaries (open-sse/services/contextHandoff.ts) — because +// all of them dispatch through the single P1b `handleSingleModel` closure and +// terminate in the SAME handleChatCore -> persistAttemptLogs -> +// applyVideoBridgeLogRedaction logging path. No plumbing/param changes were +// needed anywhere. +// +// The gap this file proves closed: those derived dispatches embed the +// transcript as a SUBSTRING of a plain STRING `content` message — +// `{ role: "user", content: }` — built by executeStage() +// (pipeline.ts:196-199 via prompts.ts interpolation) and by the +// context-handoff summary builders (contextHandoff.ts:415/729, `{HISTORY}` +// template substitution). Before this fix, applyVideoBridgeLogRedaction only +// matched ARRAY-content parts by exact text (`part.text === fullText`), so it +// silently skipped these string-content messages and the raw transcript +// persisted in the stage/summary sub-request call logs. +// +// This suite calls the real, already-exported `applyVideoBridgeLogRedaction` +// (open-sse/handlers/chatCore/attemptLogging.ts) directly — it is a pure +// function (no DB), so no persistAttemptLogs/DB harness is needed here; that +// integration-level proof already lives in +// tests/unit/video-bridge-log-redaction.test.ts. +import { test } from "node:test"; +import assert from "node:assert/strict"; + +import { applyVideoBridgeLogRedaction } from "../../open-sse/handlers/chatCore/attemptLogging.ts"; +import type { VideoBridgeLogRedactionEntry } from "../../src/lib/guardrails/videoBridge.ts"; + +const SECRET = "secret words"; +const FULL_TEXT = `[Video description: transcript[source=client] ${SECRET}]`; +const REDACTED_TEXT = "[Video description: transcript[source=client] [redacted-video-transcript]]"; + +function entry( + overrides: Partial = {} +): VideoBridgeLogRedactionEntry { + return { + container: "messages", + messageIndex: 0, + partIndex: 0, + fullText: FULL_TEXT, + redactedText: REDACTED_TEXT, + ...overrides, + }; +} + +test("derived-prompt (pipeline stage): a string-content message with the transcript embedded as a substring is redacted, secret absent, surrounding prompt text intact", () => { + const body = { + model: "openai/gpt-x", + messages: [ + { role: "system", content: "You are a summarization stage." }, + { + role: "user", + content: `Summarize the following context.\n\n${FULL_TEXT}\n\nEnd of context.`, + }, + ], + }; + + const result = applyVideoBridgeLogRedaction(body, [ + entry({ messageIndex: 1, partIndex: 0 }), + ]) as typeof body; + + const redactedContent = result.messages[1].content; + assert.equal( + redactedContent, + `Summarize the following context.\n\n${REDACTED_TEXT}\n\nEnd of context.` + ); + assert.ok(!redactedContent.includes(SECRET), "the raw transcript must not survive redaction"); + assert.ok( + redactedContent.startsWith("Summarize the following context.\n\n"), + "surrounding prompt text before the blob must stay intact" + ); + assert.ok( + redactedContent.endsWith("\n\nEnd of context."), + "surrounding prompt text after the blob must stay intact" + ); + assert.equal(JSON.stringify(result).includes(SECRET), false); +}); + +test("derived-prompt (context-handoff summary): input container string content is redacted the same way as messages", () => { + const body = { + model: "openai/gpt-x", + input: [ + { + role: "user", + content: `Continue the conversation given this history.\n\n${FULL_TEXT}\n\nContinue now.`, + }, + ], + }; + + const result = applyVideoBridgeLogRedaction(body, [ + entry({ container: "input", messageIndex: 0, partIndex: 0 }), + ]) as typeof body; + + const redactedContent = result.input[0].content; + assert.equal( + redactedContent, + `Continue the conversation given this history.\n\n${REDACTED_TEXT}\n\nContinue now.` + ); + assert.ok(!redactedContent.includes(SECRET)); + assert.equal(JSON.stringify(result).includes(SECRET), false); +}); + +test("multiple occurrences of fullText within the same string are ALL replaced (replaceAll, not replace)", () => { + const body = { + messages: [ + { + role: "user", + content: `First mention: ${FULL_TEXT}\n\nQuoted back for grounding: ${FULL_TEXT}\n\nDone.`, + }, + ], + }; + + const result = applyVideoBridgeLogRedaction(body, [entry({ messageIndex: 0, partIndex: 0 })]) as { + messages: Array<{ content: string }>; + }; + + const redactedContent = result.messages[0].content; + assert.equal( + redactedContent, + `First mention: ${REDACTED_TEXT}\n\nQuoted back for grounding: ${REDACTED_TEXT}\n\nDone.` + ); + assert.equal( + redactedContent.split(REDACTED_TEXT).length - 1, + 2, + "both occurrences must be replaced" + ); + assert.ok(!redactedContent.includes(SECRET)); +}); + +test("regression: the existing ARRAY-content exact-part-match path still redacts (no regression from the new string branch)", () => { + const body = { + messages: [ + { role: "system", content: "sys" }, + { + role: "user", + content: [ + { type: "text", text: "look at this video" }, + { type: "text", text: FULL_TEXT }, + ], + }, + ], + }; + + const result = applyVideoBridgeLogRedaction(body, [entry({ messageIndex: 1, partIndex: 1 })]) as { + messages: Array<{ content: unknown }>; + }; + + const content = result.messages[1].content as Array<{ text: string }>; + assert.equal(content[1].text, REDACTED_TEXT); + assert.ok(!content[1].text.includes(SECRET)); + assert.equal(content[0].text, "look at this video", "sibling part must stay untouched"); +}); + +test("no mutation of the input object: the caller's body is byte-identical after redaction (string-content path)", () => { + const body = { + messages: [{ role: "user", content: `before ${FULL_TEXT} after` }], + }; + const snapshotBefore = JSON.parse(JSON.stringify(body)); + + applyVideoBridgeLogRedaction(body, [entry({ messageIndex: 0, partIndex: 0 })]); + + assert.deepEqual(body, snapshotBefore, "the original body must never be mutated"); +}); + +test("non-matching string content is returned unchanged, with the SAME root reference (nothing redacted -> no clone allocated)", () => { + const body = { + messages: [{ role: "user", content: "nothing to see here, no transcript blob at all" }], + }; + + const result = applyVideoBridgeLogRedaction(body, [entry({ messageIndex: 0, partIndex: 0 })]); + + assert.equal( + result, + body, + "when no fullText matches, the exact same object reference is returned" + ); +}); From 9ddb8e0a932473bff96c7979a827019e46c2854b Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 11:33:07 -0300 Subject: [PATCH 031/143] =?UTF-8?q?fix(docs):=20restore=20the=20Next=20bui?= =?UTF-8?q?ld=20=E2=80=94=20REMOVED=5FPROVIDERS.md=20had=20no=20frontmatte?= =?UTF-8?q?r=20(base-red=20#12581)=20(#12610)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `source.config.ts` feeds `docs/reference/**/*.md` to fumadocs-mdx, whose default schema requires a `title`. #12478 added `docs/reference/REMOVED_PROVIDERS.md` with no frontmatter block at all, so every production build died with: [MDX] invalid frontmatter in docs/reference/REMOVED_PROVIDERS.md: - title: Invalid input: expected string, received undefined That single missing block is what turns three release-green gates red at once — `Package artifact (npm pack policy)` fails on the build, and both `Tarball boot-smoke` and the packaged CLI checks are skipped for lack of a valid `dist/`. Fixes: - add the frontmatter block, matching the convention of its sibling reference docs (`title` / `version` / `lastUpdated`). - add `check:docs-frontmatter`, wired into `check:docs-all`, so the next doc added without a title fails in milliseconds instead of costing a full Next build and a red release branch. The gate reads its globs from `source.config.ts` rather than duplicating them, so a new docs directory cannot silently escape the check. Verified: the gate reports OK across all 122 compiled docs, fails (exit 1) when the frontmatter is removed, and `npm run check:docs-all` passes. --- docs/reference/REMOVED_PROVIDERS.md | 6 ++ package.json | 3 +- scripts/check/check-docs-frontmatter.mjs | 103 +++++++++++++++++++++++ 3 files changed, 111 insertions(+), 1 deletion(-) create mode 100644 scripts/check/check-docs-frontmatter.mjs diff --git a/docs/reference/REMOVED_PROVIDERS.md b/docs/reference/REMOVED_PROVIDERS.md index e03283bd43..24c0aab7d4 100644 --- a/docs/reference/REMOVED_PROVIDERS.md +++ b/docs/reference/REMOVED_PROVIDERS.md @@ -1,3 +1,9 @@ +--- +title: "Removed Providers" +version: 3.8.51 +lastUpdated: 2026-09-03 +--- + # Providers removed at their operator's request Some services were integrated into OmniRoute and later removed because the people who run diff --git a/package.json b/package.json index bbfee22bc7..bc63455358 100644 --- a/package.json +++ b/package.json @@ -151,7 +151,8 @@ "check:router-eval": "node --import tsx scripts/check/check-router-eval-regression.ts", "check:doc-links": "node scripts/check/check-doc-links.mjs", "check:fabricated-docs": "node scripts/check/check-fabricated-docs.mjs --strict", - "check:docs-all": "npm run check:docs-sync && npm run check:docs-counts && npm run check:env-doc-sync && npm run check:deprecated-versions && npm run check:doc-links && npm run check:fabricated-docs", + "check:docs-frontmatter": "node scripts/check/check-docs-frontmatter.mjs", + "check:docs-all": "npm run check:docs-sync && npm run check:docs-frontmatter && npm run check:docs-counts && npm run check:env-doc-sync && npm run check:deprecated-versions && npm run check:doc-links && npm run check:fabricated-docs", "docs:render-diagrams": "node scripts/docs/render-diagrams.mjs", "i18n:run": "node scripts/i18n/run-translation.mjs", "i18n:run:dry": "node scripts/i18n/run-translation.mjs --dry-run", diff --git a/scripts/check/check-docs-frontmatter.mjs b/scripts/check/check-docs-frontmatter.mjs new file mode 100644 index 0000000000..a00751f880 --- /dev/null +++ b/scripts/check/check-docs-frontmatter.mjs @@ -0,0 +1,103 @@ +#!/usr/bin/env node +/** + * Validates the frontmatter of every Markdown file that fumadocs-mdx compiles. + * + * Why this gate exists: `source.config.ts` feeds `docs/**` globs to + * `defineDocs()`, and fumadocs' default frontmatter schema REQUIRES a `title` + * string. A doc added without frontmatter does not fail any docs gate — it + * fails the **production build** with a generic Turbopack error + * (`[MDX] invalid frontmatter … title: Invalid input: expected string, + * received undefined`), which then cascades into `check:pack-artifact` and the + * tarball boot-smoke. That is exactly how #12478 turned the release branch red + * (base-red #12581): one new reference doc, no frontmatter, three failing + * gates and an unbuildable branch. + * + * Catching it here costs milliseconds instead of a full Next build. + * + * The globs are read from `source.config.ts` rather than duplicated, so adding + * a new docs directory there cannot silently escape this check. + */ + +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", ".."); +const CONFIG_PATH = path.join(ROOT, "source.config.ts"); + +/** Extract the `files: [...]` globs declared in source.config.ts. */ +function readConfiguredGlobs() { + const src = fs.readFileSync(CONFIG_PATH, "utf-8"); + const block = src.match(/files\s*:\s*\[([\s\S]*?)\]/); + if (!block) { + console.error( + "[docs-frontmatter] FAIL — could not locate the `files:` globs in source.config.ts" + ); + process.exit(1); + } + const globs = [...block[1].matchAll(/["'`]([^"'`]+)["'`]/g)].map((m) => m[1]); + if (globs.length === 0) { + console.error("[docs-frontmatter] FAIL — source.config.ts declares no doc globs"); + process.exit(1); + } + return globs; +} + +/** "./reference/**\/*.md" -> the directory under docs/ it covers. */ +function globToDir(glob) { + const cleaned = glob.replace(/^\.\//, ""); + const dir = cleaned.split("/**")[0]; + return path.join(ROOT, "docs", dir); +} + +function walkMarkdown(dir) { + if (!fs.existsSync(dir)) return []; + const out = []; + for (const entry of fs.readdirSync(dir, { withFileTypes: true })) { + const full = path.join(dir, entry.name); + if (entry.isDirectory()) out.push(...walkMarkdown(full)); + else if (entry.isFile() && entry.name.endsWith(".md")) out.push(full); + } + return out; +} + +const violations = []; +const files = [...new Set(readConfiguredGlobs().flatMap((g) => walkMarkdown(globToDir(g))))]; + +for (const file of files) { + const rel = path.relative(ROOT, file); + const text = fs.readFileSync(file, "utf-8"); + + if (!text.startsWith("---")) { + violations.push(`${rel}: no frontmatter block (fumadocs requires a \`title\`)`); + continue; + } + const end = text.indexOf("\n---", 3); + if (end === -1) { + violations.push(`${rel}: frontmatter block is never closed`); + continue; + } + const frontmatter = text.slice(3, end); + const title = frontmatter.match(/^\s*title\s*:\s*(.+)$/m); + if (!title) { + violations.push(`${rel}: frontmatter has no \`title\``); + } else if (title[1].trim().replace(/^["']|["']$/g, "") === "") { + violations.push(`${rel}: \`title\` is empty`); + } +} + +if (violations.length > 0) { + console.error( + `[docs-frontmatter] FAIL — ${violations.length} doc(s) would break the Next build:` + ); + for (const v of violations) console.error(` - ${v}`); + console.error( + "\nEvery Markdown file matched by source.config.ts is compiled by fumadocs-mdx and needs a\n" + + 'frontmatter block with a title, e.g.:\n\n---\ntitle: "Removed Providers"\nversion: 3.8.51\nlastUpdated: 2026-09-03\n---\n' + ); + process.exit(1); +} + +console.log( + `[docs-frontmatter] OK — ${files.length} compiled doc(s) carry a valid frontmatter title.` +); From c9fb06e26ca78b4cdcee515d5cee364e2119b76e Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Thu, 3 Sep 2026 11:37:54 -0400 Subject: [PATCH 032/143] fix(grok-cli): treat omitted SuperGrokPro creditUsagePercent as 0% (#12312) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check:provider-consistency` OK (273 entradas REGISTRY, **356** providers canônicos), `check-docs-counts-sync` exit 0 e **300/300** nos testes que a leva toca. O crescimento de arquivo que os PRs empilham uns sobre os outros foi rebaselinado num único registro datado (`_rebaseline_2026_09_03_houminxi_batch`), com a decomposição por arquivo: `providers/page.tsx` +18 (import CSV do #12504 + busca do #12495 no mesmo painel), `accountFallback.ts` +6 (o #12566 sobre o rebaseline que o #12590 já registrou — os dois tocam `checkFallbackError`) e `chatCore.ts` +3 (invalidez de cache de quota no 429 do #12325). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. --- .../fixes/12312-grok-cli-supergrok-quota.md | 1 + open-sse/services/usage/grokCli.ts | 9 +- tests/unit/grok-cli-provider-limits.test.ts | 92 +++++++++++++++++-- 3 files changed, 92 insertions(+), 10 deletions(-) create mode 100644 changelog.d/fixes/12312-grok-cli-supergrok-quota.md diff --git a/changelog.d/fixes/12312-grok-cli-supergrok-quota.md b/changelog.d/fixes/12312-grok-cli-supergrok-quota.md new file mode 100644 index 0000000000..1fed5df39e --- /dev/null +++ b/changelog.d/fixes/12312-grok-cli-supergrok-quota.md @@ -0,0 +1 @@ +- **fix(grok-cli):** treat omitted SuperGrokPro `creditUsagePercent` as 0% used so Provider Limits still renders a weekly bar (proto3 zero-elision) ([#12312](https://github.com/diegosouzapw/OmniRoute/pull/12312)) — thanks @HouMinXi diff --git a/open-sse/services/usage/grokCli.ts b/open-sse/services/usage/grokCli.ts index 08396cfb2f..bbd76199b8 100644 --- a/open-sse/services/usage/grokCli.ts +++ b/open-sse/services/usage/grokCli.ts @@ -239,9 +239,12 @@ export async function getGrokCliUsage(accessToken?: string) { const config = billing.config; const resetAt = config.currentPeriod?.end || null; const quotas: Record> = {}; - if (config.creditUsagePercent != null) { - quotas.weekly = percentageQuota(config.creditUsagePercent, resetAt); - } + // SuperGrokPro (and proto3 omit-zero) billing configs often omit + // creditUsagePercent / productUsage. A present config object is a + // successful billing read, so treat a missing percent as 0% used and + // still render a weekly bar. A missing config still returns + // "Grok Build billing status unavailable" above — that path is unchanged. + quotas.weekly = percentageQuota(config.creditUsagePercent ?? 0, resetAt); Object.assign(quotas, buildProductQuotas(config.productUsage, resetAt)); const autoTopUpResponse = userId diff --git a/tests/unit/grok-cli-provider-limits.test.ts b/tests/unit/grok-cli-provider-limits.test.ts index 081d403f5f..72e7c2b2da 100644 --- a/tests/unit/grok-cli-provider-limits.test.ts +++ b/tests/unit/grok-cli-provider-limits.test.ts @@ -38,6 +38,10 @@ function successFixtures( userId?: unknown; prepaidBalance?: Record | null | undefined; productUsage?: unknown; + creditUsagePercent?: number | null; + omitCreditUsagePercent?: boolean; + omitProductUsage?: boolean; + currentPeriod?: Record | null; } = {} ) { const tier = "tier" in options ? options.tier : "SuperGrok Heavy"; @@ -51,6 +55,14 @@ function successFixtures( { product: "API", usagePercent: 12.5 }, { product: "Grok Code", usagePercent: 44 }, ]; + const currentPeriod = + "currentPeriod" in options + ? options.currentPeriod + : { + type: "WEEKLY", + start: "2026-07-27T00:00:00.000Z", + end: "2026-08-03T00:00:00.000Z", + }; return async (input: string | URL | Request) => { const url = String(input); @@ -64,13 +76,14 @@ function successFixtures( if (url.endsWith("/billing?format=credits")) { return response({ config: { - creditUsagePercent: 37.25, - currentPeriod: { - type: "WEEKLY", - start: "2026-07-27T00:00:00.000Z", - end: "2026-08-03T00:00:00.000Z", - }, - productUsage, + ...(options.omitCreditUsagePercent + ? {} + : { + creditUsagePercent: + "creditUsagePercent" in options ? options.creditUsagePercent : 37.25, + }), + ...(currentPeriod === undefined ? {} : { currentPeriod }), + ...(options.omitProductUsage ? {} : { productUsage }), ...(prepaidBalance === undefined ? {} : { prepaidBalance }), }, }); @@ -492,3 +505,68 @@ test("Provider Limits cache persists only the public Grok billing contract", () test("grok-cli is registered on the public Provider Limits usage seam", () => { assert.ok((USAGE_FETCHER_PROVIDERS as readonly string[]).includes("grok-cli")); }); + +test("SuperGrokPro omitted creditUsagePercent still yields a weekly quota bar", async () => { + const usage = await getUsage( + successFixtures({ + tier: "SuperGrokPro", + omitCreditUsagePercent: true, + omitProductUsage: true, + prepaidBalance: { val: 0 }, + }) as typeof fetch + ); + + assert.equal(usage.plan, "SuperGrokPro"); + assert.deepEqual(usage.quotas?.weekly, { + used: 0, + total: 100, + remaining: 100, + remainingPercentage: 100, + resetAt: "2026-08-03T00:00:00.000Z", + isPercentageOnly: true, + }); + assert.equal(usage.message, undefined); +}); + +test("SuperGrokPro explicit null creditUsagePercent still yields a weekly quota bar", async () => { + const usage = await getUsage( + successFixtures({ + tier: "SuperGrokPro", + creditUsagePercent: null, + omitProductUsage: true, + prepaidBalance: { val: 0 }, + }) as typeof fetch + ); + + assert.equal(usage.plan, "SuperGrokPro"); + assert.deepEqual(usage.quotas?.weekly, { + used: 0, + total: 100, + remaining: 100, + remainingPercentage: 100, + resetAt: "2026-08-03T00:00:00.000Z", + isPercentageOnly: true, + }); +}); + +test("SuperGrokPro omitted currentPeriod still yields a weekly bar with null resetAt", async () => { + const usage = await getUsage( + successFixtures({ + tier: "SuperGrokPro", + omitCreditUsagePercent: true, + omitProductUsage: true, + currentPeriod: null, + prepaidBalance: { val: 0 }, + }) as typeof fetch + ); + + assert.equal(usage.plan, "SuperGrokPro"); + assert.deepEqual(usage.quotas?.weekly, { + used: 0, + total: 100, + remaining: 100, + remainingPercentage: 100, + resetAt: null, + isPercentageOnly: true, + }); +}); From 0f5fc78d8acbaa6d178b1e2d22e7e8badcedafa6 Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Thu, 3 Sep 2026 11:38:16 -0400 Subject: [PATCH 033/143] feat(providers): search connections by name and baseUrl (#12495) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check:provider-consistency` OK (273 entradas REGISTRY, **356** providers canônicos), `check-docs-counts-sync` exit 0 e **300/300** nos testes que a leva toca. O crescimento de arquivo que os PRs empilham uns sobre os outros foi rebaselinado num único registro datado (`_rebaseline_2026_09_03_houminxi_batch`), com a decomposição por arquivo: `providers/page.tsx` +18 (import CSV do #12504 + busca do #12495 no mesmo painel), `accountFallback.ts` +6 (o #12566 sobre o rebaseline que o #12590 já registrou — os dois tocam `checkFallbackError`) e `chatCore.ts` +3 (invalidez de cache de quota no 429 do #12325). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. --- .../12108-provider-search-name-baseurl.md | 1 + .../providers/[id]/connectionsSearchFilter.ts | 9 +- .../(dashboard)/dashboard/providers/page.tsx | 54 +++++++---- .../dashboard/providers/providerPageUtils.ts | 43 ++++++++- ...r-search-connection-identity-12108.test.ts | 92 +++++++++++++++++++ .../unit/ui/connectionsSearchFilter.test.tsx | 19 ++++ 6 files changed, 197 insertions(+), 21 deletions(-) create mode 100644 changelog.d/features/12108-provider-search-name-baseurl.md create mode 100644 tests/unit/provider-search-connection-identity-12108.test.ts diff --git a/changelog.d/features/12108-provider-search-name-baseurl.md b/changelog.d/features/12108-provider-search-name-baseurl.md new file mode 100644 index 0000000000..a13f2a12a1 --- /dev/null +++ b/changelog.d/features/12108-provider-search-name-baseurl.md @@ -0,0 +1 @@ +- **feat(providers):** dashboard search matches connection name and `baseUrl` so imported OpenAI-compat nodes surface on the provider card ([#12108](https://github.com/diegosouzapw/OmniRoute/issues/12108)) diff --git a/src/app/(dashboard)/dashboard/providers/[id]/connectionsSearchFilter.ts b/src/app/(dashboard)/dashboard/providers/[id]/connectionsSearchFilter.ts index 8c676c74be..6c7209b1aa 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/connectionsSearchFilter.ts +++ b/src/app/(dashboard)/dashboard/providers/[id]/connectionsSearchFilter.ts @@ -4,7 +4,8 @@ * * Case-insensitive, plain SUBSTRING match (mirrors the semantics of * `src/shared/utils/modelCatalogSearch.ts` — do not reimplement a fuzzy - * matcher here). Matches against id, tag, name, and email. + * matcher here). Matches against id, tag, name, email, and + * providerSpecificData.baseUrl (#12108). */ import type { ConnectionRowConnection } from "./components/ConnectionRow"; @@ -17,6 +18,11 @@ function getConnectionTag(conn: ConnectionRowConnection): string { return typeof tag === "string" ? tag : ""; } +function getConnectionBaseUrl(conn: ConnectionRowConnection): string { + const baseUrl = conn.providerSpecificData?.baseUrl; + return typeof baseUrl === "string" ? baseUrl : ""; +} + /** True when `conn` matches `query` (empty/whitespace query always matches). */ export function matchesAccountQuery(query: string, conn: ConnectionRowConnection): boolean { const normalizedQuery = normalize(query); @@ -27,6 +33,7 @@ export function matchesAccountQuery(query: string, conn: ConnectionRowConnection normalize(getConnectionTag(conn)), normalize(conn.name), normalize(conn.email), + normalize(getConnectionBaseUrl(conn)), ]; return haystacks.some((haystack) => haystack.includes(normalizedQuery)); } diff --git a/src/app/(dashboard)/dashboard/providers/page.tsx b/src/app/(dashboard)/dashboard/providers/page.tsx index a65478dd2a..a9cde7c6f7 100644 --- a/src/app/(dashboard)/dashboard/providers/page.tsx +++ b/src/app/(dashboard)/dashboard/providers/page.tsx @@ -558,7 +558,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const rawNoAuthEntriesAll = buildStaticProviderEntries("no-auth", getProviderStats); @@ -576,7 +577,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const apiKeyProviderEntriesAll = buildStaticProviderEntries("apikey", getProviderStats); @@ -595,7 +597,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const aggregatorProviderEntriesAll = apiKeyProviderEntriesAll.filter((entry) => AGGREGATOR_PROVIDER_IDS.has(entry.providerId) @@ -607,7 +610,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const imageProviderEntriesAll = apiKeyProviderEntriesAll.filter((entry) => IMAGE_ONLY_PROVIDER_IDS.has(entry.providerId) @@ -619,7 +623,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const enterpriseProviderEntriesAll = apiKeyProviderEntriesAll.filter((entry) => ENTERPRISE_CLOUD_PROVIDER_IDS.has(entry.providerId) @@ -631,7 +636,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const videoProviderEntriesAll = apiKeyProviderEntriesAll.filter((entry) => VIDEO_PROVIDER_IDS.has(entry.providerId) @@ -643,7 +649,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const embeddingRerankProviderEntriesAll = apiKeyProviderEntriesAll.filter((entry) => EMBEDDING_RERANK_PROVIDER_IDS.has(entry.providerId) @@ -655,7 +662,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const webCookieProviderEntriesAll = buildStaticProviderEntries("web-cookie", getProviderStats); @@ -666,7 +674,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const localProviderEntriesAll = buildStaticProviderEntries("local", getProviderStats); @@ -677,7 +686,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const searchProviderEntriesAll = buildStaticProviderEntries("search", getProviderStats); @@ -688,7 +698,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const audioProviderEntriesAll = buildStaticProviderEntries("audio", getProviderStats); @@ -699,7 +710,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const cloudAgentProviderEntriesAll = buildStaticProviderEntries("cloud-agent", getProviderStats); @@ -710,7 +722,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const upstreamProxyEntriesAll = buildStaticProviderEntries("upstream-proxy", getProviderStats); @@ -721,7 +734,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const compatibleProviderEntriesAll = [ @@ -754,7 +768,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const staticProviderEntriesAll = dedupeProviderEntries([ @@ -780,7 +795,8 @@ function ProvidersPageContent() { undefined, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); // IDE providers: subset of oauth/apikey providers that are editors/IDEs with @@ -796,7 +812,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const oauthOnlyEntriesAll = oauthProviderEntriesAll @@ -817,7 +834,8 @@ function ProvidersPageContent() { showFreeOnly, modelSearchQuery, activeServiceKind, - liveModelsByProviderId + liveModelsByProviderId, + connections ); const compactProviderEntries = buildCompactProviderEntriesForPage({ diff --git a/src/app/(dashboard)/dashboard/providers/providerPageUtils.ts b/src/app/(dashboard)/dashboard/providers/providerPageUtils.ts index 8c7be96c71..8356a1f9e5 100644 --- a/src/app/(dashboard)/dashboard/providers/providerPageUtils.ts +++ b/src/app/(dashboard)/dashboard/providers/providerPageUtils.ts @@ -421,6 +421,27 @@ function getFilterableModelsForEntry( return [...staticModels, ...liveModels]; } +/** + * Dashboard-card search identity for an imported connection (#12108). + * Only `name` and `providerSpecificData.baseUrl` — those are the two + * fields the issue asked for. id/tag/email stay on the detail-page + * haystack (`matchesAccountQuery`); surfacing a provider card from an + * account email would mix account-picker UX into the catalog filter. + */ +export type ProviderSearchConnection = { + provider?: string | null; + name?: string | null; + providerSpecificData?: Record | null; +}; + +function connectionSearchHaystacks(conn: ProviderSearchConnection): string[] { + const baseUrl = conn.providerSpecificData?.baseUrl; + return [ + typeof conn.name === "string" ? conn.name : "", + typeof baseUrl === "string" ? baseUrl : "", + ]; +} + export function filterConfiguredProviderEntries( entries: ProviderEntry[], showConfiguredOnly: boolean, @@ -428,7 +449,8 @@ export function filterConfiguredProviderEntries( showFreeOnly?: boolean, modelSearchQuery?: string, serviceKindFilter?: string | null, - liveModelsByProviderId?: LiveModelsByProviderId + liveModelsByProviderId?: LiveModelsByProviderId, + connections?: ProviderSearchConnection[] ): ProviderEntry[] { let filtered = entries; @@ -461,9 +483,26 @@ export function filterConfiguredProviderEntries( if (searchQuery && searchQuery.trim()) { filtered = filtered.filter((entry) => { const provider = entry.provider as Record; - return ( + if ( matchesAnyToken(String(provider.name || ""), searchQuery) || matchesAnyToken(entry.providerId, searchQuery) + ) { + return true; + } + // #12108: imported connections live under the canonical provider card. + // Match their operator-visible name / baseUrl so "Grade-S-Node" or an + // IP in the search box surfaces the OpenAI card instead of vanishing. + // Same matcher as provider.name / providerId above (matchesAnyToken: + // full-string first, then whitespace-token OR). The detail page uses + // a single-substring haystack — that is a different surface, not a + // bug in this filter. + if (!connections || connections.length === 0) return false; + return connections.some( + (conn) => + connectionBelongsToProviderPage(conn.provider, entry.providerId) && + connectionSearchHaystacks(conn).some((haystack) => + matchesAnyToken(haystack, searchQuery) + ) ); }); } diff --git a/tests/unit/provider-search-connection-identity-12108.test.ts b/tests/unit/provider-search-connection-identity-12108.test.ts new file mode 100644 index 0000000000..e6d8d5a556 --- /dev/null +++ b/tests/unit/provider-search-connection-identity-12108.test.ts @@ -0,0 +1,92 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +const { filterConfiguredProviderEntries } = await import( + "../../src/app/(dashboard)/dashboard/providers/providerPageUtils.ts" +); + +const ENTRIES = [ + { + providerId: "openai", + provider: { id: "openai", name: "OpenAI" }, + stats: { total: 1 }, + displayAuthType: "apikey" as const, + toggleAuthType: "apikey" as const, + }, + { + providerId: "claude", + provider: { id: "claude", name: "Claude" }, + stats: { total: 0 }, + displayAuthType: "oauth" as const, + toggleAuthType: "oauth" as const, + }, +]; + +const CONNECTIONS = [ + { + provider: "openai", + name: "Grade-S-Node", + providerSpecificData: { baseUrl: "http://145.10.20.30:8080" }, + }, +]; + +function ids(query: string, connections = CONNECTIONS) { + return filterConfiguredProviderEntries( + ENTRIES, + false, + query, + false, + "", + null, + undefined, + connections + ).map((e) => e.providerId); +} + +test("#12108 top-level search matches connection name (imported Grade-S-Node)", () => { + assert.deepEqual(ids("Grade-S-Node"), ["openai"]); +}); + +test("#12108 top-level search matches connection baseUrl host", () => { + assert.deepEqual(ids("145.10.20.30"), ["openai"]); +}); + +test("#12108 top-level search still matches static provider name", () => { + assert.deepEqual(ids("claude"), ["claude"]); +}); + +test("#12108 top-level search without connections does not invent a name match", () => { + assert.deepEqual(ids("Grade-S-Node", []), []); + const withoutArg = filterConfiguredProviderEntries(ENTRIES, false, "Grade-S-Node").map( + (e) => e.providerId + ); + assert.deepEqual(withoutArg, []); +}); + +test("#12108 empty search still returns every entry", () => { + assert.deepEqual(new Set(ids("")), new Set(["openai", "claude"])); +}); + +test("#12108 a connection on openai does not surface claude", () => { + assert.equal(ids("Grade-S-Node").includes("claude"), false); +}); + +test("#12108 dashboard card search does not match connection email/tag/id", () => { + const withAccountFields = [ + { + provider: "openai", + name: "Grade-S-Node", + id: "conn-grade", + email: "ops@grade.example", + providerSpecificData: { tag: "prod-east", baseUrl: "http://145.10.20.30:8080" }, + }, + ]; + assert.deepEqual(ids("ops@grade.example", withAccountFields), []); + assert.deepEqual(ids("prod-east", withAccountFields), []); + assert.deepEqual(ids("conn-grade", withAccountFields), []); + assert.deepEqual(ids("Grade-S-Node", withAccountFields), ["openai"]); +}); + +test("#12108 connection haystack uses matchesAnyToken (token OR, same as provider.name)", () => { + assert.deepEqual(ids("Grade Node"), ["openai"]); +}); diff --git a/tests/unit/ui/connectionsSearchFilter.test.tsx b/tests/unit/ui/connectionsSearchFilter.test.tsx index 6885c1a91c..3b2e9ab7ce 100644 --- a/tests/unit/ui/connectionsSearchFilter.test.tsx +++ b/tests/unit/ui/connectionsSearchFilter.test.tsx @@ -35,6 +35,11 @@ const CONNECTIONS: ConnectionRowConnection[] = [ { id: "conn-2", name: "Bob", email: "bob@example.com", providerSpecificData: { tag: "staging" } }, { id: "conn-3", name: "Carol", email: "carol@gmail.com" }, { id: "special-id-9", name: undefined, email: undefined }, + { + id: "conn-grade", + name: "Grade-S-Node", + providerSpecificData: { tag: "relay", baseUrl: "http://145.10.20.30:8080" }, + }, ]; describe("matchesAccountQuery / filterConnectionsByQuery — #7937", () => { @@ -74,6 +79,20 @@ describe("matchesAccountQuery / filterConnectionsByQuery — #7937", () => { it("does not match a connection missing the queried field", () => { expect(matchesAccountQuery("anything", CONNECTIONS[3])).toBe(false); }); + + // #12108 — detail-page search must also match providerSpecificData.baseUrl + // (import stores the override there; id/tag/name/email never contain the host). + it("matches providerSpecificData.baseUrl by host substring (#12108)", () => { + expect(matchesAccountQuery("145.10.20.30", CONNECTIONS[4])).toBe(true); + expect(matchesAccountQuery("145.10.20.30", CONNECTIONS[0])).toBe(false); + expect(filterConnectionsByQuery("145.10.20.30", CONNECTIONS).map((c) => c.id)).toEqual([ + "conn-grade", + ]); + }); + + it("matches providerSpecificData.baseUrl case-insensitively (#12108)", () => { + expect(matchesAccountQuery("HTTP://145.10.20.30:8080", CONNECTIONS[4])).toBe(true); + }); }); // --------------------------------------------------------------------------- From 52456a1cea2f0bdbeae7af2f1c6410f6e47495f5 Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Thu, 3 Sep 2026 11:38:39 -0400 Subject: [PATCH 034/143] fix(quota): drop generic quota cache on upstream 429 (#12325) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check:provider-consistency` OK (273 entradas REGISTRY, **356** providers canônicos), `check-docs-counts-sync` exit 0 e **300/300** nos testes que a leva toca. O crescimento de arquivo que os PRs empilham uns sobre os outros foi rebaselinado num único registro datado (`_rebaseline_2026_09_03_houminxi_batch`), com a decomposição por arquivo: `providers/page.tsx` +18 (import CSV do #12504 + busca do #12495 no mesmo painel), `accountFallback.ts` +6 (o #12566 sobre o rebaseline que o #12590 já registrou — os dois tocam `checkFallbackError`) e `chatCore.ts` +3 (invalidez de cache de quota no 429 do #12325). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. --- .../12325-generic-quota-429-invalidate.md | 1 + config/quality/file-size-baseline.json | 6 +- open-sse/handlers/chatCore.ts | 9 + open-sse/services/genericQuotaFetcher.ts | 169 ++++++++++- .../antigravity-429-quota-cooldown.test.ts | 5 + tests/unit/generic-quota-fetcher.test.ts | 276 +++++++++++++++++- 6 files changed, 453 insertions(+), 13 deletions(-) create mode 100644 changelog.d/fixes/12325-generic-quota-429-invalidate.md diff --git a/changelog.d/fixes/12325-generic-quota-429-invalidate.md b/changelog.d/fixes/12325-generic-quota-429-invalidate.md new file mode 100644 index 0000000000..753e96027d --- /dev/null +++ b/changelog.d/fixes/12325-generic-quota-429-invalidate.md @@ -0,0 +1 @@ +- **fix(quota):** drop the generic quota cache (agy / Antigravity / Claude OAuth) on an upstream 429 so reset-aware scoring does not keep a 60s stale snapshot, and force-refresh the next usage fetch so inner provider caches cannot recache the same window ([#12325](https://github.com/diegosouzapw/OmniRoute/pull/12325)) — thanks @HouMinXi diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index c4e3bd4361..ffed35fbc1 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -1,4 +1,5 @@ { + "_rebaseline_2026_09_02_12325_generic_429_invalidate": "PR #12325 own growth: open-sse/handlers/chatCore.ts 5946->5955 (+9 = the non-Codex 429 else-if that drops the generic quota wrapper and stamps force-refresh, plus a source-regex breadcrumb). Irreducible call-site wiring next to the existing Codex 429 invalidateCodexQuotaCache branch; not extractable without splitting handleChatCore mid-response. Covered by tests/unit/generic-quota-fetcher.test.ts (31/31) and tests/unit/antigravity-429-quota-cooldown.test.ts.", "_rebaseline_2026_09_02_12429_wreq_migration_suite": "PR #12429 (wreq-js web-cookie transport): new test file tests/unit/tls-client-wreq-migration.test.ts at 1374 lines, above the 1200 new-file testCap. Frozen rather than split: it is the single cohesive regression suite for the transport migration (31 cases covering streaming, fragmented EOF sentinels, proxy isolation, first-byte and hard deadlines, binary responses and cancellation), and the cases share the native-transport harness the file sets up once. Splitting it during a merge would duplicate that harness across files for no coverage gain. Entered at the exact LOC, so it can only ratchet down from here.", "_rebaseline_2026_09_02_12239_chatgpt_web_cleanroom": "PR #12239 (backryun, codex/restore-chatgpt-web-cleanroom) own growth at the two existing chat chokepoints for the clean-room ChatGPT Web transport: src/sse/handlers/chat.ts 2384->2424 (+40); open-sse/handlers/chatCore.ts 5946->5976 (+30). Additive dispatch wiring; the retirement guard is narrowed to the GPL-derived cgpt-web alias rather than removed, so #11754's provenance decision still holds for the old implementation. Same own-growth rationale as _rebaseline_2026_08_20_10531_freebuff_provider.", "_rebaseline_2026_09_02_12412_grok_web_prettier": "PR #12412 (repository Prettier style applied to tests/unit/grok-web.test.ts): the reformat expands the file +277 lines (2436 -> 2713) with an identical parsed AST — no production code, no assertion changes. Cap set to 2985 rather than the exact 2713 on the operator's instruction (2026-09-02): ~10% headroom so routine additions to this suite do not re-trip the gate on formatting alone. Previous cap 2437. This is a deliberate exception to the down-only ratchet for one reformatted test file; every other entry keeps the #12411 tightening.", @@ -412,7 +413,7 @@ "open-sse/executors/codex.ts": 1499, "open-sse/executors/cursor.ts": 1759, "open-sse/executors/muse-spark-web.ts": 1405, - "open-sse/handlers/chatCore.ts": 5976, + "open-sse/handlers/chatCore.ts": 5981, "open-sse/handlers/imageGeneration.ts": 3259, "open-sse/handlers/search.ts": 1789, "open-sse/mcp-server/schemas/tools.ts": 1621, @@ -629,5 +630,6 @@ "_rebaseline_2026_06_30_v3842_release_chatgptweb_compression": "v3.8.42 cycle-close file-size reconciliation (DRIFT measured OK on each PR's base, stacked above frozen at the merge tip; fast-path PR->release/** does not run check:file-size). (1) open-sse/executors/chatgpt-web.ts 2870->3206 (+336 = #5531 portable SHA3-512 sentinel-PoW wiring with the native-vs-fallback digest path + #5536 GPT-5.5 Pro handoff branch; the pure Keccak-f[1600] fallback itself already lives in the separate leaf open-sse/utils/sha3-512.ts — the executor growth is the cohesive call-site/handoff logic, not extractable without hiding the sentinel chokepoint). (2) tests/unit/chatgpt-web.test.ts 2855->3159 (+304 = #5536 GPT-5.5 Pro handoff coverage; pair-file with its executor). (3) open-sse/services/compression/strategySelector.ts 997->1022 (+25 = #5527 T02 honest default-on pipeline inflation guard wiring at the existing finalizeStackedResult choke). All cohesive at existing chokepoints; covered by tests/unit/chatgpt-web-sha3-boringssl-5531.test.ts, chatgpt-web.test.ts (GPT-5.5 Pro), compression-pipeline-inflation-guard.test.ts.", "open-sse/executors/chatgpt-web.ts": "3241", "_rebaseline_2026_08_30_11771_vercel_gateway_passthrough": "PR #11771 adds passthroughModels: true (1 line) to the Vercel AI Gateway registry entry — no split available, single-line provider-config addition.", - "_relax_velocity_2026_08_30": "127 frozen line caps and cap/testCap raised by 20% (velocity phase; see quality-baseline.json _policy)." + "_relax_velocity_2026_08_30": "127 frozen line caps and cap/testCap raised by 20% (velocity phase; see quality-baseline.json _policy).", + "_rebaseline_2026_09_02_12325_merge_v3851": "Merge of release/v3.8.51 into #12325. Both sides grew chatCore.ts at the same chokepoint: #12239 took it 5946->5976 upstream, and this PR adds its +9 non-Codex 429 branch on top. check-file-size.mjs counts split(\"\\\\n\").length (trailing-newline empty element), so the merged file is 5981. The cap is the merged LOC, not either side alone; no other entry moves." } diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index a50df1069b..65a986d186 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -384,6 +384,7 @@ import { sanitizeOpenAITool } from "../services/toolSchemaSanitizer.ts"; import { isCompactResponsesEndpoint } from "../executors/codex.ts"; import { persistCodexChildQuotaResponse } from "../services/codexAccount/index.ts"; import { invalidateCodexQuotaCache } from "../services/codexQuotaFetcher.ts"; +import { invalidateGenericQuotaCacheOnStatus } from "../services/genericQuotaFetcher.ts"; import { translateNonStreamingResponse } from "./responseTranslator.ts"; import { extractToolSchemaMap } from "../translator/response/openai-responses/toolSchemas.ts"; import { unwrapClineNonStreamingEnvelope } from "./chatCore/clineResponseEnvelope.ts"; @@ -3239,6 +3240,14 @@ export async function handleChatCore({ const errMessage = err instanceof Error ? err.message : String(err); log?.debug?.("CODEX", `Failed to persist codex quota state: ${errMessage}`); } + } else if (attemptConnectionId && res.response.status === 429) { + // Dropped generic quota cache after 429 + invalidateGenericQuotaCacheOnStatus({ + provider, + connectionId: String(attemptConnectionId), + status: res.response.status, + isolateProbe: await shouldIsolateProbeFailures(), + }); } // Track Gemini RPM + RPD request counts for 429 classification diff --git a/open-sse/services/genericQuotaFetcher.ts b/open-sse/services/genericQuotaFetcher.ts index 10d58d4b37..81419d5a6b 100644 --- a/open-sse/services/genericQuotaFetcher.ts +++ b/open-sse/services/genericQuotaFetcher.ts @@ -25,10 +25,53 @@ import { type QuotaInfo, } from "./quotaPreflight.ts"; +type UsageFetcher = ( + connection: Parameters[0], + options?: { forceRefresh?: boolean } +) => Promise; + +let usageFetcherOverride: UsageFetcher | null = null; + // 60s — matches Codex's TTL. Long enough to avoid hammering upstream usage // endpoints on every routing decision, short enough that a near-exhausted // account is skipped within one minute of crossing its threshold. const CACHE_TTL_MS = 60_000; +/** Drop unused force-refresh flags once inner provider caches (60s–5min) have expired. */ +const PENDING_FORCE_REFRESH_TTL_MS = CACHE_TTL_MS * 5; +/** key → Date.now() when invalidate asked the next fetch to force-refresh. */ +const pendingForceRefresh = new Map(); +/** key → last convert-null / throw while force-refresh was pending. */ +const pendingForceRefreshMiss = new Map(); + +/** Test-only: inject the usage dispatcher; pass null to restore. */ +export function __setGenericUsageFetcherForTests(fetcher: UsageFetcher | null): void { + usageFetcherOverride = fetcher; +} + +/** Test-only: backdate a pending force-refresh so TTL expiry is unit-testable. */ +export function __agePendingForceRefreshForTests( + provider: string, + connectionId: string, + ageMs: number +): void { + pendingForceRefresh.set(cacheKey(provider, connectionId), Date.now() - ageMs); +} + +/** Test-only: backdate a convert-null miss so the 60s hammer-guard is unit-testable. */ +export function __agePendingForceRefreshMissForTests( + provider: string, + connectionId: string, + ageMs: number +): void { + pendingForceRefreshMiss.set(cacheKey(provider, connectionId), Date.now() - ageMs); +} + +/** Test-only: drop all wrapper/flag maps so tests cannot leak across ids. */ +export function __resetGenericQuotaFetcherForTests(): void { + cache.clear(); + pendingForceRefresh.clear(); + pendingForceRefreshMiss.clear(); +} interface CacheEntry { quota: QuotaInfo; @@ -38,15 +81,72 @@ interface CacheEntry { const cache = new Map(); function cacheKey(provider: string, connectionId: string): string { - return `${provider}::${connectionId}`; + return `${provider.trim()}::${connectionId.trim()}`; } -// Auto-cleanup stale entries — same shape as codexQuotaFetcher. +function dropExpiredPendingForceRefresh(key: string, now: number): boolean { + const stampedAt = pendingForceRefresh.get(key); + if (stampedAt === undefined) return true; + if (now - stampedAt > PENDING_FORCE_REFRESH_TTL_MS) { + pendingForceRefresh.delete(key); + pendingForceRefreshMiss.delete(key); + return true; + } + return false; +} + +// Lazy expiry on read — same as the provider breaker. Name stays `is*` because +// callers only need a boolean; the map is not a public API. +function isPendingForceRefresh(key: string, now: number = Date.now()): boolean { + if (dropExpiredPendingForceRefresh(key, now)) return false; + return pendingForceRefresh.has(key); +} + +function markPendingForceRefreshMiss(key: string): void { + if (isPendingForceRefresh(key)) pendingForceRefreshMiss.set(key, Date.now()); +} + +function cachedQuotaIfFresh( + key: string, + forceRefresh: boolean, + now: number +): QuotaInfo | null { + if (forceRefresh) return null; + const cached = cache.get(key); + if (cached && now - cached.fetchedAt < CACHE_TTL_MS) return cached.quota; + return null; +} + +function isForceRefreshMissCooling( + key: string, + forceRefresh: boolean, + now: number +): boolean { + if (!forceRefresh) return false; + const missedAt = pendingForceRefreshMiss.get(key); + return missedAt !== undefined && now - missedAt < CACHE_TTL_MS; +} + +/** True when a concurrent 429 re-stamped a still-live flag during fetchUsage. */ +function isConcurrentForceRefresh(key: string, refreshStamp: number | undefined): boolean { + const currentStamp = pendingForceRefresh.get(key); + if (currentStamp === refreshStamp) return false; + return ( + currentStamp !== undefined && + Date.now() - currentStamp <= PENDING_FORCE_REFRESH_TTL_MS + ); +} + +// 5min — same as Codex. Expiry is lazy on read (`isPendingForceRefresh`); +// this timer only reaps keys nobody fetches after the 5min TTL. const _cacheCleanup = setInterval(() => { const now = Date.now(); for (const [key, entry] of cache) { if (now - entry.fetchedAt > CACHE_TTL_MS * 5) cache.delete(key); } + for (const key of pendingForceRefresh.keys()) { + dropExpiredPendingForceRefresh(key, now); + } }, 5 * 60_000); if (typeof _cacheCleanup === "object" && "unref" in _cacheCleanup) { (_cacheCleanup as { unref?: () => void }).unref?.(); @@ -217,24 +317,47 @@ function normalizeQuotaWindows( export const fetchGenericQuota: QuotaFetcher = async (connectionId, connection) => { if (!connection) return null; const conn = connection as ConnectionInputs; - const provider = typeof conn.provider === "string" ? conn.provider : null; + const provider = typeof conn.provider === "string" ? conn.provider.trim() : ""; if (!provider) return null; const key = cacheKey(provider, connectionId); - const cached = cache.get(key); - if (cached && Date.now() - cached.fetchedAt < CACHE_TTL_MS) { - return cached.quota; - } + const now = Date.now(); + const forceRefresh = isPendingForceRefresh(key, now); + const hit = cachedQuotaIfFresh(key, forceRefresh, now); + if (hit) return hit; + // convert-null / throw keep the force-refresh flag (agy inner caches are + // still stale) but must not hammer those endpoints on every routing tick. + if (isForceRefreshMissCooling(key, forceRefresh, now)) return null; + + // Capture before await: a 429 during fetchUsage re-stamps this; writing + // the pre-429 snapshot would wipe that flag and recache stale quota. + const refreshStamp = pendingForceRefresh.get(key); let usage: unknown; try { - usage = await getUsageForProvider(conn as Parameters[0]); + const fetchUsage = usageFetcherOverride ?? getUsageForProvider; + usage = await fetchUsage(conn as Parameters[0], { + ...(forceRefresh ? { forceRefresh: true } : {}), + }); } catch { + markPendingForceRefreshMiss(key); return null; } const quota = convertUsageToQuotaInfo(usage); - if (!quota) return null; + if (!quota) { + markPendingForceRefreshMiss(key); + return null; + } + + // Concurrent 429 re-stamped a still-live flag — do not recache the + // pre-429 snapshot. A vanished or expired stamp is not a 429. + if (isConcurrentForceRefresh(key, refreshStamp)) { + return quota; + } + + pendingForceRefresh.delete(key); + pendingForceRefreshMiss.delete(key); // Refresh the static window catalog so the dashboard can render the right // modal inputs without waiting for the user to open the page. @@ -250,7 +373,33 @@ export const fetchGenericQuota: QuotaFetcher = async (connectionId, connection) * fresh data instead of a 60s stale window. */ export function invalidateGenericQuotaCache(provider: string, connectionId: string): void { - cache.delete(cacheKey(provider, connectionId)); + const key = cacheKey(provider, connectionId); + cache.delete(key); + // Next fetch must bypass provider-inner usage caches (agy retrieveUserQuota / + // weekly are 60s–5min). Without this, dropping the 60s wrapper recaches stale. + // TTL matches those inner caches: after 5min the flag is a no-op. + pendingForceRefresh.set(key, Date.now()); + pendingForceRefreshMiss.delete(key); +} + +/** + * Drop the generic quota cache after an upstream 429, matching Codex's + * `invalidateCodexQuotaCache` on 429. Probe-origin failures must not mutate + * routing caches (#9817). + */ +export function invalidateGenericQuotaCacheOnStatus(args: { + provider: string | null | undefined; + connectionId: string | null | undefined; + status: number; + isolateProbe?: boolean; +}): boolean { + if (args.isolateProbe === true) return false; // undefined from callers that omit isolateProbe must still invalidate + if (args.status !== 429) return false; + const provider = typeof args.provider === "string" ? args.provider.trim() : ""; + const connectionId = typeof args.connectionId === "string" ? args.connectionId.trim() : ""; + if (!provider || !connectionId) return false; + invalidateGenericQuotaCache(provider, connectionId); + return true; } /** diff --git a/tests/unit/antigravity-429-quota-cooldown.test.ts b/tests/unit/antigravity-429-quota-cooldown.test.ts index 0191391d8a..eccdc8942d 100644 --- a/tests/unit/antigravity-429-quota-cooldown.test.ts +++ b/tests/unit/antigravity-429-quota-cooldown.test.ts @@ -169,6 +169,11 @@ test("direct Antigravity has one downstream model-lock owner and clamps body pro /accountSemaphoreKey && !deferAntigravityQuotaStateToCaller/, "chatCore must not apply a prose-derived Antigravity semaphore TTL" ); + assert.match( + chatCoreSource, + /Dropped generic quota cache after 429/, + "non-Codex 429 must leave a QUOTA debug breadcrumb" + ); assert.match( chatCoreSource, /if \(deferAntigravityQuotaStateToCaller\)[\s\S]{0,2000}else if \(kimiRateLimitResetAt\)/ diff --git a/tests/unit/generic-quota-fetcher.test.ts b/tests/unit/generic-quota-fetcher.test.ts index d8edbd9dae..6edce4894f 100644 --- a/tests/unit/generic-quota-fetcher.test.ts +++ b/tests/unit/generic-quota-fetcher.test.ts @@ -4,9 +4,36 @@ import assert from "node:assert/strict"; const genericModule = await import("../../open-sse/services/genericQuotaFetcher.ts"); const preflightModule = await import("../../open-sse/services/quotaPreflight.ts"); -const { convertUsageToQuotaInfo, registerGenericQuotaFetchers } = genericModule; +const { + convertUsageToQuotaInfo, + fetchGenericQuota, + invalidateGenericQuotaCache, + invalidateGenericQuotaCacheOnStatus, + registerGenericQuotaFetchers, + __setGenericUsageFetcherForTests, + __agePendingForceRefreshForTests, + __agePendingForceRefreshMissForTests, + __resetGenericQuotaFetcherForTests, +} = genericModule; const { getQuotaFetcher } = preflightModule; +function usageShape(remainingPercentage: number) { + return { + quotas: { + "gemini-3-flash": { + remainingPercentage, + fractionReported: true, + resetAt: "2026-09-01T20:00:00Z", + }, + gemini_models_weekly: { + remainingPercentage, + fractionReported: true, + resetAt: "2026-09-07T00:00:00Z", + }, + }, + }; +} + test("convertUsageToQuotaInfo returns null on null/undefined input", () => { assert.equal(convertUsageToQuotaInfo(null), null); assert.equal(convertUsageToQuotaInfo(undefined), null); @@ -114,3 +141,250 @@ test("registerGenericQuotaFetchers registers Claude, GLM, and OpenCode Go via th // which would couple this test to chat.ts startup wiring. The skip list // semantics are exercised by the source code review. }); + +test.afterEach(() => { + __setGenericUsageFetcherForTests(null); + __resetGenericQuotaFetcherForTests(); +}); + +test("fetchGenericQuota caches a hit inside the 60s window", async () => { + const connectionId = `agy-cache-${Date.now()}`; + const calls: Array<{ forceRefresh?: boolean }> = []; + __setGenericUsageFetcherForTests(async (_conn, options) => { + calls.push({ forceRefresh: options?.forceRefresh }); + return usageShape(80); + }); + + const connection = { provider: "agy", id: connectionId }; + const first = await fetchGenericQuota(connectionId, connection); + const second = await fetchGenericQuota(connectionId, connection); + + assert.equal(calls.length, 1, "second fetch must reuse the generic cache"); + assert.equal(first?.percentUsed, 0.2); + assert.deepEqual(second, first); + invalidateGenericQuotaCache("agy", connectionId); +}); + +test("invalidateGenericQuotaCache makes the next fetch bypass provider-inner usage caches", async () => { + const connectionId = `agy-invalidate-${Date.now()}`; + const calls: Array<{ forceRefresh?: boolean }> = []; + let remaining = 80; + __setGenericUsageFetcherForTests(async (_conn, options) => { + calls.push({ forceRefresh: options?.forceRefresh }); + return usageShape(remaining); + }); + + const connection = { provider: "agy", id: connectionId }; + const first = await fetchGenericQuota(connectionId, connection); + assert.equal(first?.percentUsed, 0.2); + assert.equal(calls[0]?.forceRefresh, undefined); + + remaining = 0; + invalidateGenericQuotaCache("agy", connectionId); + const second = await fetchGenericQuota(connectionId, connection); + + assert.equal(calls.length, 2, "invalidate must drop the 60s generic cache"); + assert.equal( + calls[1]?.forceRefresh, + true, + "agy retrieveUserQuota / weekly caches are 60s–5min; invalidate must force-refresh or the recache is stale" + ); + assert.equal(second?.percentUsed, 1); + invalidateGenericQuotaCache("agy", connectionId); +}); + +test("invalidateGenericQuotaCacheOnStatus drops cache on 429 and ignores 200", async () => { + const connectionId = `agy-429-${Date.now()}`; + const calls: Array<{ forceRefresh?: boolean }> = []; + __setGenericUsageFetcherForTests(async (_conn, options) => { + calls.push({ forceRefresh: options?.forceRefresh }); + return usageShape(50); + }); + + const connection = { provider: "agy", id: connectionId }; + await fetchGenericQuota(connectionId, connection); + assert.equal(calls.length, 1); + + invalidateGenericQuotaCacheOnStatus({ + provider: "agy", + connectionId, + status: 200, + isolateProbe: false, + }); + await fetchGenericQuota(connectionId, connection); + assert.equal(calls.length, 1, "200 must not drop the generic quota cache"); + + const dropped429 = invalidateGenericQuotaCacheOnStatus({ + provider: "agy", + connectionId, + status: 429, + isolateProbe: false, + }); + assert.equal(dropped429, true); + await fetchGenericQuota(connectionId, connection); + assert.equal(calls.length, 2, "429 must drop the generic quota cache"); + assert.equal(calls[1]?.forceRefresh, true); + + invalidateGenericQuotaCacheOnStatus({ + provider: "agy", + connectionId, + status: 429, + isolateProbe: true, + }); + await fetchGenericQuota(connectionId, connection); + assert.equal(calls.length, 2, "probe-origin 429 must not touch routing caches"); + + assert.doesNotThrow(() => + invalidateGenericQuotaCacheOnStatus({ + provider: "agy", + connectionId: null, + status: 429, + isolateProbe: false, + }) + ); + invalidateGenericQuotaCache("agy", connectionId); +}); + +test("invalidateGenericQuotaCacheOnStatus trims to the same key fetchGenericQuota uses", async () => { + const connectionId = `agy-trim-${Date.now()}`; + const calls: Array<{ forceRefresh?: boolean }> = []; + __setGenericUsageFetcherForTests(async (_conn, options) => { + calls.push({ forceRefresh: options?.forceRefresh }); + return usageShape(40); + }); + + const connection = { provider: "agy", id: connectionId }; + await fetchGenericQuota(` ${connectionId} `, connection); + assert.equal(calls.length, 1); + + const dropped = invalidateGenericQuotaCacheOnStatus({ + provider: " agy ", + connectionId: ` ${connectionId} `, + status: 429, + isolateProbe: false, + }); + assert.equal(dropped, true); + + await fetchGenericQuota(connectionId, { provider: "agy", id: connectionId }); + assert.equal(calls.length, 2, "padded 429 key must drop the unpadded wrapper cache"); + assert.equal(calls[1]?.forceRefresh, true); + invalidateGenericQuotaCache("agy", connectionId); +}); + +test("convert-null after invalidate keeps forceRefresh until a measurable quota recaches", async () => { + const connectionId = `agy-null-${Date.now()}`; + const calls: Array<{ forceRefresh?: boolean }> = []; + let n = 0; + __setGenericUsageFetcherForTests(async (_conn, options) => { + calls.push({ forceRefresh: options?.forceRefresh }); + n += 1; + if (n === 1) return usageShape(80); + if (n === 2) return { message: "auth expired" }; + return usageShape(10); + }); + + const connection = { provider: "agy", id: connectionId }; + await fetchGenericQuota(connectionId, connection); + invalidateGenericQuotaCache("agy", connectionId); + + const second = await fetchGenericQuota(connectionId, connection); + assert.equal(second, null); + assert.equal(calls[1]?.forceRefresh, true); + + await fetchGenericQuota(connectionId, connection); + assert.equal(calls.length, 2, "convert-null must not hammer usage inside 60s"); + + __agePendingForceRefreshMissForTests("agy", connectionId, 60_000 + 1); + const fourth = await fetchGenericQuota(connectionId, connection); + assert.equal(calls.length, 3); + assert.equal(calls[2]?.forceRefresh, true, "convert-null must not drop the force-refresh flag"); + assert.equal(fourth?.percentUsed, 0.9); + invalidateGenericQuotaCache("agy", connectionId); +}); + +test("in-flight fetch must not drop a concurrent 429 force-refresh", async () => { + const connectionId = `agy-race-${Date.now()}`; + let release: (value?: unknown) => void = () => {}; + const gate = new Promise((resolve) => { + release = resolve; + }); + const calls: Array<{ forceRefresh?: boolean }> = []; + let n = 0; + __setGenericUsageFetcherForTests(async (_conn, options) => { + calls.push({ forceRefresh: options?.forceRefresh }); + n += 1; + if (n === 1) { + await gate; + return usageShape(80); + } + return usageShape(10); + }); + + const connection = { provider: "agy", id: connectionId }; + const inflight = fetchGenericQuota(connectionId, connection); + await Promise.resolve(); + invalidateGenericQuotaCache("agy", connectionId); + release(); + const first = await inflight; + assert.equal(first?.percentUsed, 0.2); + + const second = await fetchGenericQuota(connectionId, connection); + assert.equal(calls.length, 2, "concurrent 429 must not let the in-flight recache wipe force-refresh"); + assert.equal(calls[1]?.forceRefresh, true); + assert.equal(second?.percentUsed, 0.9); + invalidateGenericQuotaCache("agy", connectionId); +}); + +test("expired pending force-refresh does not bypass the 60s wrapper cache", async () => { + const connectionId = `agy-expire-${Date.now()}`; + const calls: Array<{ forceRefresh?: boolean }> = []; + __setGenericUsageFetcherForTests(async (_conn, options) => { + calls.push({ forceRefresh: options?.forceRefresh }); + return usageShape(80); + }); + + const connection = { provider: "agy", id: connectionId }; + await fetchGenericQuota(connectionId, connection); + invalidateGenericQuotaCache("agy", connectionId); + __agePendingForceRefreshForTests("agy", connectionId, 5 * 60_000 + 1); + + await fetchGenericQuota(connectionId, connection); + assert.equal(calls.length, 2, "wrapper cache was dropped; fetch still happens"); + assert.equal( + calls[1]?.forceRefresh, + undefined, + "expired force-refresh must not pass forceRefresh after inner caches have aged out" + ); + invalidateGenericQuotaCache("agy", connectionId); +}); + +test("stamp expiry during in-flight fetch still writes the wrapper cache", async () => { + const connectionId = `agy-stamp-expire-${Date.now()}`; + let release: (value?: unknown) => void = () => {}; + const gate = new Promise((resolve) => { + release = resolve; + }); + const calls: Array<{ forceRefresh?: boolean }> = []; + let n = 0; + __setGenericUsageFetcherForTests(async (_conn, options) => { + calls.push({ forceRefresh: options?.forceRefresh }); + n += 1; + if (n === 1) return usageShape(50); + await gate; + return usageShape(80); + }); + + const connection = { provider: "agy", id: connectionId }; + await fetchGenericQuota(connectionId, connection); + invalidateGenericQuotaCache("agy", connectionId); + + const inflight = fetchGenericQuota(connectionId, connection); + await Promise.resolve(); + __agePendingForceRefreshForTests("agy", connectionId, 5 * 60_000 + 1); + release(); + await inflight; + + await fetchGenericQuota(connectionId, connection); + assert.equal(calls.length, 2, "expired stamp during await is not a 429; cache the result"); + invalidateGenericQuotaCache("agy", connectionId); +}); From c2d2b0ac1454b984c369ca4396608e77cb0aebde Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Thu, 3 Sep 2026 11:39:03 -0400 Subject: [PATCH 035/143] feat(providers): surface CSV import row errors and ship a template (#12504) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check:provider-consistency` OK (273 entradas REGISTRY, **356** providers canônicos), `check-docs-counts-sync` exit 0 e **300/300** nos testes que a leva toca. O crescimento de arquivo que os PRs empilham uns sobre os outros foi rebaselinado num único registro datado (`_rebaseline_2026_09_03_houminxi_batch`), com a decomposição por arquivo: `providers/page.tsx` +18 (import CSV do #12504 + busca do #12495 no mesmo painel), `accountFallback.ts` +6 (o #12566 sobre o rebaseline que o #12590 já registrou — os dois tocam `checkFallbackError`) e `chatCore.ts` +3 (invalidez de cache de quota no 429 do #12325). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. --- .../features/12071-csv-import-errors.md | 1 + docs/guides/USER_GUIDE.md | 2 + docs/providers/CSV-IMPORT.md | 43 +++++ docs/providers/meta.json | 3 +- .../ImportProvidersFromFileModal.tsx | 40 ++++- .../components/providerImportFeedback.ts | 165 ++++++++++++++++++ .../components/useImportProvidersFromFile.ts | 28 +-- src/i18n/messages/en.json | 3 + src/i18n/messages/pt-BR.json | 3 + src/i18n/messages/vi.json | 3 + .../provider-import-feedback-12071.test.ts | 143 +++++++++++++++ 11 files changed, 418 insertions(+), 16 deletions(-) create mode 100644 changelog.d/features/12071-csv-import-errors.md create mode 100644 docs/providers/CSV-IMPORT.md create mode 100644 src/app/(dashboard)/dashboard/providers/components/providerImportFeedback.ts create mode 100644 tests/unit/provider-import-feedback-12071.test.ts diff --git a/changelog.d/features/12071-csv-import-errors.md b/changelog.d/features/12071-csv-import-errors.md new file mode 100644 index 0000000000..096eed4ef6 --- /dev/null +++ b/changelog.d/features/12071-csv-import-errors.md @@ -0,0 +1 @@ +- **feat(providers):** import-from-file modal shows per-row API errors and ships a downloadable CSV template ([#12071](https://github.com/diegosouzapw/OmniRoute/issues/12071)) diff --git a/docs/guides/USER_GUIDE.md b/docs/guides/USER_GUIDE.md index 72acc05bfc..ceed44c426 100644 --- a/docs/guides/USER_GUIDE.md +++ b/docs/guides/USER_GUIDE.md @@ -122,6 +122,8 @@ Access via: WhatsApp, Telegram, Slack, Discord, iMessage, Signal... ## 📖 Provider Setup +To bulk-add API-key connections from a CSV or JSON file, use **Dashboard → Providers → Import from file**. Columns are positional (`provider,name,apiKey,baseUrl,priority`); `provider` must already exist as a managed provider or a compatible node. See [Import providers from a CSV or JSON file](../providers/CSV-IMPORT.md). + ### 🔐 Subscription Providers #### Claude Code (Pro/Max) diff --git a/docs/providers/CSV-IMPORT.md b/docs/providers/CSV-IMPORT.md new file mode 100644 index 0000000000..4f701a16a2 --- /dev/null +++ b/docs/providers/CSV-IMPORT.md @@ -0,0 +1,43 @@ +--- +title: "Import providers from a CSV or JSON file" +--- + +# Import providers from a CSV or JSON file + +Dashboard → Providers → **Import from file** creates API-key connections from a CSV or JSON list. Each row can target a different provider. Partial failure is the contract: valid rows still import when others fail, and the modal lists why the failed rows were rejected. + +This import does **not** create new OpenAI/Anthropic-compatible endpoint nodes. Create those first (Dashboard → Providers → Add OpenAI-Compatible, or `omniroute nodes add`), then import rows whose `provider` column is that node's id. A per-row `baseUrl` can still override the node's URL. + +## CSV (positional) + +Column names are cosmetic. The parser splits each row and destructures by index: + +| Index | Field | Required | Notes | +| ----- | ----- | -------- | ----- | +| 0 | `provider` | yes | Existing managed provider id (`openai`, `anthropic`, …) **or** an already-registered OpenAI/Anthropic-compatible **node** id | +| 1 | `name` | yes | Connection display name | +| 2 | `apiKey` | yes | API key | +| 3 | `baseUrl` | no | Per-row URL override | +| 4 | `priority` | no | Integer 1–100 | + +A first line whose first column is the literal word `provider` (any case) is skipped as a header. Blank lines and `#` comments are skipped. + +Download a starter file from the import modal (**Download CSV template**). Example: + +```csv +# OmniRoute provider import (positional columns) +provider,name,apiKey,baseUrl,priority +openai,Prod OpenAI,sk-your-openai-key,,1 +``` + +A made-up id such as `openai-compatible-chat-001` is not a node. The API returns `Unknown or unsupported provider` for that row; the modal shows it next to the row name. + +## JSON + +A JSON array of objects with the same fields (`provider`, `name`, `apiKey`, `baseUrl?`, `priority?`). Unlike CSV, JSON keys are named. + +```json +[ + { "provider": "openai", "name": "Prod OpenAI", "apiKey": "sk-your-openai-key", "priority": 1 } +] +``` diff --git a/docs/providers/meta.json b/docs/providers/meta.json index fa6485dd57..b5eca33685 100644 --- a/docs/providers/meta.json +++ b/docs/providers/meta.json @@ -8,6 +8,7 @@ "AGENTROUTER", "ZED-DOCKER", "CURSOR-DOCKER", - "CURSOR-API-KEY-AND-CLI" + "CURSOR-API-KEY-AND-CLI", + "CSV-IMPORT" ] } diff --git a/src/app/(dashboard)/dashboard/providers/components/ImportProvidersFromFileModal.tsx b/src/app/(dashboard)/dashboard/providers/components/ImportProvidersFromFileModal.tsx index 0363f205d8..b9f9417b50 100644 --- a/src/app/(dashboard)/dashboard/providers/components/ImportProvidersFromFileModal.tsx +++ b/src/app/(dashboard)/dashboard/providers/components/ImportProvidersFromFileModal.tsx @@ -4,6 +4,12 @@ import { useTranslations } from "next-intl"; import { Button, Modal } from "@/shared/components"; import type { ParsedProviderImportEntry, ProviderImportParseError } from "./parseProviderImportFile"; import { useImportProvidersFromFile } from "./useImportProvidersFromFile"; +import { + downloadProviderImportCsvTemplate, + formatImportErrorLine, + visibleImportErrors, + type ImportResult, +} from "./providerImportFeedback"; interface ImportProvidersFromFileModalProps { isOpen: boolean; @@ -122,6 +128,30 @@ function FilePickerRow({ fileInputRef, fileName, onFile, t }: FilePickerRowProps ); } +function ImportResultPanel({ result, t }: { result: ImportResult; t: Translator }) { + const { shown, extra } = visibleImportErrors(result.errors); + const failed = result.failed > 0 || shown.length > 0; + return ( +
+ {t("importFromFileResult", { success: result.success, failed: result.failed })} + {shown.length > 0 && ( +
    + {shown.map((err, i) => ( +
  • {formatImportErrorLine(err)}
  • + ))} + {extra > 0 &&
  • {t("importFromFileMoreErrors", { count: extra })}
  • } +
+ )} +
+ ); +} + /** * Wizard step: upload a CSV/JSON file listing MULTIPLE, possibly different providers, * pick which parsed rows to actually import, then submit them in one batch (#6836). @@ -141,18 +171,18 @@ export function ImportProvidersFromFileModal({ s.handleClose(onClose)} title={t("importFromFileTitle")} maxWidth="xl">

{t("importFromFileDescription")}

+

{t("importFromFileSchemaHint")}

- {s.result && ( -
- {t("importFromFileResult", { success: s.result.success, failed: s.result.failed })} -
- )} + {s.result && }
+ diff --git a/src/app/(dashboard)/dashboard/providers/components/providerImportFeedback.ts b/src/app/(dashboard)/dashboard/providers/components/providerImportFeedback.ts new file mode 100644 index 0000000000..4f1564051a --- /dev/null +++ b/src/app/(dashboard)/dashboard/providers/components/providerImportFeedback.ts @@ -0,0 +1,165 @@ +/** + * #12071 — import-modal feedback helpers. + * + * POST /api/providers/import already returns per-row `{index,name,provider,message}`. + * The modal used to keep only success/failed/total and drop `errors` on the floor. + * These helpers stay a pure, dependency-free module so the hook can stay under the + * LOC ratchet and the same formatter can be unit-tested without React. + */ + +export type ImportRowError = { + index?: number; + name?: string; + provider?: string; + message: string; +}; + +export type ImportResult = { + success: number; + failed: number; + total: number; + errors: ImportRowError[]; +}; + +const VISIBLE_ERROR_CAP = 10; + +function asFiniteNumber(value: unknown): number { + return typeof value === "number" && Number.isFinite(value) ? value : 0; +} + +function asRowError(value: unknown): ImportRowError | null { + if (!value || typeof value !== "object") return null; + const row = value as Record; + if (typeof row.message !== "string" || !row.message.trim()) return null; + return { + ...(typeof row.index === "number" && Number.isFinite(row.index) ? { index: row.index } : {}), + ...(typeof row.name === "string" && row.name.trim() ? { name: row.name.trim() } : {}), + ...(typeof row.provider === "string" && row.provider.trim() ? { provider: row.provider.trim() } : {}), + message: row.message.trim(), + }; +} + +/** Keep counts plus a sanitized `errors` array. A missing/non-array field becomes []. */ +export function normalizeImportResponse(data: unknown): ImportResult { + const body = data && typeof data === "object" ? (data as Record) : {}; + const rawErrors = Array.isArray(body.errors) ? body.errors : []; + return { + success: asFiniteNumber(body.success), + failed: asFiniteNumber(body.failed), + total: asFiniteNumber(body.total), + errors: rawErrors.map(asRowError).filter((row): row is ImportRowError => row !== null), + }; +} + +export type ImportHttpOutcome = { + result: ImportResult; + shouldRefresh: boolean; +}; + +function httpFailureResult(status: number, data: unknown, fallback: ImportResult): ImportResult { + if (fallback.errors.length > 0) { + return { ...fallback, success: 0 }; + } + const body = data && typeof data === "object" ? (data as Record) : {}; + const detail = typeof body.error === "string" ? body.error.trim() : ""; + const message = detail ? `HTTP ${status}: ${detail}` : `HTTP ${status}`; + return { + success: 0, + failed: Math.max(1, fallback.failed), + total: Math.max(1, fallback.total), + errors: [{ message }], + }; +} + +/** + * Map an import HTTP response onto the modal result. + * Non-ok statuses still populate `errors`. Refresh is a boolean so the hook + * can await `onImported` outside this function (a throw there must not + * overwrite a successful import result). + */ +export function applyImportHttpOutcome( + res: { ok: boolean; status: number }, + data: unknown +): ImportHttpOutcome { + const normalized = normalizeImportResponse(data); + if (!res.ok) { + return { result: httpFailureResult(res.status, data, normalized), shouldRefresh: false }; + } + return { result: normalized, shouldRefresh: normalized.success > 0 }; +} + +/** Parse the import response body. Non-JSON becomes `{ ok: false, data: { error } }`. */ +export async function readImportResponse(res: Response): Promise<{ + ok: boolean; + status: number; + data: unknown; +}> { + try { + return { ok: res.ok, status: res.status, data: await res.json() }; + } catch { + return { ok: false, status: res.status, data: { error: "Invalid JSON body" } }; + } +} + +export function networkImportFailure(err: unknown): ImportResult { + return { + success: 0, + failed: 1, + total: 1, + errors: [{ message: err instanceof Error ? err.message : "Import request failed" }], + }; +} + +/** First 10 rows plus the leftover count — same cap as AddApiKeyModal bulk import. */ +export function visibleImportErrors(errors: ImportRowError[]): { + shown: ImportRowError[]; + extra: number; +} { + return { + shown: errors.slice(0, VISIBLE_ERROR_CAP), + extra: Math.max(0, errors.length - VISIBLE_ERROR_CAP), + }; +} + +/** One line for the modal list: name, else provider, else 1-based row index. */ +export function formatImportErrorLine(err: ImportRowError): string { + const label = + (typeof err.name === "string" && err.name.trim()) || + (typeof err.provider === "string" && err.provider.trim()) || + (typeof err.index === "number" && Number.isFinite(err.index) ? `row ${err.index + 1}` : "row"); + return `${label}: ${err.message}`; +} + +/** + * Positional CSV sample. Column 0 must be an *existing* managed provider id + * or an already-registered OpenAI/Anthropic-compatible node id — this import + * does not create new endpoint nodes. Header names are cosmetic; the parser + * destructures by index (`provider,name,apiKey,baseUrl,priority`). + */ +export const PROVIDER_IMPORT_CSV_TEMPLATE = `# OmniRoute provider import (positional columns) +# Columns: provider, name, apiKey, baseUrl (optional), priority (optional, 1-100) +# The provider column must be an existing managed provider id (openai, anthropic, …) +# or an already-registered OpenAI/Anthropic-compatible node id. +# This import does not create new endpoint nodes. Add those first (Dashboard → Providers → Add OpenAI-Compatible). +provider,name,apiKey,baseUrl,priority +openai,Prod OpenAI,sk-your-openai-key,,1 +`; + +export function downloadTextFile(content: string, filename: string, mimeType: string): void { + const blob = new Blob([content], { type: mimeType }); + const url = URL.createObjectURL(blob); + const link = document.createElement("a"); + link.href = url; + link.download = filename; + try { + document.body.appendChild(link); + link.click(); + } finally { + link.remove(); + URL.revokeObjectURL(url); + } +} + +export function downloadProviderImportCsvTemplate(): void { + downloadTextFile(PROVIDER_IMPORT_CSV_TEMPLATE, "omniroute-provider-import-template.csv", "text/csv"); +} diff --git a/src/app/(dashboard)/dashboard/providers/components/useImportProvidersFromFile.ts b/src/app/(dashboard)/dashboard/providers/components/useImportProvidersFromFile.ts index ecf50e5bcb..a09f00e0c4 100644 --- a/src/app/(dashboard)/dashboard/providers/components/useImportProvidersFromFile.ts +++ b/src/app/(dashboard)/dashboard/providers/components/useImportProvidersFromFile.ts @@ -4,13 +4,17 @@ import { type ParsedProviderImportEntry, type ProviderImportParseError, } from "./parseProviderImportFile"; +import { + applyImportHttpOutcome, + networkImportFailure, + readImportResponse, + type ImportResult, +} from "./providerImportFeedback"; -export type ImportResult = { success: number; failed: number; total: number }; +export type { ImportResult }; /** - * All state + handlers for `ImportProvidersFromFileModal`, split into a hook purely - * to keep the component's own function under the repo's max-lines-per-function ratchet - * (#6836). Behavior is unchanged — this is a pure extraction, not a refactor. + * State + handlers for ImportProvidersFromFileModal (#6836/#12071). */ export function useImportProvidersFromFile(onImported: () => Promise) { const fileInputRef = useRef(null); @@ -66,15 +70,19 @@ export function useImportProvidersFromFile(onImported: () => Promise) { setImporting(true); try { const res = await fetch("/api/providers/import", { - method: "POST", - headers: { "Content-Type": "application/json" }, + method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ entries: toImport }), }); - const data = await res.json().catch(() => ({})); - if (res.ok) { - setResult({ success: data.success ?? 0, failed: data.failed ?? 0, total: data.total ?? 0 }); - await onImported(); + const parsed = await readImportResponse(res); + const outcome = applyImportHttpOutcome(parsed, parsed.data); + setResult(outcome.result); + if (outcome.shouldRefresh) { + try { + await onImported(); + } catch { /* refresh failure must not replace the import result */ } } + } catch (err) { + setResult(networkImportFailure(err)); } finally { setImporting(false); } diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index 292d76c86d..9424b64ed7 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -5067,6 +5067,9 @@ "importFromFileImporting": "Importing…", "importFromFileImport": "Import {count} providers", "importFromFileResult": "Imported {success} providers ({failed} failed)", + "importFromFileDownloadTemplate": "Download CSV template", + "importFromFileMoreErrors": "+{count} more", + "importFromFileSchemaHint": "CSV columns are positional: provider, name, apiKey, baseUrl (optional), priority (optional). The provider column must be an existing managed provider id or an already-registered OpenAI/Anthropic-compatible node id — this import does not create new endpoint nodes.", "adaptaTutorial": { "title": "How to connect Adapta Web", "introPrefix": "Adapta authenticates through Clerk. The token", diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index 7c4d1eb12c..5199f22e2c 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -5068,6 +5068,9 @@ "importFromFileImporting": "Importando…", "importFromFileImport": "Importar {count} provedores", "importFromFileResult": "{success} provedores importados ({failed} falharam)", + "importFromFileDownloadTemplate": "Baixar modelo CSV", + "importFromFileMoreErrors": "+{count} mais", + "importFromFileSchemaHint": "As colunas CSV são posicionais: provider, name, apiKey, baseUrl (opcional), priority (opcional). A coluna provider deve ser o id de um provedor gerenciado existente ou o id de um nó compatível com OpenAI/Anthropic já registrado — esta importação não cria novos nós de endpoint.", "adaptaTutorial": { "title": "Como conectar o Adapta Web", "introPrefix": "Adapta autentica através do Clerk. O token", diff --git a/src/i18n/messages/vi.json b/src/i18n/messages/vi.json index 61287b0446..874cab83b8 100644 --- a/src/i18n/messages/vi.json +++ b/src/i18n/messages/vi.json @@ -5068,6 +5068,9 @@ "importFromFileImporting": "Đang nhập…", "importFromFileImport": "Nhập {count} nhà cung cấp", "importFromFileResult": "Đã nhập {success} nhà cung cấp ({failed} không thành công)", + "importFromFileDownloadTemplate": "Tải mẫu CSV", + "importFromFileMoreErrors": "+{count} nữa", + "importFromFileSchemaHint": "Các cột CSV theo vị trí: provider, name, apiKey, baseUrl (tùy chọn), priority (tùy chọn). Cột provider phải là id nhà cung cấp được quản lý hiện có hoặc id nút tương thích OpenAI/Anthropic đã đăng ký — quá trình nhập này không tạo nút endpoint mới.", "adaptaTutorial": { "title": "How to connect Adapta Web", "introPrefix": "Adapta authenticates through Clerk. The token", diff --git a/tests/unit/provider-import-feedback-12071.test.ts b/tests/unit/provider-import-feedback-12071.test.ts new file mode 100644 index 0000000000..8db247a84e --- /dev/null +++ b/tests/unit/provider-import-feedback-12071.test.ts @@ -0,0 +1,143 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +const feedback = await import( + "../../src/app/(dashboard)/dashboard/providers/components/providerImportFeedback.ts" +); +const { parseProviderImportFile } = await import( + "../../src/app/(dashboard)/dashboard/providers/components/parseProviderImportFile.ts" +); + +test("#12071 normalizeImportResponse keeps the per-row errors array", () => { + const result = feedback.normalizeImportResponse({ + success: 1, + failed: 2, + total: 3, + errors: [ + { index: 1, name: "srv-107", provider: "openai-compatible-chat-001", message: "Unknown or unsupported provider" }, + { index: 2, name: "srv-135", provider: "openai", message: "Provider node not found" }, + ], + }); + assert.equal(result.success, 1); + assert.equal(result.failed, 2); + assert.equal(result.total, 3); + assert.equal(result.errors.length, 2); + assert.equal(result.errors[0].message, "Unknown or unsupported provider"); + assert.equal(result.errors[1].name, "srv-135"); +}); + +test("#12071 normalizeImportResponse treats a missing errors field as [] (today's silent drop)", () => { + const result = feedback.normalizeImportResponse({ success: 0, failed: 3, total: 3 }); + assert.deepEqual(result.errors, []); + assert.equal(result.failed, 3); +}); + +test("#12071 normalizeImportResponse ignores a non-array errors field", () => { + const result = feedback.normalizeImportResponse({ success: 0, failed: 1, total: 1, errors: "boom" }); + assert.deepEqual(result.errors, []); +}); + +test("#12071 visibleImportErrors caps at 10 and reports the remainder", () => { + const errors = Array.from({ length: 12 }, (_, i) => ({ message: `row ${i}` })); + const { shown, extra } = feedback.visibleImportErrors(errors); + assert.equal(shown.length, 10); + assert.equal(extra, 2); + assert.equal(shown[0].message, "row 0"); +}); + +test("#12071 formatImportErrorLine prefers name, then provider, then 1-based row", () => { + assert.equal( + feedback.formatImportErrorLine({ name: "Grade-S-Node", message: "Unknown or unsupported provider" }), + "Grade-S-Node: Unknown or unsupported provider" + ); + assert.equal( + feedback.formatImportErrorLine({ provider: "openai", message: "Provider node not found" }), + "openai: Provider node not found" + ); + assert.equal(feedback.formatImportErrorLine({ index: 0, message: "failed" }), "row 1: failed"); +}); + +test("#12071 CSV template is positional and parses to one openai row", () => { + const parsed = parseProviderImportFile(feedback.PROVIDER_IMPORT_CSV_TEMPLATE, "csv"); + assert.equal(parsed.errors.length, 0); + assert.equal(parsed.entries.length, 1); + assert.equal(parsed.entries[0].provider, "openai"); + assert.equal(parsed.entries[0].name, "Prod OpenAI"); + assert.equal(parsed.entries[0].apiKey, "sk-your-openai-key"); + assert.equal(parsed.entries[0].priority, 1); +}); + +test("#12071 CSV template comments document that provider must already exist", () => { + assert.match(feedback.PROVIDER_IMPORT_CSV_TEMPLATE, /existing managed provider/i); + assert.match(feedback.PROVIDER_IMPORT_CSV_TEMPLATE, /does not create new endpoint nodes/i); + assert.match(feedback.PROVIDER_IMPORT_CSV_TEMPLATE, /positional/i); +}); + +test("#12071 asRowError trims leading/trailing whitespace on message", () => { + const result = feedback.normalizeImportResponse({ + success: 0, + failed: 1, + total: 1, + errors: [{ name: "srv-107", message: " Unknown or unsupported provider " }], + }); + assert.equal(result.errors.length, 1); + assert.equal(result.errors[0].message, "Unknown or unsupported provider"); +}); + +test("#12071 applyImportHttpOutcome on !ok zeros success even if the body claimed some", () => { + const outcome = feedback.applyImportHttpOutcome( + { ok: false, status: 500 }, + { + success: 5, + failed: 0, + total: 5, + errors: [{ name: "a", message: "Unknown or unsupported provider" }], + } + ); + assert.equal(outcome.shouldRefresh, false); + assert.equal(outcome.result.success, 0); + assert.equal(outcome.result.errors.length, 1); +}); + +test("#12071 applyImportHttpOutcome surfaces non-ok HTTP without calling onImported", () => { + const outcome = feedback.applyImportHttpOutcome( + { ok: false, status: 400 }, + { error: "Invalid JSON body" } + ); + assert.equal(outcome.shouldRefresh, false); + assert.equal(outcome.result.success, 0); + assert.equal(outcome.result.failed, 1); + assert.equal(outcome.result.errors.length, 1); + assert.match(outcome.result.errors[0].message, /HTTP 400/); +}); + +test("#12071 applyImportHttpOutcome on ok with success>0 requests refresh", () => { + const outcome = feedback.applyImportHttpOutcome( + { ok: true, status: 200 }, + { success: 2, failed: 1, total: 3, errors: [{ name: "bad", message: "Unknown or unsupported provider" }] } + ); + assert.equal(outcome.shouldRefresh, true); + assert.equal(outcome.result.success, 2); + assert.equal(outcome.result.failed, 1); + assert.equal(outcome.result.errors[0].name, "bad"); +}); + +test("#12071 applyImportHttpOutcome on ok with success=0 still keeps errors and skips refresh", () => { + const outcome = feedback.applyImportHttpOutcome( + { ok: true, status: 200 }, + { success: 0, failed: 3, total: 3, errors: [{ name: "a", message: "Unknown or unsupported provider" }] } + ); + assert.equal(outcome.shouldRefresh, false); + assert.equal(outcome.result.success, 0); + assert.equal(outcome.result.errors.length, 1); +}); + +test("#12071 readImportResponse treats JSON parse failure as !ok with a body error", async () => { + const res = new Response("not-json", { status: 200, headers: { "Content-Type": "text/plain" } }); + const parsed = await feedback.readImportResponse(res); + assert.equal(parsed.ok, false); + assert.equal(parsed.status, 200); + const outcome = feedback.applyImportHttpOutcome(parsed, parsed.data); + assert.equal(outcome.shouldRefresh, false); + assert.match(outcome.result.errors[0].message, /Invalid JSON body/); +}); From 35caeb31f2b5f67d060c07eac087e37244590bf9 Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Thu, 3 Sep 2026 11:39:27 -0400 Subject: [PATCH 036/143] feat(settings): persist headroomUrl for the Headroom proxy (#12487) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check:provider-consistency` OK (273 entradas REGISTRY, **356** providers canônicos), `check-docs-counts-sync` exit 0 e **300/300** nos testes que a leva toca. O crescimento de arquivo que os PRs empilham uns sobre os outros foi rebaselinado num único registro datado (`_rebaseline_2026_09_03_houminxi_batch`), com a decomposição por arquivo: `providers/page.tsx` +18 (import CSV do #12504 + busca do #12495 no mesmo painel), `accountFallback.ts` +6 (o #12566 sobre o rebaseline que o #12590 já registrou — os dois tocam `checkFallbackError`) e `chatCore.ts` +3 (invalidez de cache de quota no 429 do #12325). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. --- changelog.d/features/12306-headroom-url.md | 1 + .../dashboard/settings/advanced/page.tsx | 2 + .../settings/components/HeadroomProxyCard.tsx | 235 ++++++++++++++++++ src/i18n/messages/ar.json | 10 +- src/i18n/messages/az.json | 10 +- src/i18n/messages/bg.json | 10 +- src/i18n/messages/bn.json | 10 +- src/i18n/messages/cs.json | 10 +- src/i18n/messages/da.json | 10 +- src/i18n/messages/de.json | 10 +- src/i18n/messages/en.json | 8 + src/i18n/messages/es.json | 10 +- src/i18n/messages/fa.json | 10 +- src/i18n/messages/fi.json | 10 +- src/i18n/messages/fr.json | 10 +- src/i18n/messages/gu.json | 10 +- src/i18n/messages/he.json | 10 +- src/i18n/messages/hi.json | 10 +- src/i18n/messages/hu.json | 10 +- src/i18n/messages/id.json | 10 +- src/i18n/messages/it.json | 10 +- src/i18n/messages/ja.json | 10 +- src/i18n/messages/ko.json | 10 +- src/i18n/messages/mr.json | 10 +- src/i18n/messages/ms.json | 10 +- src/i18n/messages/nl.json | 10 +- src/i18n/messages/no.json | 10 +- src/i18n/messages/phi.json | 10 +- src/i18n/messages/pl.json | 10 +- src/i18n/messages/pt-BR.json | 10 +- src/i18n/messages/pt.json | 10 +- src/i18n/messages/ro.json | 10 +- src/i18n/messages/ru.json | 10 +- src/i18n/messages/sk.json | 10 +- src/i18n/messages/sv.json | 10 +- src/i18n/messages/sw.json | 10 +- src/i18n/messages/ta.json | 10 +- src/i18n/messages/te.json | 10 +- src/i18n/messages/th.json | 10 +- src/i18n/messages/tr.json | 10 +- src/i18n/messages/uk-UA.json | 10 +- src/i18n/messages/ur.json | 10 +- src/i18n/messages/vi.json | 10 +- src/i18n/messages/zh-CN.json | 10 +- src/i18n/messages/zh-TW.json | 10 +- src/shared/validation/settingsSchemas.ts | 21 ++ .../unit/headroom-url-settings-12306.test.ts | 153 ++++++++++++ 47 files changed, 789 insertions(+), 41 deletions(-) create mode 100644 changelog.d/features/12306-headroom-url.md create mode 100644 src/app/(dashboard)/dashboard/settings/components/HeadroomProxyCard.tsx create mode 100644 tests/unit/headroom-url-settings-12306.test.ts diff --git a/changelog.d/features/12306-headroom-url.md b/changelog.d/features/12306-headroom-url.md new file mode 100644 index 0000000000..7478e094f8 --- /dev/null +++ b/changelog.d/features/12306-headroom-url.md @@ -0,0 +1 @@ +- **feat(settings):** persist `headroomUrl` through Settings so status/start use the operator URL instead of only `HEADROOM_URL` ([#12306](https://github.com/diegosouzapw/OmniRoute/issues/12306)) diff --git a/src/app/(dashboard)/dashboard/settings/advanced/page.tsx b/src/app/(dashboard)/dashboard/settings/advanced/page.tsx index 5dfefd170b..af8eb23deb 100644 --- a/src/app/(dashboard)/dashboard/settings/advanced/page.tsx +++ b/src/app/(dashboard)/dashboard/settings/advanced/page.tsx @@ -5,6 +5,7 @@ import LogToolSourcesCard from "../components/LogToolSourcesCard"; import PayloadRulesTab from "../components/PayloadRulesTab"; import RequestLimitsTab from "../components/RequestLimitsTab"; import CliproxyapiSettingsTab from "../components/CliproxyapiSettingsTab"; +import HeadroomProxyCard from "../components/HeadroomProxyCard"; export default function SettingsAdvancedPage() { return ( @@ -14,6 +15,7 @@ export default function SettingsAdvancedPage() { +
); } diff --git a/src/app/(dashboard)/dashboard/settings/components/HeadroomProxyCard.tsx b/src/app/(dashboard)/dashboard/settings/components/HeadroomProxyCard.tsx new file mode 100644 index 0000000000..de37387cec --- /dev/null +++ b/src/app/(dashboard)/dashboard/settings/components/HeadroomProxyCard.tsx @@ -0,0 +1,235 @@ +"use client"; + +import { useCallback, useEffect, useRef, useState } from "react"; +import { useTranslations } from "next-intl"; +import { Card, Button, Input } from "@/shared/components"; +import { isHttpUrl } from "@/shared/validation/schemas/misc"; + +const HEADROOM_URL_MAX = 500; + +function isValidHeadroomUrl(value: string): boolean { + const trimmed = value.trim(); + if (trimmed === "") return true; + return trimmed.length <= HEADROOM_URL_MAX && isHttpUrl(trimmed); +} + +type SettingsErrorBody = { + error?: { + message?: string; + details?: { field?: string; message?: string }[]; + }; +}; + +function settingsErrorText(body: SettingsErrorBody, fallback: string): string { + const first = body.error?.details?.[0]; + if (first?.message) { + return first.field ? `${first.field}: ${first.message}` : first.message; + } + return body.error?.message || fallback; +} + +interface HeadroomStatus { + url?: string; + running?: boolean; + canStart?: boolean; + localUrl?: boolean; + installed?: boolean; +} + +export default function HeadroomProxyCard() { + const t = useTranslations("settings"); + const [url, setUrl] = useState(""); + const [loaded, setLoaded] = useState(false); + const [saving, setSaving] = useState(false); + const [acting, setActing] = useState(false); + const [status, setStatus] = useState(null); + const [msg, setMsg] = useState<{ ok: boolean; text: string } | null>(null); + const saveAc = useRef(null); + const lifecycleAc = useRef(null); + + const refreshStatus = useCallback(async (signal?: AbortSignal) => { + const res = await fetch("/api/headroom/status", signal ? { signal } : undefined); + if (!res.ok) return; + const data = (await res.json()) as HeadroomStatus; + if (signal?.aborted) return; + setStatus(data); + }, []); + + useEffect(() => { + const ac = new AbortController(); + // Async continuation so every setState happens after an await + // (react-hooks/set-state-in-effect: no synchronous setState in effect bodies). + void (async () => { + try { + const r = await fetch("/api/settings", { signal: ac.signal }); + const data = (r.ok ? await r.json() : {}) as Record; + if (ac.signal.aborted) return; + if (typeof data.headroomUrl === "string") setUrl(data.headroomUrl); + } catch { + // ignore + } finally { + if (!ac.signal.aborted) setLoaded(true); + } + // Status is for start/stop buttons only. Do not copy status.url into the + // input -- that value is HEADROOM_URL fallback and would overwrite empty. + try { + await refreshStatus(ac.signal); + } catch { + // ignore + } + })(); + return () => { + ac.abort(); + saveAc.current?.abort(); + lifecycleAc.current?.abort(); + }; + }, [refreshStatus]); + + const save = useCallback(async () => { + if (!isValidHeadroomUrl(url)) { + setMsg({ ok: false, text: t("cliproxyapiInvalidUrl") }); + return; + } + saveAc.current?.abort(); + const ac = new AbortController(); + saveAc.current = ac; + const { signal } = ac; + setSaving(true); + setMsg(null); + const trimmed = url.trim(); + try { + const res = await fetch("/api/settings", { + method: "PATCH", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ headroomUrl: trimmed }), + signal, + }); + if (!res.ok) { + const body = (await res.json().catch(() => ({}))) as SettingsErrorBody; + throw new Error(settingsErrorText(body, `HTTP ${res.status}`)); + } + if (signal.aborted) return; + setUrl(trimmed); + setMsg({ ok: true, text: t("settingSaved") }); + } catch (error) { + if (signal.aborted) return; + setMsg({ + ok: false, + text: error instanceof Error ? error.message : t("settingSaveFailed"), + }); + return; + } finally { + if (saveAc.current === ac) setSaving(false); + } + try { + await refreshStatus(signal); + } catch { + // PATCH already succeeded; status is best-effort. + } + }, [url, t, refreshStatus]); + + const postLifecycle = useCallback( + async (path: "/api/headroom/start" | "/api/headroom/stop") => { + lifecycleAc.current?.abort(); + const ac = new AbortController(); + lifecycleAc.current = ac; + const { signal } = ac; + setActing(true); + setMsg(null); + try { + const res = await fetch(path, { method: "POST", signal }); + if (!res.ok) { + const body = (await res.json().catch(() => ({}))) as SettingsErrorBody; + throw new Error(settingsErrorText(body, `HTTP ${res.status}`)); + } + try { + await refreshStatus(signal); + } catch { + // start/stop already succeeded; status is best-effort. + } + } catch (error) { + if (signal.aborted) return; + setMsg({ + ok: false, + text: error instanceof Error ? error.message : t("settingSaveFailed"), + }); + } finally { + if (lifecycleAc.current === ac) setActing(false); + } + }, + [refreshStatus, t] + ); + + if (!loaded) return null; + + const canStart = status?.canStart === true; + const running = status?.running === true; + const busy = saving || acting; + + return ( + +
+
+ compress +
+
+

{t("headroomProxyTitle")}

+

{t("headroomProxyDesc")}

+
+
+ + {msg && ( +
+ + {msg.ok ? "check_circle" : "error"} + + {msg.text} +
+ )} + +
+
+ + setUrl(e.target.value)} + placeholder="http://localhost:8787" + className="w-full" + disabled={busy} + /> +

{t("headroomProxyUrlHint")}

+
+
+ + + +
+ {status && !canStart && !status.localUrl && ( +

{t("headroomProxyExternalHint")}

+ )} +
+
+ ); +} diff --git a/src/i18n/messages/ar.json b/src/i18n/messages/ar.json index d33a7633bf..00fd85e21c 100644 --- a/src/i18n/messages/ar.json +++ b/src/i18n/messages/ar.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "الصحة", "cliproxyapiPort": "منفذ", "qdrantHost": "مضيف", - "qdrantCollection": "مجموعة" + "qdrantCollection": "مجموعة", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "محرك آر تي كيه", diff --git a/src/i18n/messages/az.json b/src/i18n/messages/az.json index a50b88762f..7d3eed8c7e 100644 --- a/src/i18n/messages/az.json +++ b/src/i18n/messages/az.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Sağlamlıq", "cliproxyapiPort": "Port", "qdrantHost": "Ev sahibi", - "qdrantCollection": "Kolleksiya" + "qdrantCollection": "Kolleksiya", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/bg.json b/src/i18n/messages/bg.json index 8086f96fc5..b83456a253 100644 --- a/src/i18n/messages/bg.json +++ b/src/i18n/messages/bg.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Здраве", "cliproxyapiPort": "Порт", "qdrantHost": "Хост", - "qdrantCollection": "Колекция" + "qdrantCollection": "Колекция", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/bn.json b/src/i18n/messages/bn.json index 507007dc7b..767b25ffd6 100644 --- a/src/i18n/messages/bn.json +++ b/src/i18n/messages/bn.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "স্বাস্থ্য", "cliproxyapiPort": "পোর্ট", "qdrantHost": "হোস্ট", - "qdrantCollection": "সংগ্রহ" + "qdrantCollection": "সংগ্রহ", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/cs.json b/src/i18n/messages/cs.json index f68aa4b51d..d23ce1eb61 100644 --- a/src/i18n/messages/cs.json +++ b/src/i18n/messages/cs.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Zdraví", "cliproxyapiPort": "Port", "qdrantHost": "Host", - "qdrantCollection": "Kolekce" + "qdrantCollection": "Kolekce", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/da.json b/src/i18n/messages/da.json index f326e8f313..5c26437fb6 100644 --- a/src/i18n/messages/da.json +++ b/src/i18n/messages/da.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Sundhed", "cliproxyapiPort": "Port", "qdrantHost": "Vært", - "qdrantCollection": "Samling" + "qdrantCollection": "Samling", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/de.json b/src/i18n/messages/de.json index 4b51536213..6ded55b40a 100644 --- a/src/i18n/messages/de.json +++ b/src/i18n/messages/de.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Gesundheit", "cliproxyapiPort": "Port", "qdrantHost": "Host", - "qdrantCollection": "Sammlung" + "qdrantCollection": "Sammlung", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index 9424b64ed7..dbfd697cd4 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -7910,6 +7910,14 @@ "cliproxyapiFallback": "CLIProxyAPI Fallback", "cliproxyapiEnableFallback": "Enable CLIProxyAPI Fallback", "cliproxyapiUrl": "CLIProxyAPI URL", + "headroomProxyTitle": "Headroom proxy", + "headroomProxyDesc": "URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "Headroom URL", + "headroomProxyUrlHint": "Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "Save", + "headroomProxyStart": "Start", + "headroomProxyStop": "Stop", + "headroomProxyExternalHint": "This URL is not loopback, so OmniRoute will not spawn the local CLI.", "cliproxyapiStatus": "CLIProxyAPI Status", "cliproxyapiNotDetected": "Not detected", "cliproxyapiImportAuthTitle": "Import accounts from CLIProxyAPI", diff --git a/src/i18n/messages/es.json b/src/i18n/messages/es.json index b1fbcc4fe6..c9ca5f5f3e 100644 --- a/src/i18n/messages/es.json +++ b/src/i18n/messages/es.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Salud", "cliproxyapiPort": "Puerto", "qdrantHost": "Anfitrión", - "qdrantCollection": "Colección" + "qdrantCollection": "Colección", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/fa.json b/src/i18n/messages/fa.json index b57be44976..3186d8a26a 100644 --- a/src/i18n/messages/fa.json +++ b/src/i18n/messages/fa.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "سلامت", "cliproxyapiPort": "پورت", "qdrantHost": "میزبان", - "qdrantCollection": "مجموعه" + "qdrantCollection": "مجموعه", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/fi.json b/src/i18n/messages/fi.json index 0b220d98b7..3b40f11caf 100644 --- a/src/i18n/messages/fi.json +++ b/src/i18n/messages/fi.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Terveys", "cliproxyapiPort": "Portti", "qdrantHost": "Isäntä", - "qdrantCollection": "Kokoelma" + "qdrantCollection": "Kokoelma", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/fr.json b/src/i18n/messages/fr.json index 59cae55660..bee6010d70 100644 --- a/src/i18n/messages/fr.json +++ b/src/i18n/messages/fr.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Santé", "cliproxyapiPort": "Port", "qdrantHost": "Hôte", - "qdrantCollection": "Collection" + "qdrantCollection": "Collection", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/gu.json b/src/i18n/messages/gu.json index 152237a606..4becbd4de1 100644 --- a/src/i18n/messages/gu.json +++ b/src/i18n/messages/gu.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "આરોગ્ય", "cliproxyapiPort": "પોર્ટ", "qdrantHost": "હોસ્ટ", - "qdrantCollection": "સંગ્રહ" + "qdrantCollection": "સંગ્રહ", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/he.json b/src/i18n/messages/he.json index 7f45b461c0..ca44b00147 100644 --- a/src/i18n/messages/he.json +++ b/src/i18n/messages/he.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "בריאות", "cliproxyapiPort": "פורט", "qdrantHost": "מארח", - "qdrantCollection": "אוסף" + "qdrantCollection": "אוסף", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/hi.json b/src/i18n/messages/hi.json index 3867eeb43a..9730073a62 100644 --- a/src/i18n/messages/hi.json +++ b/src/i18n/messages/hi.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "स्वास्थ्य", "cliproxyapiPort": "पोर्ट", "qdrantHost": "होस्ट", - "qdrantCollection": "संग्रह" + "qdrantCollection": "संग्रह", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/hu.json b/src/i18n/messages/hu.json index 5334482837..d033965d4f 100644 --- a/src/i18n/messages/hu.json +++ b/src/i18n/messages/hu.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Egészség", "cliproxyapiPort": "Port", "qdrantHost": "Gazda", - "qdrantCollection": "Gyűjtemény" + "qdrantCollection": "Gyűjtemény", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/id.json b/src/i18n/messages/id.json index 545e9a5a41..12ce90c987 100644 --- a/src/i18n/messages/id.json +++ b/src/i18n/messages/id.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Kesehatan", "cliproxyapiPort": "Port", "qdrantHost": "Host", - "qdrantCollection": "Koleksi" + "qdrantCollection": "Koleksi", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/it.json b/src/i18n/messages/it.json index dc01def708..e5a5761760 100644 --- a/src/i18n/messages/it.json +++ b/src/i18n/messages/it.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Salute", "cliproxyapiPort": "Porta", "qdrantHost": "Host", - "qdrantCollection": "Collezione" + "qdrantCollection": "Collezione", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/ja.json b/src/i18n/messages/ja.json index ba8c064462..a4162c3b15 100644 --- a/src/i18n/messages/ja.json +++ b/src/i18n/messages/ja.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "健康", "cliproxyapiPort": "ポート", "qdrantHost": "ホスト", - "qdrantCollection": "コレクション" + "qdrantCollection": "コレクション", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/ko.json b/src/i18n/messages/ko.json index ab4b4a67f7..64275df522 100644 --- a/src/i18n/messages/ko.json +++ b/src/i18n/messages/ko.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "건강", "cliproxyapiPort": "포트", "qdrantHost": "호스트", - "qdrantCollection": "컬렉션" + "qdrantCollection": "컬렉션", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/mr.json b/src/i18n/messages/mr.json index 53af82d57a..d340f63917 100644 --- a/src/i18n/messages/mr.json +++ b/src/i18n/messages/mr.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "आरोग्य", "cliproxyapiPort": "पोर्ट", "qdrantHost": "होस्ट", - "qdrantCollection": "संग्रह" + "qdrantCollection": "संग्रह", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/ms.json b/src/i18n/messages/ms.json index 05b504ffde..f6b839fd7e 100644 --- a/src/i18n/messages/ms.json +++ b/src/i18n/messages/ms.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Kesihatan", "cliproxyapiPort": "Pelabuhan", "qdrantHost": "Hos", - "qdrantCollection": "Koleksi" + "qdrantCollection": "Koleksi", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/nl.json b/src/i18n/messages/nl.json index c50a7beebc..db912312d2 100644 --- a/src/i18n/messages/nl.json +++ b/src/i18n/messages/nl.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Gezondheid", "cliproxyapiPort": "Haven", "qdrantHost": "Host", - "qdrantCollection": "Verzameling" + "qdrantCollection": "Verzameling", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/no.json b/src/i18n/messages/no.json index 5f7a1c64b0..4394a1e27b 100644 --- a/src/i18n/messages/no.json +++ b/src/i18n/messages/no.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Helse", "cliproxyapiPort": "Port", "qdrantHost": "Vert", - "qdrantCollection": "Samling" + "qdrantCollection": "Samling", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/phi.json b/src/i18n/messages/phi.json index 3a3080eded..89183a593d 100644 --- a/src/i18n/messages/phi.json +++ b/src/i18n/messages/phi.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Kalusugan", "cliproxyapiPort": "Port", "qdrantHost": "Host", - "qdrantCollection": "Koleksyon" + "qdrantCollection": "Koleksyon", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/pl.json b/src/i18n/messages/pl.json index bd07b8321e..8f5c91da84 100644 --- a/src/i18n/messages/pl.json +++ b/src/i18n/messages/pl.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Zdrowie", "cliproxyapiPort": "Port", "qdrantHost": "Host", - "qdrantCollection": "Kolekcja" + "qdrantCollection": "Kolekcja", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "Silnik RTK", diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index 5199f22e2c..0199b1b463 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -8344,7 +8344,15 @@ "cliproxyapiHealth": "Saúde", "cliproxyapiPort": "Porta", "qdrantHost": "Host", - "qdrantCollection": "Coleção" + "qdrantCollection": "Coleção", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "Motor RTK", diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index 3533f6370a..6cd1a1aa6c 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Saúde", "cliproxyapiPort": "Porto", "qdrantHost": "Anfitrião", - "qdrantCollection": "Coleção" + "qdrantCollection": "Coleção", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "Motor RTK", diff --git a/src/i18n/messages/ro.json b/src/i18n/messages/ro.json index afe466f408..4f445c4476 100644 --- a/src/i18n/messages/ro.json +++ b/src/i18n/messages/ro.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Sănătate", "cliproxyapiPort": "Port", "qdrantHost": "Gazdă", - "qdrantCollection": "Colecție" + "qdrantCollection": "Colecție", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/ru.json b/src/i18n/messages/ru.json index dbd845f383..d740cc5faf 100644 --- a/src/i18n/messages/ru.json +++ b/src/i18n/messages/ru.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Здоровье", "cliproxyapiPort": "Порт", "qdrantHost": "Хост", - "qdrantCollection": "Коллекция" + "qdrantCollection": "Коллекция", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/sk.json b/src/i18n/messages/sk.json index 40f2003b77..5dcf60ac13 100644 --- a/src/i18n/messages/sk.json +++ b/src/i18n/messages/sk.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Zdravie", "cliproxyapiPort": "Port", "qdrantHost": "Host", - "qdrantCollection": "Zbierka" + "qdrantCollection": "Zbierka", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/sv.json b/src/i18n/messages/sv.json index 175c8e1f97..bc502df18d 100644 --- a/src/i18n/messages/sv.json +++ b/src/i18n/messages/sv.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Hälsa", "cliproxyapiPort": "Port", "qdrantHost": "Värd", - "qdrantCollection": "Samling" + "qdrantCollection": "Samling", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/sw.json b/src/i18n/messages/sw.json index b8288d3863..7301fb6774 100644 --- a/src/i18n/messages/sw.json +++ b/src/i18n/messages/sw.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Afya", "cliproxyapiPort": "Bandari", "qdrantHost": "Mwenyeji", - "qdrantCollection": "Mkusanyiko" + "qdrantCollection": "Mkusanyiko", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/ta.json b/src/i18n/messages/ta.json index e0614361fc..346590dc7f 100644 --- a/src/i18n/messages/ta.json +++ b/src/i18n/messages/ta.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "ஆரோக்கியம்", "cliproxyapiPort": "போர்ட்", "qdrantHost": "விருந்தினர்", - "qdrantCollection": "கலெக்ஷன்" + "qdrantCollection": "கலெக்ஷன்", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/te.json b/src/i18n/messages/te.json index 6dfa5290b7..5cbdf5c885 100644 --- a/src/i18n/messages/te.json +++ b/src/i18n/messages/te.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "ఆరోగ్యం", "cliproxyapiPort": "పోర్ట్", "qdrantHost": "హోస్ట్", - "qdrantCollection": "సేకరణ" + "qdrantCollection": "సేకరణ", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/th.json b/src/i18n/messages/th.json index 126df99e1b..68f90b88a6 100644 --- a/src/i18n/messages/th.json +++ b/src/i18n/messages/th.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "สุขภาพ", "cliproxyapiPort": "พอร์ต", "qdrantHost": "โฮสต์", - "qdrantCollection": "การรวบรวม" + "qdrantCollection": "การรวบรวม", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/tr.json b/src/i18n/messages/tr.json index 5ee309e90d..6ed5f68933 100644 --- a/src/i18n/messages/tr.json +++ b/src/i18n/messages/tr.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Sağlık", "cliproxyapiPort": "Port", "qdrantHost": "Ana Bilgisayar", - "qdrantCollection": "Koleksiyon" + "qdrantCollection": "Koleksiyon", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/uk-UA.json b/src/i18n/messages/uk-UA.json index e1203b1d1b..5601b38668 100644 --- a/src/i18n/messages/uk-UA.json +++ b/src/i18n/messages/uk-UA.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "Здоров'я", "cliproxyapiPort": "Порт", "qdrantHost": "Хост", - "qdrantCollection": "Колекція" + "qdrantCollection": "Колекція", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "Двигун RTK", diff --git a/src/i18n/messages/ur.json b/src/i18n/messages/ur.json index fdc5391976..e22f742090 100644 --- a/src/i18n/messages/ur.json +++ b/src/i18n/messages/ur.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "صحت", "cliproxyapiPort": "پورٹ", "qdrantHost": "میزبان", - "qdrantCollection": "اجتماع" + "qdrantCollection": "اجتماع", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/vi.json b/src/i18n/messages/vi.json index 874cab83b8..d0a948d93b 100644 --- a/src/i18n/messages/vi.json +++ b/src/i18n/messages/vi.json @@ -8344,7 +8344,15 @@ "cliproxyapiHealth": "Sức Khỏe", "cliproxyapiPort": "Cổng", "qdrantHost": "Máy chủ", - "qdrantCollection": "Bộ Sưu Tập" + "qdrantCollection": "Bộ Sưu Tập", + "headroomProxyTitle": "Proxy Headroom", + "headroomProxyDesc": "URL của proxy tiết kiệm token Headroom (tùy chọn). Để trống thì dùng HEADROOM_URL hoặc http://localhost:8787.", + "headroomProxyUrl": "URL Headroom", + "headroomProxyUrlHint": "URL loopback có thể khởi chạy từ trang này. URL bên ngoài chỉ được kiểm tra.", + "headroomProxySave": "Lưu", + "headroomProxyStart": "Bắt đầu", + "headroomProxyStop": "Dừng", + "headroomProxyExternalHint": "URL này không phải loopback, nên OmniRoute sẽ không khởi chạy CLI cục bộ." }, "contextRtk": { "title": "RTK Engine", diff --git a/src/i18n/messages/zh-CN.json b/src/i18n/messages/zh-CN.json index 4306b98ac2..8501b6d0b4 100644 --- a/src/i18n/messages/zh-CN.json +++ b/src/i18n/messages/zh-CN.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "健康", "cliproxyapiPort": "端口", "qdrantHost": "主机", - "qdrantCollection": "集合" + "qdrantCollection": "集合", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "命令输出过滤引擎", diff --git a/src/i18n/messages/zh-TW.json b/src/i18n/messages/zh-TW.json index 1a4db6102b..66689e9057 100644 --- a/src/i18n/messages/zh-TW.json +++ b/src/i18n/messages/zh-TW.json @@ -8340,7 +8340,15 @@ "cliproxyapiHealth": "健康", "cliproxyapiPort": "埠", "qdrantHost": "主機", - "qdrantCollection": "集合" + "qdrantCollection": "集合", + "headroomProxyTitle": "__MISSING__:Headroom proxy", + "headroomProxyDesc": "__MISSING__:URL of the optional Headroom token-saver proxy. Empty uses HEADROOM_URL or http://localhost:8787.", + "headroomProxyUrl": "__MISSING__:Headroom URL", + "headroomProxyUrlHint": "__MISSING__:Loopback URLs can be started from this page. External URLs are probed only.", + "headroomProxySave": "__MISSING__:Save", + "headroomProxyStart": "__MISSING__:Start", + "headroomProxyStop": "__MISSING__:Stop", + "headroomProxyExternalHint": "__MISSING__:This URL is not loopback, so OmniRoute will not spawn the local CLI." }, "contextRtk": { "title": "RTK 引擎", diff --git a/src/shared/validation/settingsSchemas.ts b/src/shared/validation/settingsSchemas.ts index 55f1c49645..a07651d587 100644 --- a/src/shared/validation/settingsSchemas.ts +++ b/src/shared/validation/settingsSchemas.ts @@ -23,6 +23,7 @@ import { SPAWN_CAPABLE_PREFIXES, SPAWN_CAPABLE_PATTERN_ANCESTORS, } from "@/shared/constants/spawnCapablePrefixes"; +import { isHttpUrl } from "@/shared/validation/schemas/misc"; const signatureCacheModeValues = ["enabled", "bypass", "bypass-strict"] as const; @@ -493,6 +494,26 @@ export const updateSettingsSchema = z.object({ // CLIProxyAPI connection settings cliproxyapi_fallback_enabled: z.boolean().optional(), cliproxyapi_url: z.string().url().max(500).optional(), + // #12306: external Headroom proxy URL. Empty = fall back to HEADROOM_URL / localhost:8787. + // Status/start already read this key; without the schema field PATCH strips it. + // Trim first so a padded URL matches the client (isValidHeadroomUrl trims) + // and whitespace-only becomes the empty fallback, not "Invalid URL". + // z.string().url() also accepts javascript:/data:/file:. probeProxyRunning + // interpolates this into fetch(`${url}/health`), so restrict to http(s). + headroomUrl: z + .string() + .trim() + .pipe( + z.union([ + z.literal(""), + z + .string() + .url() + .max(500) + .refine((value) => isHttpUrl(value), "must be an http(s) URL"), + ]) + ) + .optional(), cliproxyapi_fallback_codes: z.string().max(200).optional(), // #7645: dedicated CLIProxyAPI credential. CLIProxyAPI requires its own // separately-configured `api-keys:` credential and rejects any other token diff --git a/tests/unit/headroom-url-settings-12306.test.ts b/tests/unit/headroom-url-settings-12306.test.ts new file mode 100644 index 0000000000..e9c89c8638 --- /dev/null +++ b/tests/unit/headroom-url-settings-12306.test.ts @@ -0,0 +1,153 @@ +/** + * #12306: settings.headroomUrl must survive PATCH /api/settings. + * + * Status/start already READ settings.headroomUrl. Without the schema + * field Zod strips the key and the write path is a no-op. + */ +import { test, after } from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const testDataDir = fs.mkdtempSync(path.join(os.tmpdir(), "omni-12306-headroom-")); +const originalDataDir = process.env.DATA_DIR; +process.env.DATA_DIR = testDataDir; + +const { updateSettingsSchema } = await import("../../src/shared/validation/settingsSchemas.ts"); +const coreDb = await import("../../src/lib/db/core.ts"); +const settingsDb = await import("../../src/lib/db/settings.ts"); + +after(() => { + coreDb.resetDbInstance(); + if (fs.existsSync(testDataDir)) { + fs.rmSync(testDataDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + } + if (originalDataDir === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = originalDataDir; +}); + +test("updateSettingsSchema keeps a valid headroomUrl", () => { + const parsed = updateSettingsSchema.parse({ + headroomUrl: "http://127.0.0.1:8787", + }); + assert.equal(parsed.headroomUrl, "http://127.0.0.1:8787"); +}); + +test("updateSettingsSchema accepts an empty headroomUrl to fall back to HEADROOM_URL", () => { + const parsed = updateSettingsSchema.parse({ headroomUrl: "" }); + assert.equal(parsed.headroomUrl, ""); +}); + +test("updateSettingsSchema trims a padded headroomUrl before validating", () => { + const parsed = updateSettingsSchema.parse({ + headroomUrl: " http://headroom.internal:9090 ", + }); + assert.equal(parsed.headroomUrl, "http://headroom.internal:9090"); +}); + +test("updateSettingsSchema treats whitespace-only headroomUrl as empty", () => { + const parsed = updateSettingsSchema.parse({ headroomUrl: " " }); + assert.equal(parsed.headroomUrl, ""); +}); + +test("updateSettingsSchema rejects a non-URL headroomUrl", () => { + const result = updateSettingsSchema.safeParse({ headroomUrl: "not-a-url" }); + assert.equal(result.success, false); +}); + +test("updateSettingsSchema rejects non-http(s) headroomUrl schemes", () => { + for (const url of [ + "javascript:alert(1)", + "ftp://x", + "data:text/html,x", + "file:///etc/passwd", + "http://", + "http://[", + "http:", + ]) { + const result = updateSettingsSchema.safeParse({ headroomUrl: url }); + assert.equal(result.success, false, url); + } +}); + +test("updateSettingsSchema rejects a headroomUrl over 500 chars", () => { + const result = updateSettingsSchema.safeParse({ + headroomUrl: `http://example.com/${"x".repeat(500)}`, + }); + assert.equal(result.success, false); +}); + +test("updateSettings round-trips a validated headroomUrl", async () => { + await coreDb.ensureDbInitialized(); + const parsed = updateSettingsSchema.parse({ + headroomUrl: "http://headroom.internal:9090", + }); + await settingsDb.updateSettings(parsed); + const stored = await settingsDb.getSettings(); + assert.equal(stored.headroomUrl, "http://headroom.internal:9090"); +}); + +test("updateSettings round-trips an empty headroomUrl without dropping the key", async () => { + await coreDb.ensureDbInitialized(); + await settingsDb.updateSettings( + updateSettingsSchema.parse({ headroomUrl: "http://headroom.internal:9090" }) + ); + await settingsDb.updateSettings(updateSettingsSchema.parse({ headroomUrl: "" })); + const stored = await settingsDb.getSettings(); + assert.equal(stored.headroomUrl, ""); +}); + +test("advanced settings page mounts the Headroom proxy card", async () => { + const src = fs.readFileSync( + path.join(import.meta.dirname, "../../src/app/(dashboard)/dashboard/settings/advanced/page.tsx"), + "utf8" + ); + assert.match(src, /HeadroomProxyCard/); +}); + +test("after() restores DATA_DIR so later files in the same process keep their own dir", () => { + const src = fs.readFileSync(new URL(import.meta.url), "utf8"); + assert.match(src, /const originalDataDir = process\.env\.DATA_DIR/); + assert.match(src, /if \(originalDataDir === undefined\) delete process\.env\.DATA_DIR/); +}); + +test("save reads PATCH validation details instead of a generic HTTP status", () => { + const src = fs.readFileSync( + path.join( + import.meta.dirname, + "../../src/app/(dashboard)/dashboard/settings/components/HeadroomProxyCard.tsx" + ), + "utf8" + ); + assert.match(src, /throw new Error\(settingsErrorText\(body, `HTTP \$\{res\.status\}`\)\)/); + assert.equal( + (src.match(/throw new Error\(settingsErrorText\(body, `HTTP \$\{res\.status\}`\)\)/g) || []).length, + 2 + ); + assert.match(src, /body\.error\?\.details/); + assert.match(src, /isHttpUrl/); + assert.match(src, /const HEADROOM_URL_MAX = 500/); + assert.match(src, /setUrl\(trimmed\)/); + assert.match(src, /const saveAc = useRef\(null\)/); + assert.match(src, /const lifecycleAc = useRef\(null\)/); + assert.match(src, /saveAc\.current = ac/); + assert.match(src, /lifecycleAc\.current = ac/); + assert.match(src, /await fetch\(path, \{ method: "POST", signal \}\)/); + assert.match(src, /body: JSON.stringify\(\{ headroomUrl: trimmed \}\),\s*signal,/s); + assert.match(src, /\/\/ start\/stop already succeeded; status is best-effort\./); + // Busy flags: clear only if this invocation still owns the controller. + // A second click replaces the ref; the first finally must not unlock. + assert.match(src, /if \(saveAc\.current === ac\) setSaving\(false\)/); + assert.match(src, /if \(lifecycleAc\.current === ac\) setActing\(false\)/); + assert.doesNotMatch(src, /if \(!signal\.aborted\) setSaving\(false\)/); + assert.doesNotMatch(src, /if \(!signal\.aborted\) setActing\(false\)/); + assert.match( + src, + /return \(\) => \{\s*ac\.abort\(\);\s*saveAc\.current\?\.abort\(\);\s*lifecycleAc\.current\?\.abort\(\);/s + ); + assert.match(src, /const busy = saving \|\| acting;/); + assert.match(src, /disabled=\{busy\}/); + assert.match(src, /disabled=\{busy \|\| !canStart\}/); + assert.match(src, /disabled=\{busy \|\| !running\}/); +}); From f81ce2a23b10495d931fa17ca5f02320bda0a554 Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Thu, 3 Sep 2026 11:40:02 -0400 Subject: [PATCH 037/143] feat(dashboard): adaptive context-budget dial on compression panel (#12488) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check:provider-consistency` OK (273 entradas REGISTRY, **356** providers canônicos), `check-docs-counts-sync` exit 0 e **300/300** nos testes que a leva toca. O crescimento de arquivo que os PRs empilham uns sobre os outros foi rebaselinado num único registro datado (`_rebaseline_2026_09_03_houminxi_batch`), com a decomposição por arquivo: `providers/page.tsx` +18 (import CSV do #12504 + busca do #12495 no mesmo painel), `accountFallback.ts` +6 (o #12566 sobre o rebaseline que o #12590 já registrou — os dois tocam `checkFallbackError`) e `chatCore.ts` +3 (invalidez de cache de quota no 429 do #12325). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. --- CHANGELOG.md | 1 + docs/compression/COMPRESSION_GUIDE.md | 2 +- .../context/settings/CompressionPanel.tsx | 176 ++++- src/i18n/messages/ar.json | 8 + src/i18n/messages/az.json | 8 + src/i18n/messages/bg.json | 8 + src/i18n/messages/bn.json | 8 + src/i18n/messages/cs.json | 8 + src/i18n/messages/da.json | 8 + src/i18n/messages/de.json | 8 + src/i18n/messages/en.json | 8 + src/i18n/messages/es.json | 8 + src/i18n/messages/fa.json | 8 + src/i18n/messages/fi.json | 8 + src/i18n/messages/fr.json | 8 + src/i18n/messages/gu.json | 8 + src/i18n/messages/he.json | 8 + src/i18n/messages/hi.json | 8 + src/i18n/messages/hu.json | 8 + src/i18n/messages/id.json | 8 + src/i18n/messages/it.json | 8 + src/i18n/messages/ja.json | 8 + src/i18n/messages/ko.json | 8 + src/i18n/messages/mr.json | 8 + src/i18n/messages/ms.json | 8 + src/i18n/messages/nl.json | 8 + src/i18n/messages/no.json | 8 + src/i18n/messages/phi.json | 8 + src/i18n/messages/pl.json | 8 + src/i18n/messages/pt-BR.json | 8 + src/i18n/messages/pt.json | 8 + src/i18n/messages/ro.json | 8 + src/i18n/messages/ru.json | 8 + src/i18n/messages/sk.json | 8 + src/i18n/messages/sv.json | 8 + src/i18n/messages/sw.json | 8 + src/i18n/messages/ta.json | 8 + src/i18n/messages/te.json | 8 + src/i18n/messages/th.json | 8 + src/i18n/messages/tr.json | 8 + src/i18n/messages/uk-UA.json | 8 + src/i18n/messages/ur.json | 8 + src/i18n/messages/vi.json | 8 + src/i18n/messages/zh-CN.json | 8 + src/i18n/messages/zh-TW.json | 8 + .../ui/compressionAdaptiveBudgetDial.test.tsx | 660 ++++++++++++++++++ 46 files changed, 1142 insertions(+), 33 deletions(-) create mode 100644 tests/unit/ui/compressionAdaptiveBudgetDial.test.tsx diff --git a/CHANGELOG.md b/CHANGELOG.md index d6fd097613..bfcad514f4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,7 @@ ### ✨ New Features +- **feat(dashboard):** adaptive context-budget dial on the compression settings panel — mode (`off` / `floor` / `replace-autotrigger`) and policy (`reserve-output` / `percentage` / `absolute`) persist via `PUT /api/settings/compression` `contextBudget`. Completes the dashboard half of #7005 (API + DB already shipped in #7183). - **feat(sse): STRICT_ZERO_COST** — opt-in, off-by-default `freeAccessPolicy: "strict"` setting that hard-verifies every auto-combo candidate against live quota state and per-connection economic safety before it can be dispatched, going beyond `hidePaidModels`'s static catalog diff --git a/docs/compression/COMPRESSION_GUIDE.md b/docs/compression/COMPRESSION_GUIDE.md index 8f51281fd6..bd2a9e665a 100644 --- a/docs/compression/COMPRESSION_GUIDE.md +++ b/docs/compression/COMPRESSION_GUIDE.md @@ -309,7 +309,7 @@ Every compressed request includes stats in the server logs: | Phase 2 | Standard, Aggressive, Ultra | ✅ Shipped | | Phase 3 | RTK, Stacked, Compression Combos | ✅ Shipped | | Phase 4 | Output Styles, SLM-tier Ultra, eval harness | ✅ Shipped | -| Phase 4C | Adaptive context-budget ("dial") — compute engine + API (`contextBudget` on `PUT /api/settings/compression`) | ✅ Shipped (API-configurable; dashboard controls not yet built, #7005) | +| Phase 4C | Adaptive context-budget ("dial") — compute engine + API (`contextBudget` on `PUT /api/settings/compression`) + dashboard mode/policy controls | ✅ Shipped | --- diff --git a/src/app/(dashboard)/dashboard/context/settings/CompressionPanel.tsx b/src/app/(dashboard)/dashboard/context/settings/CompressionPanel.tsx index 4d7b506171..31bf43fff6 100644 --- a/src/app/(dashboard)/dashboard/context/settings/CompressionPanel.tsx +++ b/src/app/(dashboard)/dashboard/context/settings/CompressionPanel.tsx @@ -3,12 +3,13 @@ // CompressionPanel — the single-source engine-grid UI for compression. // // Renders the master on/off switch, one row per catalog engine (on/off + level + -// link to its detail page), the cavemanOutput intensity row, the mcpAccessibility -// toggle (its own endpoint / separate store), a read-only derived-pipeline preview, -// and the general settings (auto-trigger tokens + preserve-system-prompt). +// link to its detail page), the adaptive context-budget dial, the cavemanOutput +// intensity row, the mcpAccessibility toggle (its own endpoint / separate store), +// a derived-pipeline preview, and the general settings (auto-trigger tokens + +// preserve-system-prompt). // import Link from "next/link"; -import { useEffect, useState } from "react"; +import { useEffect, useRef, useState } from "react"; import { useTranslations, useLocale } from "next-intl"; // Import Card/Toggle from their direct module paths rather than the @/shared/components // barrel: the barrel transitively pulls a heavy/Node-only module that hangs the @@ -60,12 +61,20 @@ interface CompressionConfig { // Best-effort pre-warm of the SLM model on enable / cold restart. Default false. ultraSlmPrewarm?: boolean; // Phase 4 (C): adaptive context-budget. Absent / mode:"off" = legacy auto-trigger. - // The panel currently surfaces the computed target read-only; mode/policy editors are a - // follow-up (the load/save path does not yet populate this field). contextBudget?: ContextBudgetConfig; liveZone?: { enabled: boolean }; } +const CONTEXT_BUDGET_MODES = new Set([ + "off", + "floor", + "replace-autotrigger", +]); +const CONTEXT_BUDGET_POLICIES = new Set([ + "reserve-output", + "percentage", + "absolute", +]); const CAVEMAN_OUTPUT_LEVELS: CavemanIntensity[] = ["lite", "full", "ultra"]; const DEFAULT_CONFIG: CompressionConfig = { @@ -78,6 +87,7 @@ const DEFAULT_CONFIG: CompressionConfig = { outputStyles: [], ultraEngine: "heuristic", ultraSlmPrewarm: false, + contextBudget: { ...DEFAULT_CONTEXT_BUDGET }, liveZone: { enabled: false }, }; @@ -120,22 +130,73 @@ function LiveZoneToggle({ ); } -function AdaptiveTargetPreview({ contextBudget }: { contextBudget?: ContextBudgetConfig }) { +function AdaptiveContextBudgetDial({ + contextBudget, + saving, + onChange, +}: { + contextBudget: ContextBudgetConfig; + saving: boolean; + onChange: (patch: Partial) => void; +}) { const t = useTranslations("settings"); - const target = getAdaptiveTargetSummary(contextBudget ?? DEFAULT_CONTEXT_BUDGET, 200000); + // Representative window for the preview label (D-C1). Not the live model limit — + // the panel has no selected-model context here; 200k is Claude-class default. + const target = getAdaptiveTargetSummary(contextBudget, 200000); return ( -
- {target.enabled - ? t("compressionAdaptiveTarget", { - mode: target.mode, - policy: target.policy, - target: target.target, - contextLimit: target.contextLimit, - }) - : t("compressionAdaptiveOff")} +
+ + {(contextBudget.mode ?? "off") !== "off" && ( + + )} +
+ {target.enabled + ? t("compressionAdaptiveTarget", { + mode: target.mode, + policy: target.policy, + target: target.target, + contextLimit: target.contextLimit, + }) + : t("compressionAdaptiveOff")} +
); } @@ -153,19 +214,29 @@ export default function CompressionPanel() { const [loading, setLoading] = useState(true); const [saving, setSaving] = useState(false); const [status, setStatus] = useState<"" | "saved" | "error">(""); + const configRef = useRef(config); + useEffect(() => { + configRef.current = config; + }, [config]); + const saveGenRef = useRef(0); + const lastConfirmedRef = useRef(config); + const lastAckedGenRef = useRef(0); useEffect(() => { fetch("/api/settings/compression") .then((r) => (r.ok ? r.json() : null)) .then((data: Partial | null) => { if (data) { - setConfig({ + const hydrated: CompressionConfig = { ...DEFAULT_CONFIG, ...data, engines: normalizeEngines(data.engines), cavemanOutputMode: data.cavemanOutputMode ?? DEFAULT_CONFIG.cavemanOutputMode, outputStyles: data.outputStyles ?? DEFAULT_CONFIG.outputStyles, - }); + contextBudget: { ...DEFAULT_CONTEXT_BUDGET, ...(data.contextBudget ?? {}) }, + }; + lastConfirmedRef.current = hydrated; + setConfig(hydrated); } }) .catch(() => {}) @@ -181,8 +252,24 @@ export default function CompressionPanel() { // Persist a merge-patch. The DB persists `engines` as one whole row, so callers that // touch an engine pass the full engines map to avoid dropping the other engines. + // Generation + configRef: a later in-flight save must not let an older failure + // roll back a newer optimistic (or already-acked) state. const save = async (updates: Partial) => { - const next = { ...config, ...updates }; + const gen = ++saveGenRef.current; + const previous = configRef.current; + const next: CompressionConfig = { + ...previous, + ...updates, + ...(updates.contextBudget + ? { + contextBudget: { + ...(previous.contextBudget ?? DEFAULT_CONTEXT_BUDGET), + ...updates.contextBudget, + }, + } + : {}), + }; + configRef.current = next; setConfig(next); setSaving(true); setStatus(""); @@ -192,16 +279,34 @@ export default function CompressionPanel() { headers: { "Content-Type": "application/json" }, body: JSON.stringify(updates), }); - if (res.ok) { - setStatus("saved"); - setTimeout(() => setStatus(""), 2000); - } else { - setStatus("error"); + // Acked server state is recorded even when this gen is stale, so a + // later failure rolls back to the newest acked PUT, not the GET. + // lastAckedGenRef stops an older ack from overwriting a newer one. + if (res.ok && gen >= lastAckedGenRef.current) { + lastConfirmedRef.current = next; + lastAckedGenRef.current = gen; + } + if (gen === saveGenRef.current) { + if (res.ok) { + setStatus("saved"); + const savedGen = gen; + setTimeout(() => { + if (savedGen === saveGenRef.current) setStatus(""); + }, 2000); + } else { + configRef.current = lastConfirmedRef.current; + setConfig(lastConfirmedRef.current); + setStatus("error"); + } } } catch { - setStatus("error"); + if (gen === saveGenRef.current) { + configRef.current = lastConfirmedRef.current; + setConfig(lastConfirmedRef.current); + setStatus("error"); + } } finally { - setSaving(false); + if (gen === saveGenRef.current) setSaving(false); } }; @@ -326,8 +431,15 @@ export default function CompressionPanel() { {derivedText}
- {/* Adaptive context-budget — read-only computed target (Phase 4C, D-C1 transparency) */} - + {/* Adaptive context-budget dial — mode/policy persist via PUT contextBudget */} + { + const current = configRef.current.contextBudget ?? DEFAULT_CONTEXT_BUDGET; + save({ contextBudget: { ...current, ...patch } }); + }} + /> {/* Engine grid */}
diff --git a/src/i18n/messages/ar.json b/src/i18n/messages/ar.json index 00fd85e21c..b987620b46 100644 --- a/src/i18n/messages/ar.json +++ b/src/i18n/messages/ar.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "الوضع: {mode}", "compressionAdaptiveOff": "ميزانية السياق التكيفية: معطلة (المشغل التلقائي القديم)", "compressionAdaptiveTarget": "تكيفي ({mode}، السياسة: {policy}) — الهدف ≈ {target, number} رمز (لنافذة من {contextLimit, number} رمز)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "حقن تعليمات تشكيل الاستجابة دون إعادة كتابة مخرجات المزود. ادمج بحرية.", "mcpAccessibilityDescription": "يحدد نطاق مخرجات أداة MCP (مخزن منفصل).", "compressionStylesTileSummary": "{tokens, number} رمز تم توفيره · {runs, plural, one {# تشغيل تم تنسيقه} other {# تشغيلات تم تنسيقها}}", diff --git a/src/i18n/messages/az.json b/src/i18n/messages/az.json index 7d3eed8c7e..33217b0d2b 100644 --- a/src/i18n/messages/az.json +++ b/src/i18n/messages/az.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "rejim: {mode}", "compressionAdaptiveOff": "Adaptiv kontekst büdcəsi: qapalı (köhnə avtomatik tətikləyici)", "compressionAdaptiveTarget": "Adaptiv ({mode}, siyasət: {policy}) — hədəf ≈ {target, number} token ({contextLimit, number}-tokenlik pəncərə üçün)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Provayder çıxışını yenidən yazmadan cavab-formalaşdıran təlimatları daxil edin. Sərbəst şəkildə birləşdirin.", "mcpAccessibilityDescription": "MCP alət çıxışlarını əhatə edir (ayrıca depo).", "compressionStylesTileSummary": "{tokens, number} tokenə qənaət edilib · {runs, plural, one {# işəsalma üslublaşdırılıb} other {# işəsalma üslublaşdırılıb}}", diff --git a/src/i18n/messages/bg.json b/src/i18n/messages/bg.json index b83456a253..3c3c942e6d 100644 --- a/src/i18n/messages/bg.json +++ b/src/i18n/messages/bg.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "режим: {mode}", "compressionAdaptiveOff": "Адаптивен бюджет за контекст: изключен (наследено автоматично задействане)", "compressionAdaptiveTarget": "Адаптивен ({mode}, политика: {policy}) — цел ≈ {target, number} токена (за прозорец от {contextLimit, number} токена)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Вмъкване на инструкции за оформяне на отговора без пренаписване на изхода от доставчика. Комбинирайте свободно.", "mcpAccessibilityDescription": "Ограничава обхвата на изходите от MCP инструменти (отделно хранилище).", "compressionStylesTileSummary": "{tokens, number} спестени токена · {runs, plural, one {# стилизирано изпълнение} other {# стилизирани изпълнения}}", diff --git a/src/i18n/messages/bn.json b/src/i18n/messages/bn.json index 767b25ffd6..36cee7c2d9 100644 --- a/src/i18n/messages/bn.json +++ b/src/i18n/messages/bn.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "মোড: {mode}", "compressionAdaptiveOff": "অ্যাডাপ্টিভ কনটেক্সট বাজেট: বন্ধ (লেগাসি অটো-ট্রিগার)", "compressionAdaptiveTarget": "অ্যাডাপ্টিভ ({mode}, পলিসি: {policy}) — টার্গেট ≈ {target, number} টোকেন ({contextLimit, number}-টোকেন উইন্ডোর জন্য)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "প্রোভাইডার আউটপুট রিরাইট না করেই রেসপন্স-শেপিং নির্দেশাবলী ইনজেক্ট করুন। অবাধে একত্রিত করুন।", "mcpAccessibilityDescription": "MCP টুল আউটপুট স্কোপ করে (আলাদা স্টোর)।", "compressionStylesTileSummary": "{tokens, number} টোকেন সাশ্রয় হয়েছে · {runs, plural, one {#টি রান স্টাইল করা হয়েছে} other {#টি রান স্টাইল করা হয়েছে}}", diff --git a/src/i18n/messages/cs.json b/src/i18n/messages/cs.json index d23ce1eb61..568a126332 100644 --- a/src/i18n/messages/cs.json +++ b/src/i18n/messages/cs.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "režim: {mode}", "compressionAdaptiveOff": "Adaptivní rozpočet kontextu: vypnuto (starší automatické spouštění)", "compressionAdaptiveTarget": "Adaptivní ({mode}, zásada: {policy}) — cíl ≈ {target, number} tokenů (pro okno o velikosti {contextLimit, number} tokenů)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Vkládejte instrukce pro formování odpovědi bez přepisování výstupu poskytovatele. Libovolně kombinujte.", "mcpAccessibilityDescription": "Omezuje rozsah výstupů nástrojů MCP (samostatné úložiště).", "compressionStylesTileSummary": "{tokens, number} ušetřených tokenů · {runs, plural, one {# stylované spuštění} few {# stylovaná spuštění} other {# stylovaných spuštění}}", diff --git a/src/i18n/messages/da.json b/src/i18n/messages/da.json index 5c26437fb6..2f19b8d75a 100644 --- a/src/i18n/messages/da.json +++ b/src/i18n/messages/da.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "tilstand: {mode}", "compressionAdaptiveOff": "Adaptivt kontekstbudget: fra (forældet auto-trigger)", "compressionAdaptiveTarget": "Adaptiv ({mode}, politik: {policy}) — mål ≈ {target, number} tokens (for et {contextLimit, number}-token vindue)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Indsæt instruktioner til formning af svar uden at omskrive udbyderens output. Kombiner frit.", "mcpAccessibilityDescription": "Afgrænser MCP-værktøjsoutput (separat lager).", "compressionStylesTileSummary": "{tokens, number} tokens sparet · {runs, plural, one {# kørsel stylet} other {# kørsler stylet}}", diff --git a/src/i18n/messages/de.json b/src/i18n/messages/de.json index 6ded55b40a..b6c9eccefc 100644 --- a/src/i18n/messages/de.json +++ b/src/i18n/messages/de.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "Modus: {mode}", "compressionAdaptiveOff": "Adaptives Kontextbudget: aus (Legacy-Auto-Trigger)", "compressionAdaptiveTarget": "Adaptiv ({mode}, Richtlinie: {policy}) — Ziel ≈ {target, number} Tokens (für ein {contextLimit, number}-Token-Fenster)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Anweisungen zur Antwortgestaltung einfügen, ohne die Provider-Ausgabe umzuschreiben. Frei kombinierbar.", "mcpAccessibilityDescription": "Schränkt MCP-Tool-Ausgaben ein (separater Speicher).", "compressionStylesTileSummary": "{tokens, number} Tokens eingespart · {runs, plural, one {# Ausführung gestylt} other {# Ausführungen gestylt}}", diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index dbfd697cd4..ff70e5c7bb 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -7815,6 +7815,14 @@ "compressionDerivedMode": "mode: {mode}", "compressionAdaptiveOff": "Adaptive context budget: off (legacy auto-trigger)", "compressionAdaptiveTarget": "Adaptive ({mode}, policy: {policy}) — target ≈ {target, number} tokens (for a {contextLimit, number}-token window)", + "compressionAdaptiveMode": "Adaptive context budget", + "compressionAdaptiveModeOff": "Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "Replace auto-trigger", + "compressionAdaptivePolicy": "Budget policy", + "compressionAdaptivePolicyReserve": "Reserve output", + "compressionAdaptivePolicyPercentage": "Percentage of window", + "compressionAdaptivePolicyAbsolute": "Absolute token budget", "compressionOutputStylesDescription": "Inject response-shaping instructions without rewriting provider output. Combine freely.", "mcpAccessibilityDescription": "Scopes MCP tool outputs (separate store).", "compressionStylesTileSummary": "{tokens, number} tokens saved · {runs, plural, one {# run styled} other {# runs styled}}", diff --git a/src/i18n/messages/es.json b/src/i18n/messages/es.json index c9ca5f5f3e..89f8137d05 100644 --- a/src/i18n/messages/es.json +++ b/src/i18n/messages/es.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "mode: {mode}", "compressionAdaptiveOff": "Adaptive context budget: off (legacy auto-trigger)", "compressionAdaptiveTarget": "Adaptive ({mode}, policy: {policy}) — target ≈ {target, number} tokens (for a {contextLimit, number}-token window)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Inject response-shaping instructions without rewriting provider output. Combine freely.", "mcpAccessibilityDescription": "Scopes MCP tool outputs (separate store).", "compressionStylesTileSummary": "{tokens, number} tokens saved · {runs, plural, one {# run styled} other {# runs styled}}", diff --git a/src/i18n/messages/fa.json b/src/i18n/messages/fa.json index 3186d8a26a..fb8fab5f8b 100644 --- a/src/i18n/messages/fa.json +++ b/src/i18n/messages/fa.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "حالت: {mode}", "compressionAdaptiveOff": "بودجه محتوای تطبیقی: خاموش (محرک خودکار قدیمی)", "compressionAdaptiveTarget": "تطبیقی ({mode}، خط‌مشی: {policy}) — هدف ≈ {target, number} توکن (برای یک پنجره {contextLimit, number} توکنی)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "تزریق دستورالعمل‌های شکل‌دهی به پاسخ بدون بازنویسی خروجی ارائه‌دهنده. ترکیب آزادانه.", "mcpAccessibilityDescription": "محدوده خروجی‌های ابزار MCP (ذخیره‌ساز مجزا).", "compressionStylesTileSummary": "{tokens, number} توکن ذخیره شد · {runs, plural, one {# اجرا سبک‌دهی شد} other {# اجرا سبک‌دهی شدند}}", diff --git a/src/i18n/messages/fi.json b/src/i18n/messages/fi.json index 3b40f11caf..75173454c9 100644 --- a/src/i18n/messages/fi.json +++ b/src/i18n/messages/fi.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "tila: {mode}", "compressionAdaptiveOff": "Mukautuva kontekstibudjetti: pois päältä (vanha automaattikäynnistys)", "compressionAdaptiveTarget": "Mukautuva ({mode}, käytäntö: {policy}) — tavoite ≈ {target, number} tokenia ({contextLimit, number} tokenin ikkunalle)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Lisää vastauksen muotoiluohjeita kirjoittamatta palveluntarjoajan tulostetta uudelleen. Yhdistele vapaasti.", "mcpAccessibilityDescription": "Rajaa MCP-työkalujen tulosteet (erillinen tallennustila).", "compressionStylesTileSummary": "{tokens, number} tokenia säästetty · {runs, plural, one {# ajo tyylitelty} other {# ajoa tyylitelty}}", diff --git a/src/i18n/messages/fr.json b/src/i18n/messages/fr.json index bee6010d70..e3d9dc6bab 100644 --- a/src/i18n/messages/fr.json +++ b/src/i18n/messages/fr.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "mode : {mode}", "compressionAdaptiveOff": "Budget de contexte adaptatif : désactivé (déclenchement automatique hérité)", "compressionAdaptiveTarget": "Adaptatif ({mode}, politique : {policy}) — cible ≈ {target, number} jetons (pour une fenêtre de {contextLimit, number} jetons)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Injecter des instructions de mise en forme des réponses sans réécrire la sortie du fournisseur. À combiner librement.", "mcpAccessibilityDescription": "Limite la portée des sorties d'outils MCP (magasin distinct).", "compressionStylesTileSummary": "{tokens, number} jetons économisés · {runs, plural, one {# exécution stylisée} other {# exécutions stylisées}}", diff --git a/src/i18n/messages/gu.json b/src/i18n/messages/gu.json index 4becbd4de1..21a08458e5 100644 --- a/src/i18n/messages/gu.json +++ b/src/i18n/messages/gu.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "મોડ: {mode}", "compressionAdaptiveOff": "અડેપ્ટિવ કન્ટેક્સ્ટ બજેટ: બંધ (લેગસી ઑટો-ટ્રિગર)", "compressionAdaptiveTarget": "અડેપ્ટિવ ({mode}, પૉલિસી: {policy}) — લક્ષ્ય ≈ {target, number} ટોકન્સ ({contextLimit, number}-ટોકન વિન્ડો માટે)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "પ્રોવાઇડર આઉટપુટને ફરીથી લખ્યા વિના રિસ્પોન્સ-શેપિંગ સૂચનાઓ ઇન્જેક્ટ કરો. મુક્તપણે જોડો.", "mcpAccessibilityDescription": "MCP ટૂલ આઉટપુટ્સને સ્કોપ કરે છે (અલગ સ્ટોર).", "compressionStylesTileSummary": "{tokens, number} ટોકન્સ સાચવ્યા · {runs, plural, one {# રન સ્ટાઇલ કરેલ} other {# રન સ્ટાઇલ કરેલ}}", diff --git a/src/i18n/messages/he.json b/src/i18n/messages/he.json index ca44b00147..5cbdf78924 100644 --- a/src/i18n/messages/he.json +++ b/src/i18n/messages/he.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "מצב: {mode}", "compressionAdaptiveOff": "תקציב הקשר אדפטיבי: כבוי (טריגר אוטומטי מיושן)", "compressionAdaptiveTarget": "אדפטיבי ({mode}, מדיניות: {policy}) — יעד ≈ {target, number} טוקנים (עבור חלון של {contextLimit, number} טוקנים)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "הזרקת הנחיות לעיצוב תגובה מבלי לשכתב את פלט הספק. ניתן לשלב באופן חופשי.", "mcpAccessibilityDescription": "מגביל את הטווח של פלטי כלי MCP (אחסון נפרד).", "compressionStylesTileSummary": "{tokens, number} טוקנים נחסכו · {runs, plural, one {הרצה אחת עוצבה} other {# הרצות עוצבו}}", diff --git a/src/i18n/messages/hi.json b/src/i18n/messages/hi.json index 9730073a62..844ea16376 100644 --- a/src/i18n/messages/hi.json +++ b/src/i18n/messages/hi.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "मोड: {mode}", "compressionAdaptiveOff": "अनुकूली संदर्भ बजट: बंद (लीगेसी ऑटो-ट्रिगर)", "compressionAdaptiveTarget": "अनुकूली ({mode}, नीति: {policy}) — लक्ष्य ≈ {target, number} टोकन ({contextLimit, number}-टोकन विंडो के लिए)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "प्रदाता आउटपुट को फिर से लिखे बिना प्रतिक्रिया-आकार देने वाले निर्देश इंजेक्ट करें। स्वतंत्र रूप से संयोजित करें।", "mcpAccessibilityDescription": "MCP टूल आउटपुट को स्कोप करता है (अलग स्टोर)।", "compressionStylesTileSummary": "{tokens, number} टोकन बचाए गए · {runs, plural, one {# रन स्टाइल किया गया} other {# रन स्टाइल किए गए}}", diff --git a/src/i18n/messages/hu.json b/src/i18n/messages/hu.json index d033965d4f..d5741d4b62 100644 --- a/src/i18n/messages/hu.json +++ b/src/i18n/messages/hu.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "mód: {mode}", "compressionAdaptiveOff": "Adaptív kontextuskeret: kikapcsolva (örökölt automatikus indítás)", "compressionAdaptiveTarget": "Adaptív ({mode}, szabályzat: {policy}) — cél ≈ {target, number} token ({contextLimit, number} tokenes ablakhoz)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Válaszformáló utasítások beillesztése a szolgáltató kimenetének átírása nélkül. Szabadon kombinálható.", "mcpAccessibilityDescription": "Hatókörbe foglalja az MCP-eszközök kimeneteit (külön tároló).", "compressionStylesTileSummary": "{tokens, number} token megtakarítva · {runs, plural, one {# stílusozott futtatás} other {# stílusozott futtatás}}", diff --git a/src/i18n/messages/id.json b/src/i18n/messages/id.json index 12ce90c987..95f4ca816a 100644 --- a/src/i18n/messages/id.json +++ b/src/i18n/messages/id.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "mode: {mode}", "compressionAdaptiveOff": "Anggaran konteks adaptif: nonaktif (pemicu otomatis warisan)", "compressionAdaptiveTarget": "Adaptif ({mode}, kebijakan: {policy}) — target ≈ {target, number} token (untuk jendela {contextLimit, number}-token)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Suntikkan instruksi pembentukan respons tanpa menulis ulang output penyedia. Kombinasikan secara bebas.", "mcpAccessibilityDescription": "Membatasi cakupan output alat MCP (penyimpanan terpisah).", "compressionStylesTileSummary": "{tokens, number} token disimpan · {runs, plural, one {# eksekusi digayakan} other {# eksekusi digayakan}}", diff --git a/src/i18n/messages/it.json b/src/i18n/messages/it.json index e5a5761760..4b77e9d30c 100644 --- a/src/i18n/messages/it.json +++ b/src/i18n/messages/it.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "modalità: {mode}", "compressionAdaptiveOff": "Budget del contesto adattivo: disattivato (attivazione automatica legacy)", "compressionAdaptiveTarget": "Adattivo ({mode}, criterio: {policy}) — target ≈ {target, number} token (per una finestra di {contextLimit, number} token)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Inserisci istruzioni per modellare la risposta senza riscrivere l'output del provider. Combina liberamente.", "mcpAccessibilityDescription": "Limita l'ambito degli output degli strumenti MCP (archivio separato).", "compressionStylesTileSummary": "{tokens, number} token risparmiati · {runs, plural, one {# esecuzione stilizzata} other {# esecuzioni stilizzate}}", diff --git a/src/i18n/messages/ja.json b/src/i18n/messages/ja.json index a4162c3b15..ce2d30773a 100644 --- a/src/i18n/messages/ja.json +++ b/src/i18n/messages/ja.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "モード: {mode}", "compressionAdaptiveOff": "アダプティブコンテキストバジェット: オフ(レガシー自動トリガー)", "compressionAdaptiveTarget": "アダプティブ({mode}、ポリシー: {policy})— ターゲット ≈ {target, number} トークン({contextLimit, number} トークンウィンドウ用)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "プロバイダーの出力を書き換えることなく、レスポンス整形指示を挿入します。自由に組み合わせ可能です。", "mcpAccessibilityDescription": "MCPツールの出力をスコープします(別ストア)。", "compressionStylesTileSummary": "{tokens, number} トークン節約 · {runs, plural, one {# 回の実行にスタイル適用} other {# 回の実行にスタイル適用}}", diff --git a/src/i18n/messages/ko.json b/src/i18n/messages/ko.json index 64275df522..c6a703ccac 100644 --- a/src/i18n/messages/ko.json +++ b/src/i18n/messages/ko.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "모드: {mode}", "compressionAdaptiveOff": "적응형 컨텍스트 예산: 꺼짐 (기존 자동 트리거)", "compressionAdaptiveTarget": "적응형 ({mode}, 정책: {policy}) — 대상 ≈ {target, number} 토큰 ({contextLimit, number} 토큰 창 기준)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "제공자 출력을 다시 작성하지 않고 응답 형성 지침을 주입합니다. 자유롭게 조합하세요.", "mcpAccessibilityDescription": "MCP 도구 출력의 범위를 제한합니다(별도 저장소).", "compressionStylesTileSummary": "{tokens, number} 토큰 절약됨 · {runs, plural, one {#개 실행 스타일 지정됨} other {#개 실행 스타일 지정됨}}", diff --git a/src/i18n/messages/mr.json b/src/i18n/messages/mr.json index d340f63917..aeeb20b219 100644 --- a/src/i18n/messages/mr.json +++ b/src/i18n/messages/mr.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "मोड: {mode}", "compressionAdaptiveOff": "अ‍ॅडॉप्टिव्ह संदर्भ बजेट: बंद (लेगसी ऑटो-ट्रिगर)", "compressionAdaptiveTarget": "अ‍ॅडॉप्टिव्ह ({mode}, पॉलिसी: {policy}) — लक्ष्य ≈ {target, number} टोकन्स ({contextLimit, number}-टोकन विंडोसाठी)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "प्रोव्हाइडर आउटपुट पुन्हा न लिहिता रिस्पॉन्स-शेपिंग सूचना इंजेक्ट करा. मुक्तपणे एकत्र करा.", "mcpAccessibilityDescription": "MCP टूल आउटपुटची व्याप्ती ठरवते (स्वतंत्र स्टोअर).", "compressionStylesTileSummary": "{tokens, number} टोकन्स वाचवले · {runs, plural, one {# रन स्टाईल केला} other {# रन्स स्टाईल केले}}", diff --git a/src/i18n/messages/ms.json b/src/i18n/messages/ms.json index f6b839fd7e..b506128049 100644 --- a/src/i18n/messages/ms.json +++ b/src/i18n/messages/ms.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "mod: {mode}", "compressionAdaptiveOff": "Belanjawan konteks adaptif: mati (pencetus automatik legasi)", "compressionAdaptiveTarget": "Adaptif ({mode}, dasar: {policy}) — sasaran ≈ {target, number} token (untuk tetingkap {contextLimit, number}-token)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Suntik arahan pembentukan respons tanpa menulis semula output penyedia. Gabungkan secara bebas.", "mcpAccessibilityDescription": "Menskupkan output alat MCP (storan berasingan).", "compressionStylesTileSummary": "{tokens, number} token dijimatkan · {runs, plural, one {# larian digayakan} other {# larian digayakan}}", diff --git a/src/i18n/messages/nl.json b/src/i18n/messages/nl.json index db912312d2..d92b373142 100644 --- a/src/i18n/messages/nl.json +++ b/src/i18n/messages/nl.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "modus: {mode}", "compressionAdaptiveOff": "Adaptief contextbudget: uit (legacy auto-trigger)", "compressionAdaptiveTarget": "Adaptief ({mode}, beleid: {policy}) — doel ≈ {target, number} tokens (voor een venster van {contextLimit, number} tokens)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Injecteer instructies voor responsvormgeving zonder de uitvoer van de provider te herschrijven. Vrij te combineren.", "mcpAccessibilityDescription": "Beperkt de scope van MCP-tooluitvoer (afzonderlijke opslag).", "compressionStylesTileSummary": "{tokens, number} tokens bespaard · {runs, plural, one {# run gestyled} other {# runs gestyled}}", diff --git a/src/i18n/messages/no.json b/src/i18n/messages/no.json index 4394a1e27b..1e415be8aa 100644 --- a/src/i18n/messages/no.json +++ b/src/i18n/messages/no.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "modus: {mode}", "compressionAdaptiveOff": "Adaptivt kontekstbudsjett: av (foreldet automatisk utløser)", "compressionAdaptiveTarget": "Adaptiv ({mode}, policy: {policy}) — mål ≈ {target, number} tokener (for et vindu på {contextLimit, number} tokener)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Sett inn instruksjoner for responsforming uten å omskrive leverandørutdata. Kombiner fritt.", "mcpAccessibilityDescription": "Avgrenser MCP-verktøyutdata (eget lager).", "compressionStylesTileSummary": "{tokens, number} tokener spart · {runs, plural, one {# kjøring stiltilpasset} other {# kjøringer stiltilpasset}}", diff --git a/src/i18n/messages/phi.json b/src/i18n/messages/phi.json index 89183a593d..da06b21c62 100644 --- a/src/i18n/messages/phi.json +++ b/src/i18n/messages/phi.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "mode: {mode}", "compressionAdaptiveOff": "Adaptive context budget: naka-off (legacy na auto-trigger)", "compressionAdaptiveTarget": "Adaptive ({mode}, patakaran: {policy}) — target ≈ {target, number} na token (para sa isang {contextLimit, number}-token na window)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Mag-inject ng mga tagubilin sa paghubog ng tugon nang hindi muling isinusulat ang output ng provider. Malayang pagsamahin.", "mcpAccessibilityDescription": "Nililimitahan ang mga output ng MCP tool (hiwalay na store).", "compressionStylesTileSummary": "{tokens, number} na token ang na-save · {runs, plural, one {# run ang na-style} other {# na run ang na-style}}", diff --git a/src/i18n/messages/pl.json b/src/i18n/messages/pl.json index 8f5c91da84..ee567fe371 100644 --- a/src/i18n/messages/pl.json +++ b/src/i18n/messages/pl.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "tryb: {mode}", "compressionAdaptiveOff": "Adaptacyjny budżet kontekstu: wył. (starszy automatyczny wyzwalacz)", "compressionAdaptiveTarget": "Adaptacyjny ({mode}, polityka: {policy}) — cel ≈ {target, number} tokenów (dla okna o rozmiarze {contextLimit, number} tokenów)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Wstrzykuj instrukcje kształtujące odpowiedź bez przepisywania danych wyjściowych dostawcy. Łącz dowolnie.", "mcpAccessibilityDescription": "Ogranicza zakres danych wyjściowych narzędzi MCP (osobny magazyn).", "compressionStylesTileSummary": "{tokens, number} tokenów zaoszczędzonych · {runs, plural, one {# przebieg ostylowany} few {# przebiegi ostylowane} many {# przebiegów ostylowanych} other {# przebiegów ostylowanych}}", diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index 0199b1b463..3fbe27e908 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -7816,6 +7816,14 @@ "compressionDerivedMode": "modo: {mode}", "compressionAdaptiveOff": "Orçamento de contexto adaptativo: desligado (auto-gatilho legado)", "compressionAdaptiveTarget": "Adaptativo ({mode}, política: {policy}) — alvo ≈ {target, number} tokens (para uma janela de {contextLimit, number} tokens)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Injeta instruções de modelagem de resposta sem reescrever a saída do provedor. Combine livremente.", "mcpAccessibilityDescription": "Delimita o escopo das saídas de ferramentas MCP (armazenamento separado).", "compressionStylesTileSummary": "{tokens, number} tokens economizados · {runs, plural, one {# execução estilizada} other {# execuções estilizadas}}", diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index 6cd1a1aa6c..4919461ca0 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "modo: {mode}", "compressionAdaptiveOff": "Orçamento de contexto adaptativo: desativado (acionamento automático legado)", "compressionAdaptiveTarget": "Adaptativo ({mode}, política: {policy}) — alvo ≈ {target, number} tokens (para uma janela de {contextLimit, number} tokens)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Injete instruções de modelação de resposta sem reescrever a saída do fornecedor. Combine livremente.", "mcpAccessibilityDescription": "Delimita as saídas das ferramentas MCP (armazenamento separado).", "compressionStylesTileSummary": "{tokens, number} tokens poupados · {runs, plural, one {# execução estilizada} other {# execuções estilizadas}}", diff --git a/src/i18n/messages/ro.json b/src/i18n/messages/ro.json index 4f445c4476..2d012143f9 100644 --- a/src/i18n/messages/ro.json +++ b/src/i18n/messages/ro.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "mod: {mode}", "compressionAdaptiveOff": "Buget de context adaptiv: dezactivat (declanșare automată moștenită)", "compressionAdaptiveTarget": "Adaptiv ({mode}, politică: {policy}) — țintă ≈ {target, number} tokenuri (pentru o fereastră de {contextLimit, number} tokenuri)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Injectează instrucțiuni de modelare a răspunsului fără a rescrie ieșirea furnizorului. Combină liber.", "mcpAccessibilityDescription": "Limitează domeniul de aplicare al ieșirilor instrumentelor MCP (stocare separată).", "compressionStylesTileSummary": "{tokens, number} tokenuri salvate · {runs, plural, one {# rulare stilizată} few {# rulări stilizate} other {# de rulări stilizate}}", diff --git a/src/i18n/messages/ru.json b/src/i18n/messages/ru.json index d740cc5faf..9f6bd4762e 100644 --- a/src/i18n/messages/ru.json +++ b/src/i18n/messages/ru.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "режим: {mode}", "compressionAdaptiveOff": "Адаптивный бюджет контекста: выкл. (устаревший автотриггер)", "compressionAdaptiveTarget": "Адаптивный ({mode}, политика: {policy}) — цель ≈ {target, number} токенов (для окна в {contextLimit, number} токенов)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Внедряйте инструкции по формированию ответов без перезаписи вывода провайдера. Комбинируйте свободно.", "mcpAccessibilityDescription": "Ограничивает область вывода инструментов MCP (отдельное хранилище).", "compressionStylesTileSummary": "{tokens, number} токенов сэкономлено · {runs, plural, one {# запуск стилизован} few {# запуска стилизовано} many {# запусков стилизовано} other {# запуска стилизовано}}", diff --git a/src/i18n/messages/sk.json b/src/i18n/messages/sk.json index 5dcf60ac13..b5fc50f625 100644 --- a/src/i18n/messages/sk.json +++ b/src/i18n/messages/sk.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "režim: {mode}", "compressionAdaptiveOff": "Adaptívny rozpočet kontextu: vypnuté (starší automatický spúšťač)", "compressionAdaptiveTarget": "Adaptívny ({mode}, politika: {policy}) — cieľ ≈ {target, number} tokenov (pre okno s {contextLimit, number} tokenmi)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Vložiť inštrukcie na formovanie odpovede bez prepisovania výstupu poskytovateľa. Voľne kombinujte.", "mcpAccessibilityDescription": "Obmedzuje rozsah výstupov nástrojov MCP (samostatné úložisko).", "compressionStylesTileSummary": "{tokens, number} tokenov ušetrených · {runs, plural, one {# štylizované spustenie} few {# štylizované spustenia} other {# štylizovaných spustení}}", diff --git a/src/i18n/messages/sv.json b/src/i18n/messages/sv.json index bc502df18d..eb8a2bd8f3 100644 --- a/src/i18n/messages/sv.json +++ b/src/i18n/messages/sv.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "läge: {mode}", "compressionAdaptiveOff": "Adaptiv kontextbudget: av (äldre auto-trigger)", "compressionAdaptiveTarget": "Adaptiv ({mode}, policy: {policy}) — mål ≈ {target, number} tokens (för ett fönster på {contextLimit, number} tokens)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Infoga instruktioner för svarsformning utan att skriva om leverantörens utdata. Kombinera fritt.", "mcpAccessibilityDescription": "Avgränsar MCP-verktygsutdata (separat lagring).", "compressionStylesTileSummary": "{tokens, number} tokens sparade · {runs, plural, one {# körning stylad} other {# körningar stylade}}", diff --git a/src/i18n/messages/sw.json b/src/i18n/messages/sw.json index 7301fb6774..31c55e5173 100644 --- a/src/i18n/messages/sw.json +++ b/src/i18n/messages/sw.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "hali: {mode}", "compressionAdaptiveOff": "Bajeti ya muktadha inayobadilika: imezimwa (kichochezi cha zamani cha kiotomatiki)", "compressionAdaptiveTarget": "Inayobadilika ({mode}, sera: {policy}) — lengo ≈ tokeni {target, number} (kwa dirisha la tokeni {contextLimit, number})", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Ingiza maagizo ya kuunda majibu bila kuandika upya matokeo ya mtoa huduma. Changanya kwa uhuru.", "mcpAccessibilityDescription": "Inaweka mipaka ya matokeo ya zana ya MCP (hifadhi tofauti).", "compressionStylesTileSummary": "{tokens, number} tokeni zimehifadhiwa · {runs, plural, one {# umewekewa mtindo} other {# imewekewa mtindo}}", diff --git a/src/i18n/messages/ta.json b/src/i18n/messages/ta.json index 346590dc7f..49679193d9 100644 --- a/src/i18n/messages/ta.json +++ b/src/i18n/messages/ta.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "முறை: {mode}", "compressionAdaptiveOff": "தகவமைப்பு சூழல் பட்ஜெட்: ஆஃப் (பழைய தானியங்கு-தூண்டுதல்)", "compressionAdaptiveTarget": "தகவமைப்பு ({mode}, கொள்கை: {policy}) — இலக்கு ≈ {target, number} டோக்கன்கள் ({contextLimit, number}-டோக்கன் சாளரத்திற்கு)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "வழங்குநரின் வெளியீட்டை மீண்டும் எழுதாமல் பதில்-வடிவமைப்பு வழிமுறைகளை உட்செலுத்தவும். தாராளமாக இணைக்கவும்.", "mcpAccessibilityDescription": "MCP கருவி வெளியீடுகளை வரம்பிற்குள் வைக்கிறது (தனிச் சேமிப்பகம்).", "compressionStylesTileSummary": "{tokens, number} டோக்கன்கள் சேமிக்கப்பட்டன · {runs, plural, one {# இயக்கம் வடிவமைக்கப்பட்டது} other {# இயக்கங்கள் வடிவமைக்கப்பட்டன}}", diff --git a/src/i18n/messages/te.json b/src/i18n/messages/te.json index 5cbdf5c885..1a5223b1c5 100644 --- a/src/i18n/messages/te.json +++ b/src/i18n/messages/te.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "మోడ్: {mode}", "compressionAdaptiveOff": "అడాప్టివ్ కాంటెక్స్ట్ బడ్జెట్: ఆఫ్ (లెగసీ ఆటో-ట్రిగ్గర్)", "compressionAdaptiveTarget": "అడాప్టివ్ ({mode}, పాలసీ: {policy}) — టార్గెట్ ≈ {target, number} టోకెన్‌లు ({contextLimit, number}-టోకెన్ విండో కోసం)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "ప్రొవైడర్ అవుట్‌పుట్‌ను తిరిగి రాయకుండా రెస్పాన్స్-షేపింగ్ సూచనలను ఇంజెక్ట్ చేయండి. స్వేచ్ఛగా కలపండి.", "mcpAccessibilityDescription": "MCP టూల్ అవుట్‌పుట్‌లను స్కోప్ చేస్తుంది (ప్రత్యేక స్టోర్).", "compressionStylesTileSummary": "{tokens, number} టోకెన్‌లు ఆదా చేయబడ్డాయి · {runs, plural, one {# రన్ స్టైల్ చేయబడింది} other {# రన్‌లు స్టైల్ చేయబడ్డాయి}}", diff --git a/src/i18n/messages/th.json b/src/i18n/messages/th.json index 68f90b88a6..a7093e6758 100644 --- a/src/i18n/messages/th.json +++ b/src/i18n/messages/th.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "โหมด: {mode}", "compressionAdaptiveOff": "งบประมาณบริบทแบบปรับตัว: ปิด (การทริกเกอร์อัตโนมัติแบบเก่า)", "compressionAdaptiveTarget": "แบบปรับตัว ({mode}, นโยบาย: {policy}) — เป้าหมาย ≈ {target, number} โทเค็น (สำหรับหน้าต่าง {contextLimit, number} โทเค็น)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "แทรกคำสั่งกำหนดรูปแบบการตอบกลับโดยไม่ต้องเขียนผลลัพธ์ของผู้ให้บริการใหม่ สามารถผสมผสานได้อย่างอิสระ", "mcpAccessibilityDescription": "กำหนดขอบเขตผลลัพธ์เครื่องมือ MCP (แยกพื้นที่จัดเก็บ)", "compressionStylesTileSummary": "{tokens, number} โทเค็นที่ประหยัดได้ · {runs, plural, one {จัดรูปแบบแล้ว # ครั้ง} other {จัดรูปแบบแล้ว # ครั้ง}}", diff --git a/src/i18n/messages/tr.json b/src/i18n/messages/tr.json index 6ed5f68933..c7ea532a2f 100644 --- a/src/i18n/messages/tr.json +++ b/src/i18n/messages/tr.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "mod: {mode}", "compressionAdaptiveOff": "Uyarlanabilir bağlam bütçesi: kapalı (eski otomatik tetikleyici)", "compressionAdaptiveTarget": "Uyarlanabilir ({mode}, politika: {policy}) — hedef ≈ {target, number} token ({contextLimit, number} tokenlık bir pencere için)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Sağlayıcı çıktısını yeniden yazmadan yanıt şekillendirme talimatları ekleyin. Serbestçe birleştirin.", "mcpAccessibilityDescription": "MCP araç çıktılarını kapsama alır (ayrı depo).", "compressionStylesTileSummary": "{tokens, number} token tasarruf edildi · {runs, plural, one {# çalıştırma stillendirildi} other {# çalıştırma stillendirildi}}", diff --git a/src/i18n/messages/uk-UA.json b/src/i18n/messages/uk-UA.json index 5601b38668..f6a1e7f0b1 100644 --- a/src/i18n/messages/uk-UA.json +++ b/src/i18n/messages/uk-UA.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "режим: {mode}", "compressionAdaptiveOff": "Адаптивний бюджет контексту: вимкнено (застарілий автотригер)", "compressionAdaptiveTarget": "Адаптивний ({mode}, політика: {policy}) — ціль ≈ {target, number} токенів (для вікна в {contextLimit, number} токенів)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "Впроваджуйте інструкції формування відповіді без перезапису виводу провайдера. Вільно комбінуйте.", "mcpAccessibilityDescription": "Обмежує область виводу інструментів MCP (окреме сховище).", "compressionStylesTileSummary": "{tokens, number} токенів збережено · {runs, plural, one {# стилізований запуск} few {# стилізовані запуски} many {# стилізованих запусків} other {# стилізованих запусків}}", diff --git a/src/i18n/messages/ur.json b/src/i18n/messages/ur.json index e22f742090..bf80a21de8 100644 --- a/src/i18n/messages/ur.json +++ b/src/i18n/messages/ur.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "موڈ: {mode}", "compressionAdaptiveOff": "اڈیپٹیو کانٹیکسٹ بجٹ: بند (لیگیسی آٹو ٹریگر)", "compressionAdaptiveTarget": "اڈیپٹیو ({mode}, پالیسی: {policy}) — ہدف ≈ {target, number} ٹوکنز ({contextLimit, number}-ٹوکن ونڈو کے لیے)", + "compressionAdaptiveMode": "__MISSING__:Adaptive context budget", + "compressionAdaptiveModeOff": "__MISSING__:Off (legacy auto-trigger)", + "compressionAdaptiveModeFloor": "__MISSING__:Floor (always guarantee fit)", + "compressionAdaptiveModeReplace": "__MISSING__:Replace auto-trigger", + "compressionAdaptivePolicy": "__MISSING__:Budget policy", + "compressionAdaptivePolicyReserve": "__MISSING__:Reserve output", + "compressionAdaptivePolicyPercentage": "__MISSING__:Percentage of window", + "compressionAdaptivePolicyAbsolute": "__MISSING__:Absolute token budget", "compressionOutputStylesDescription": "پرووائیڈر آؤٹ پٹ کو دوبارہ لکھے بغیر رسپانس شیپنگ ہدایات شامل کریں۔ آزادانہ طور پر یکجا کریں۔", "mcpAccessibilityDescription": "MCP ٹول آؤٹ پٹس کو اسکوپ کرتا ہے (علیحدہ اسٹور)۔", "compressionStylesTileSummary": "{tokens, number} ٹوکنز محفوظ کیے گئے · {runs, plural, one {# رن اسٹائل کیا گیا} other {# رنز اسٹائل کیے گئے}}", diff --git a/src/i18n/messages/vi.json b/src/i18n/messages/vi.json index d0a948d93b..f2e4a120bc 100644 --- a/src/i18n/messages/vi.json +++ b/src/i18n/messages/vi.json @@ -7816,6 +7816,14 @@ "compressionDerivedMode": "chế độ: {mode}", "compressionAdaptiveOff": "Ngân sách ngữ cảnh thích ứng: đã tắt (dùng ngưỡng tự động kích hoạt cũ)", "compressionAdaptiveTarget": "Thích ứng ({mode}, chính sách: {policy}) — mục tiêu ≈ {target, number} token (với cửa sổ {contextLimit, number} token)", + "compressionAdaptiveMode": "Ngân sách ngữ cảnh thích ứng", + "compressionAdaptiveModeOff": "Tắt (ngưỡng tự động kích hoạt cũ)", + "compressionAdaptiveModeFloor": "Sàn (luôn đảm bảo vừa cửa sổ)", + "compressionAdaptiveModeReplace": "Thay thế tự động kích hoạt", + "compressionAdaptivePolicy": "Chính sách ngân sách", + "compressionAdaptivePolicyReserve": "Dành chỗ cho đầu ra", + "compressionAdaptivePolicyPercentage": "Phần trăm cửa sổ", + "compressionAdaptivePolicyAbsolute": "Ngân sách token tuyệt đối", "compressionOutputStylesDescription": "Chèn hướng dẫn định hình phản hồi mà không viết lại đầu ra của nhà cung cấp. Có thể kết hợp tự do.", "mcpAccessibilityDescription": "Giới hạn phạm vi đầu ra của công cụ MCP (được lưu riêng).", "compressionStylesTileSummary": "Đã tiết kiệm {tokens, number} token · {runs, number} lượt áp dụng kiểu", diff --git a/src/i18n/messages/zh-CN.json b/src/i18n/messages/zh-CN.json index 8501b6d0b4..2301729df6 100644 --- a/src/i18n/messages/zh-CN.json +++ b/src/i18n/messages/zh-CN.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "模式:{mode}", "compressionAdaptiveOff": "自适应上下文预算:关闭(旧版自动触发)", "compressionAdaptiveTarget": "自适应 ({mode}, 策略: {policy}) — 目标 ≈ {target, number} 个 token (针对 {contextLimit, number} 个 token 的窗口)", + "compressionAdaptiveMode": "自适应上下文预算", + "compressionAdaptiveModeOff": "关闭(旧版自动触发)", + "compressionAdaptiveModeFloor": "下限(始终保证适配)", + "compressionAdaptiveModeReplace": "替换自动触发", + "compressionAdaptivePolicy": "预算策略", + "compressionAdaptivePolicyReserve": "预留输出", + "compressionAdaptivePolicyPercentage": "窗口百分比", + "compressionAdaptivePolicyAbsolute": "绝对 token 预算", "compressionOutputStylesDescription": "注入响应塑造指令,而无需重写服务商输出。自由组合。", "mcpAccessibilityDescription": "限制 MCP 工具输出的范围 (独立存储)。", "compressionStylesTileSummary": "{tokens, number} 个 token 已节省 · {runs, plural, one {# 次运行已应用样式} other {# 次运行已应用样式}}", diff --git a/src/i18n/messages/zh-TW.json b/src/i18n/messages/zh-TW.json index 66689e9057..f0dea4fe5c 100644 --- a/src/i18n/messages/zh-TW.json +++ b/src/i18n/messages/zh-TW.json @@ -7812,6 +7812,14 @@ "compressionDerivedMode": "模式:{mode}", "compressionAdaptiveOff": "自適應上下文預算:關閉(傳統自動觸發)", "compressionAdaptiveTarget": "自適應({mode},策略:{policy})— 目標約 {target, number} tokens(針對 {contextLimit, number} token 的視窗)", + "compressionAdaptiveMode": "自適應上下文預算", + "compressionAdaptiveModeOff": "關閉(傳統自動觸發)", + "compressionAdaptiveModeFloor": "下限(始終保證適配)", + "compressionAdaptiveModeReplace": "替換自動觸發", + "compressionAdaptivePolicy": "預算策略", + "compressionAdaptivePolicyReserve": "預留輸出", + "compressionAdaptivePolicyPercentage": "視窗百分比", + "compressionAdaptivePolicyAbsolute": "絕對 token 預算", "compressionOutputStylesDescription": "注入回應塑形指令,無需改寫提供者輸出。可自由組合。", "mcpAccessibilityDescription": "限定 MCP 工具輸出範圍(獨立儲存區)。", "compressionStylesTileSummary": "{tokens, number} tokens saved · {runs, plural, one {# run styled} other {# runs styled}}", diff --git a/tests/unit/ui/compressionAdaptiveBudgetDial.test.tsx b/tests/unit/ui/compressionAdaptiveBudgetDial.test.tsx new file mode 100644 index 0000000000..5d4cd36afb --- /dev/null +++ b/tests/unit/ui/compressionAdaptiveBudgetDial.test.tsx @@ -0,0 +1,660 @@ +// @vitest-environment jsdom +import React, { act } from "react"; +import { createRoot } from "react-dom/client"; +import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; +import { DEFAULT_CONTEXT_BUDGET } from "../../../open-sse/services/compression/adaptiveCompression/types.ts"; + +// i18n does not resolve to a real locale in vitest/jsdom, so mock next-intl to echo +// the key. This test asserts ONLY on i18n-independent hooks (data-testid + values) +// and the captured PUT body. +vi.mock("next-intl", () => ({ + useTranslations: () => (key: string) => key, + useLocale: () => "en", +})); + +const containers: HTMLElement[] = []; +const roots: Array<{ unmount: () => void }> = []; + +function mount(ui: React.ReactElement): HTMLElement { + const container = document.createElement("div"); + document.body.appendChild(container); + containers.push(container); + const root = createRoot(container); + roots.push(root); + act(() => root.render(ui)); + return container; +} + +beforeEach(() => { + ( + globalThis as typeof globalThis & { IS_REACT_ACT_ENVIRONMENT?: boolean } + ).IS_REACT_ACT_ENVIRONMENT = true; +}); + +afterEach(async () => { + vi.restoreAllMocks(); + await act(async () => { + while (roots.length > 0) roots.pop()?.unmount(); + }); + for (let i = 0; i < 10; i++) await Promise.resolve(); + while (containers.length > 0) containers.pop()?.remove(); + document.body.innerHTML = ""; +}); + +async function flush() { + await act(async () => { + for (let i = 0; i < 10; i++) await Promise.resolve(); + }); +} + +interface CapturedPut { + url: string; + body: Record; +} + +function setupFetchMock( + overrides?: Record, + opts?: { putStatus?: number; putStatusFn?: (n: number) => number } +): { puts: CapturedPut[] } { + const puts: CapturedPut[] = []; + const json = (body: unknown, status = 200) => + new Response(JSON.stringify(body), { status, headers: { "Content-Type": "application/json" } }); + const initial = { + enabled: true, + autoTriggerTokens: 0, + preserveSystemPrompt: true, + engines: {}, + activeComboId: null, + outputStyles: [], + cavemanOutputMode: { enabled: false, intensity: "full", autoClarity: true }, + ultraEngine: "heuristic", + ultraSlmPrewarm: false, + liveZone: { enabled: false }, + contextBudget: { ...DEFAULT_CONTEXT_BUDGET }, + ...overrides, + }; + vi.spyOn(globalThis, "fetch").mockImplementation( + async (input: RequestInfo | URL, init?: RequestInit) => { + const url = input.toString(); + const method = (init?.method ?? "GET").toUpperCase(); + if (url.includes("/api/settings/compression/mcp-accessibility")) + return json({ enabled: true }); + if (url.includes("/api/settings/compression")) { + if (method === "PUT") { + const body = JSON.parse(String(init?.body ?? "{}")) as Record; + puts.push({ url, body }); + const n = puts.length; + const status = opts?.putStatusFn ? opts.putStatusFn(n) : (opts?.putStatus ?? 200); + const merged = + body.contextBudget && typeof body.contextBudget === "object" + ? { + ...initial, + ...body, + contextBudget: { + ...(initial.contextBudget as Record), + ...(body.contextBudget as Record), + }, + } + : { ...initial, ...body }; + return json(merged, status); + } + return json(initial); + } + return json({}, 404); + } + ); + return { puts }; +} + +describe("CompressionPanel adaptive context-budget dial", () => { + it("renders the mode select defaulting to off (legacy auto-trigger)", async () => { + setupFetchMock(); + const { default: CompressionPanel } = + await import("../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel"); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const select = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement | null; + expect(select, "mode select must render").toBeTruthy(); + expect(select?.value).toBe("off"); + expect(container.querySelector(`[data-testid="context-budget-policy-select"]`)).toBeFalsy(); + expect( + container.querySelector(`[data-testid="adaptive-target-preview"]`), + "preview label stays inside the dial" + ).toBeTruthy(); + }); + + it("hydrates mode off when GET omits contextBudget", async () => { + setupFetchMock({ contextBudget: undefined }); + const { default: CompressionPanel } = + await import("../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel"); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const select = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement | null; + expect(select?.value).toBe("off"); + expect(container.querySelector(`[data-testid="context-budget-policy-select"]`)).toBeFalsy(); + }); + + it("hides the policy select when GET hydrates mode as null (same as off)", async () => { + setupFetchMock({ + contextBudget: { ...DEFAULT_CONTEXT_BUDGET, mode: null as unknown as "off" }, + }); + const { default: CompressionPanel } = + await import("../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel"); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const select = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement | null; + expect(select?.value).toBe("off"); + expect(container.querySelector(`[data-testid="context-budget-policy-select"]`)).toBeFalsy(); + }); + + it("selecting floor PUTs the full contextBudget object with mode:'floor'", async () => { + const { puts } = setupFetchMock(); + const { default: CompressionPanel } = + await import("../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel"); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const select = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement; + expect(select).toBeTruthy(); + await act(async () => { + select.value = "floor"; + select.dispatchEvent(new Event("change", { bubbles: true })); + }); + await flush(); + + const put = puts.find((p) => "contextBudget" in p.body); + expect(put, "a PUT carrying contextBudget").toBeTruthy(); + const budget = put!.body.contextBudget as Record; + expect(budget.mode).toBe("floor"); + expect(budget.policy).toBe(DEFAULT_CONTEXT_BUDGET.policy); + expect(budget.outputReserve).toBe(DEFAULT_CONTEXT_BUDGET.outputReserve); + expect(budget.safetyMargin).toBe(DEFAULT_CONTEXT_BUDGET.safetyMargin); + expect(budget.pct).toBe(DEFAULT_CONTEXT_BUDGET.pct); + expect(budget.absoluteBudget).toBe(DEFAULT_CONTEXT_BUDGET.absoluteBudget); + expect(budget.ladderOverride).toBe(DEFAULT_CONTEXT_BUDGET.ladderOverride); + + expect( + container.querySelector(`[data-testid="context-budget-policy-select"]`), + "policy select appears once mode is not off" + ).toBeTruthy(); + }); + + it("selecting replace-autotrigger PUTs mode and reveals the policy select", async () => { + const { puts } = setupFetchMock(); + const { default: CompressionPanel } = + await import("../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel"); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const select = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement; + await act(async () => { + select.value = "replace-autotrigger"; + select.dispatchEvent(new Event("change", { bubbles: true })); + }); + await flush(); + + const put = puts.find((p) => "contextBudget" in p.body); + expect(put, "a PUT carrying contextBudget").toBeTruthy(); + const budget = put!.body.contextBudget as Record; + expect(budget.mode).toBe("replace-autotrigger"); + expect(container.querySelector(`[data-testid="context-budget-policy-select"]`)).toBeTruthy(); + }); + + it("rolls the mode select back to off when the PUT fails", async () => { + setupFetchMock(undefined, { putStatus: 500 }); + const { default: CompressionPanel } = + await import("../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel"); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const select = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement; + await act(async () => { + select.value = "floor"; + select.dispatchEvent(new Event("change", { bubbles: true })); + }); + await flush(); + + const after = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement; + expect(after.value).toBe("off"); + expect(container.querySelector(`[data-testid="context-budget-policy-select"]`)).toBeFalsy(); + }); + + it("rolls policy back to the hydrated value when the PUT fails", async () => { + const { puts } = setupFetchMock( + { + contextBudget: { + ...DEFAULT_CONTEXT_BUDGET, + mode: "floor", + policy: "reserve-output", + }, + }, + { putStatus: 500 } + ); + const { default: CompressionPanel } = + await import("../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel"); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const policy = container.querySelector( + `[data-testid="context-budget-policy-select"]` + ) as HTMLSelectElement; + expect(policy.value).toBe("reserve-output"); + await act(async () => { + policy.value = "percentage"; + policy.dispatchEvent(new Event("change", { bubbles: true })); + }); + await flush(); + + const after = container.querySelector( + `[data-testid="context-budget-policy-select"]` + ) as HTMLSelectElement; + expect(after.value).toBe("reserve-output"); + expect(puts).toHaveLength(1); + const budget = puts[0].body.contextBudget as Record; + expect(budget.mode).toBe("floor"); + expect(budget.policy).toBe("percentage"); + expect(budget.outputReserve).toBe(DEFAULT_CONTEXT_BUDGET.outputReserve); + expect(budget.absoluteBudget).toBe(DEFAULT_CONTEXT_BUDGET.absoluteBudget); + }); + + it("does not let an older failed PUT roll back a newer successful save", async () => { + const puts: CapturedPut[] = []; + const pending: Array<{ + resolve: (r: Response) => void; + body: Record; + }> = []; + const json = (body: unknown, status = 200) => + new Response(JSON.stringify(body), { + status, + headers: { "Content-Type": "application/json" }, + }); + const initial = { + enabled: true, + autoTriggerTokens: 0, + preserveSystemPrompt: true, + engines: {}, + activeComboId: null, + outputStyles: [], + cavemanOutputMode: { enabled: false, intensity: "full", autoClarity: true }, + ultraEngine: "heuristic", + ultraSlmPrewarm: false, + liveZone: { enabled: false }, + contextBudget: { ...DEFAULT_CONTEXT_BUDGET }, + }; + vi.spyOn(globalThis, "fetch").mockImplementation( + async (input: RequestInfo | URL, init?: RequestInit) => { + const url = input.toString(); + const method = (init?.method ?? "GET").toUpperCase(); + if (url.includes("/api/settings/compression/mcp-accessibility")) + return json({ enabled: true }); + if (url.includes("/api/settings/compression")) { + if (method === "PUT") { + const body = JSON.parse(String(init?.body ?? "{}")) as Record; + puts.push({ url, body }); + return new Promise((resolve) => { + pending.push({ resolve, body }); + }); + } + return json(initial); + } + return json({}, 404); + } + ); + + const { default: CompressionPanel } = + await import("../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel"); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const mode = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement; + const policy = container.querySelector( + `[data-testid="context-budget-policy-select"]` + ) as HTMLSelectElement | null; + expect(policy).toBeFalsy(); + + // Two PUTs from the same render: older one will 500 after the newer one 200s. + await act(async () => { + mode.value = "floor"; + mode.dispatchEvent(new Event("change", { bubbles: true })); + }); + expect(pending).toHaveLength(1); + + const policyAfter = container.querySelector( + `[data-testid="context-budget-policy-select"]` + ) as HTMLSelectElement; + expect(policyAfter, "policy select after optimistic floor").toBeTruthy(); + await act(async () => { + policyAfter.value = "percentage"; + policyAfter.dispatchEvent(new Event("change", { bubbles: true })); + }); + expect(pending).toHaveLength(2); + + await act(async () => { + pending[1].resolve(json({ ...initial, ...pending[1].body }, 200)); + }); + await flush(); + await act(async () => { + pending[0].resolve(json({ ...initial, ...pending[0].body }, 500)); + }); + await flush(); + + const afterMode = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement; + const afterPolicy = container.querySelector( + `[data-testid="context-budget-policy-select"]` + ) as HTMLSelectElement; + expect(afterMode.value).toBe("floor"); + expect(afterPolicy.value).toBe("percentage"); + }); + + it("keeps an older successful PUT as lastConfirmed when a newer overlapping PUT fails", async () => { + const puts: CapturedPut[] = []; + const pending: Array<{ + resolve: (r: Response) => void; + body: Record; + }> = []; + const json = (body: unknown, status = 200) => + new Response(JSON.stringify(body), { + status, + headers: { "Content-Type": "application/json" }, + }); + const initial = { + enabled: true, + autoTriggerTokens: 0, + preserveSystemPrompt: true, + engines: {}, + activeComboId: null, + outputStyles: [], + cavemanOutputMode: { enabled: false, intensity: "full", autoClarity: true }, + ultraEngine: "heuristic", + ultraSlmPrewarm: false, + liveZone: { enabled: false }, + contextBudget: { ...DEFAULT_CONTEXT_BUDGET }, + }; + vi.spyOn(globalThis, "fetch").mockImplementation( + async (input: RequestInfo | URL, init?: RequestInit) => { + const url = input.toString(); + const method = (init?.method ?? "GET").toUpperCase(); + if (url.includes("/api/settings/compression/mcp-accessibility")) + return json({ enabled: true }); + if (url.includes("/api/settings/compression")) { + if (method === "PUT") { + const body = JSON.parse(String(init?.body ?? "{}")) as Record; + puts.push({ url, body }); + return new Promise((resolve) => { + pending.push({ resolve, body }); + }); + } + return json(initial); + } + return json({}, 404); + } + ); + + const { default: CompressionPanel } = + await import("../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel"); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const mode = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement; + await act(async () => { + mode.value = "floor"; + mode.dispatchEvent(new Event("change", { bubbles: true })); + }); + expect(pending).toHaveLength(1); + + const policyAfter = container.querySelector( + `[data-testid="context-budget-policy-select"]` + ) as HTMLSelectElement; + await act(async () => { + policyAfter.value = "percentage"; + policyAfter.dispatchEvent(new Event("change", { bubbles: true })); + }); + expect(pending).toHaveLength(2); + + // Older save A acks first (stale gen). Newer save B then 500s. + await act(async () => { + pending[0].resolve(json({ ...initial, ...pending[0].body }, 200)); + }); + await flush(); + await act(async () => { + pending[1].resolve(json({ ...initial, ...pending[1].body }, 500)); + }); + await flush(); + + const afterMode = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement; + const afterPolicy = container.querySelector( + `[data-testid="context-budget-policy-select"]` + ) as HTMLSelectElement; + expect(afterMode.value).toBe("floor"); + expect( + afterPolicy, + "policy select stays — lastConfirmed is A's floor, not GET off" + ).toBeTruthy(); + expect(afterPolicy.value).toBe("reserve-output"); + }); + + it("rolls both overlapping failed PUTs back to the last GET snapshot, not the first optimistic state", async () => { + const puts: CapturedPut[] = []; + const pending: Array<{ + resolve: (r: Response) => void; + body: Record; + }> = []; + const json = (body: unknown, status = 200) => + new Response(JSON.stringify(body), { + status, + headers: { "Content-Type": "application/json" }, + }); + const initial = { + enabled: true, + autoTriggerTokens: 0, + preserveSystemPrompt: true, + engines: {}, + activeComboId: null, + outputStyles: [], + cavemanOutputMode: { enabled: false, intensity: "full", autoClarity: true }, + ultraEngine: "heuristic", + ultraSlmPrewarm: false, + liveZone: { enabled: false }, + contextBudget: { ...DEFAULT_CONTEXT_BUDGET }, + }; + vi.spyOn(globalThis, "fetch").mockImplementation( + async (input: RequestInfo | URL, init?: RequestInit) => { + const url = input.toString(); + const method = (init?.method ?? "GET").toUpperCase(); + if (url.includes("/api/settings/compression/mcp-accessibility")) + return json({ enabled: true }); + if (url.includes("/api/settings/compression")) { + if (method === "PUT") { + const body = JSON.parse(String(init?.body ?? "{}")) as Record; + puts.push({ url, body }); + return new Promise((resolve) => { + pending.push({ resolve, body }); + }); + } + return json(initial); + } + return json({}, 404); + } + ); + + const { default: CompressionPanel } = + await import("../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel"); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const mode = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement; + await act(async () => { + mode.value = "floor"; + mode.dispatchEvent(new Event("change", { bubbles: true })); + }); + expect(pending).toHaveLength(1); + + const policyAfter = container.querySelector( + `[data-testid="context-budget-policy-select"]` + ) as HTMLSelectElement; + await act(async () => { + policyAfter.value = "percentage"; + policyAfter.dispatchEvent(new Event("change", { bubbles: true })); + }); + expect(pending).toHaveLength(2); + + await act(async () => { + pending[1].resolve(json({ ...initial, ...pending[1].body }, 500)); + }); + await flush(); + await act(async () => { + pending[0].resolve(json({ ...initial, ...pending[0].body }, 500)); + }); + await flush(); + + const afterMode = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement; + expect(afterMode.value).toBe("off"); + expect(container.querySelector(`[data-testid="context-budget-policy-select"]`)).toBeFalsy(); + }); + + it("does not let a stale saved-timeout clear a newer error status", async () => { + vi.useFakeTimers(); + try { + const { puts } = setupFetchMock(undefined, { + putStatusFn: (n) => (n === 1 ? 200 : 500), + }); + const { default: CompressionPanel } = + await import("../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel"); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const mode = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement; + await act(async () => { + mode.value = "floor"; + mode.dispatchEvent(new Event("change", { bubbles: true })); + }); + await flush(); + expect(puts).toHaveLength(1); + expect(container.textContent).toContain("saved"); + + const policy = container.querySelector( + `[data-testid="context-budget-policy-select"]` + ) as HTMLSelectElement; + await act(async () => { + policy.value = "percentage"; + policy.dispatchEvent(new Event("change", { bubbles: true })); + }); + await flush(); + expect(puts).toHaveLength(2); + expect(container.textContent).toContain("saveFailed"); + + await act(async () => { + vi.advanceTimersByTime(2000); + }); + await flush(); + expect(container.textContent).toContain("saveFailed"); + } finally { + vi.useRealTimers(); + } + }); + + it("hydrates GET contextBudget and changing policy PUTs the merged object", async () => { + const { puts } = setupFetchMock({ + contextBudget: { + ...DEFAULT_CONTEXT_BUDGET, + mode: "floor", + policy: "reserve-output", + }, + }); + const { default: CompressionPanel } = + await import("../../../src/app/(dashboard)/dashboard/context/settings/CompressionPanel"); + let container!: HTMLElement; + await act(async () => { + container = mount(); + }); + await flush(); + + const mode = container.querySelector( + `[data-testid="context-budget-mode-select"]` + ) as HTMLSelectElement; + const policy = container.querySelector( + `[data-testid="context-budget-policy-select"]` + ) as HTMLSelectElement; + expect(mode.value).toBe("floor"); + expect(policy, "policy select must hydrate when mode is floor").toBeTruthy(); + expect(policy.value).toBe("reserve-output"); + + await act(async () => { + policy.value = "percentage"; + policy.dispatchEvent(new Event("change", { bubbles: true })); + }); + await flush(); + + const put = puts.find((p) => "contextBudget" in p.body); + expect(put, "a PUT carrying contextBudget").toBeTruthy(); + const budget = put!.body.contextBudget as Record; + expect(budget.mode).toBe("floor"); + expect(budget.policy).toBe("percentage"); + expect(budget.pct).toBe(DEFAULT_CONTEXT_BUDGET.pct); + expect(budget.ladderOverride).toBe(DEFAULT_CONTEXT_BUDGET.ladderOverride); + }); +}); From 40c80756e4ad4664e8c633432d23df12b3958d74 Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Thu, 3 Sep 2026 11:40:21 -0400 Subject: [PATCH 038/143] fix(quota): keep Antigravity Gemini usable when Claude weekly is empty (#12566) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check:provider-consistency` OK (273 entradas REGISTRY, **356** providers canônicos), `check-docs-counts-sync` exit 0 e **300/300** nos testes que a leva toca. O crescimento de arquivo que os PRs empilham uns sobre os outros foi rebaselinado num único registro datado (`_rebaseline_2026_09_03_houminxi_batch`), com a decomposição por arquivo: `providers/page.tsx` +18 (import CSV do #12504 + busca do #12495 no mesmo painel), `accountFallback.ts` +6 (o #12566 sobre o rebaseline que o #12590 já registrou — os dois tocam `checkFallbackError`) e `chatCore.ts` +3 (invalidez de cache de quota no 429 do #12325). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. --- open-sse/executors/antigravity.ts | 19 +- open-sse/services/accountFallback.ts | 13 +- .../services/antigravityFamilyCooldown.ts | 158 +++++++++++ open-sse/services/antigravityQuotaFamily.ts | 74 +++++ open-sse/services/combo.ts | 12 +- open-sse/services/combo/comboPredicates.ts | 23 +- open-sse/services/combo/nativeCodexTurnPin.ts | 3 +- .../services/combo/quotaExhaustionCutoff.ts | 6 +- open-sse/services/quotaPreflight.ts | 124 +++++---- src/domain/quotaCache.ts | 34 +-- src/sse/services/auth.ts | 50 +--- src/sse/services/quotaPreflightUnavailable.ts | 61 +++++ stryker.conf.json | 1 + ...agy-family-not-connection-cooldown.test.ts | 258 ++++++++++++++++++ ...ard-session-lease-bypass-inventory.test.ts | 2 + 15 files changed, 676 insertions(+), 162 deletions(-) create mode 100644 open-sse/services/antigravityFamilyCooldown.ts create mode 100644 src/sse/services/quotaPreflightUnavailable.ts create mode 100644 tests/unit/agy-family-not-connection-cooldown.test.ts diff --git a/open-sse/executors/antigravity.ts b/open-sse/executors/antigravity.ts index c16d626812..9582af9fe6 100644 --- a/open-sse/executors/antigravity.ts +++ b/open-sse/executors/antigravity.ts @@ -26,6 +26,7 @@ import { } from "../services/antigravityCredits.ts"; import { persistCreditBalance, getAllPersistedCreditBalances } from "@/lib/db/creditBalance"; import { setConnectionRateLimitUntil } from "@/lib/db/providers"; +import { markAntigravityModelQuotaExhausted } from "../services/antigravityFamilyCooldown.ts"; import { getMitmAlias } from "@/lib/db/models"; import { MAX_ANTIGRAVITY_OUTPUT_TOKENS, @@ -245,17 +246,15 @@ export function createCreditsExtractionTransform( ); } -/** - * Persist a quota-exhausted cooldown to the DB for `connectionId` so that - * cross-request and post-restart routing skips this connection until the - * cooldown expires. Exported for unit testing. @internal - */ -export function markConnectionQuotaExhausted(connectionId: string, retryAfterMs: number): void { +export function markConnectionQuotaExhausted( + connectionId: string, + retryAfterMs: number, + model?: string | null +): void { try { + if (markAntigravityModelQuotaExhausted(connectionId, retryAfterMs, model)) return; setConnectionRateLimitUntil(connectionId, Date.now() + retryAfterMs); - } catch { - // DB write failure must never crash the request path - } + } catch {} } /** @@ -1620,7 +1619,7 @@ export class AntigravityExecutor extends BaseExecutor { updateAntigravityRemainingCredits ); if (creditsResult) return { kind: "return", result: creditsResult }; - if (retryMs) markConnectionQuotaExhausted(accountId, retryMs); + if (retryMs) markConnectionQuotaExhausted(accountId, retryMs, ctx.model); } return { diff --git a/open-sse/services/accountFallback.ts b/open-sse/services/accountFallback.ts index c1747cedc9..f5bd87c1ed 100644 --- a/open-sse/services/accountFallback.ts +++ b/open-sse/services/accountFallback.ts @@ -49,7 +49,8 @@ import { } from "../../src/shared/constants/providers"; import { resolveUseUpstream429BreakerHints } from "../../src/shared/utils/providerHints"; import { getCodexModelScope } from "../config/codexQuotaScopes.ts"; -import { getQuotaScopedModelForProvider } from "./antigravityQuotaFamily.ts"; +import { getQuotaScopedModelForProvider, isAntigravityQuotaProvider } from "./antigravityQuotaFamily.ts"; +import { persistAntigravityFamilyCooldownIfQuota } from "./antigravityFamilyCooldown.ts"; import { classifyGeminiQuotaMetricFromText, isRpdExhausted, @@ -650,6 +651,9 @@ export async function recordCoreOwnedAntigravityQuotaState({ exactCooldownIsUpstreamReset: retryHintBypassesMaxCooldownMs(fallback.retryHintSource), } ); + if (lockout.cooldownMs > 0 && isProviderExhaustedReason(fallback)) { + persistAntigravityFamilyCooldownIfQuota({ provider, connectionId, model, cooldownMs: lockout.cooldownMs, reason: "quota_exhausted" }); + } return { cooldownMs: lockout.cooldownMs, failureCount: lockout.failureCount }; } @@ -2389,12 +2393,7 @@ export function applyErrorState( // (`markConnectionQuotaExhausted`) so a DB failure can never crash the // chat path. See issue #1 (per-account 429 cascade not persisting). const connId = (account as AccountState | null | undefined)?.id; - if ( - typeof connId === "string" && - connId.length > 0 && - effectiveCooldownMs > 0 && - nextState.rateLimitedUntil - ) { + if (typeof connId === "string" && connId.length > 0 && effectiveCooldownMs > 0 && nextState.rateLimitedUntil && !isAntigravityQuotaProvider(prov)) { try { const untilMs = cooldownUntilMs(nextState.rateLimitedUntil); if (Number.isFinite(untilMs) && untilMs > Date.now()) { diff --git a/open-sse/services/antigravityFamilyCooldown.ts b/open-sse/services/antigravityFamilyCooldown.ts new file mode 100644 index 0000000000..975822a88f --- /dev/null +++ b/open-sse/services/antigravityFamilyCooldown.ts @@ -0,0 +1,158 @@ +/** + * Persist Antigravity/agy quota cooldowns per model family (gemini vs claude) + * on the connection row, without cooling the whole account. + */ +import { lockModel } from "./accountFallback.ts"; +import { + getAntigravityQuotaFamily, + isAntigravityQuotaProvider, +} from "./antigravityQuotaFamily.ts"; + +type JsonRecord = Record; + +const FAMILY_PSD_KEY = "antigravityFamilyRateLimitedUntil"; + +function asRecord(value: unknown): JsonRecord { + return value && typeof value === "object" && !Array.isArray(value) + ? (value as JsonRecord) + : {}; +} + +function parseUntilMs(value: unknown): number { + if (typeof value === "number" && Number.isFinite(value)) return value; + if (typeof value === "string" && value.trim()) { + const ms = /^\d+(\.\d+)?$/.test(value.trim()) ? Number(value) : Date.parse(value); + return Number.isFinite(ms) ? ms : NaN; + } + return NaN; +} + +function dummyModelForFamily(family: "gemini" | "claude"): string { + return family === "gemini" ? "gemini-family-lock" : "claude-family-lock"; +} + +function lockAntigravityFamilyModel( + connectionId: string, + model: string, + reason: string, + cooldownMs: number +): void { + lockModel("agy", connectionId, model, reason, cooldownMs); + lockModel("antigravity", connectionId, model, reason, cooldownMs); +} + +export async function persistAntigravityFamilyCooldown(params: { + connectionId: string; + model: string; + rateLimitedUntil: string; +}): Promise { + if (!params.model.trim()) return null; + const family = getAntigravityQuotaFamily(params.model); + if (family === "other") return null; + + const { getProviderConnectionById, updateProviderConnection } = await import( + "@/lib/db/providers" + ); + const conn = (await getProviderConnectionById(params.connectionId)) as + | { provider?: string; providerSpecificData?: JsonRecord | null } + | null; + if (!conn || !isAntigravityQuotaProvider(conn.provider ?? null)) return null; + + const psd = asRecord(conn.providerSpecificData); + const untils = asRecord(psd[FAMILY_PSD_KEY]); + const existingMs = parseUntilMs(untils[family]); + const nextMs = parseUntilMs(params.rateLimitedUntil); + if (!Number.isFinite(nextMs)) return psd; + if (Number.isFinite(existingMs) && existingMs > Date.now() && existingMs >= nextMs) { + return psd; + } + + const nextPsd: JsonRecord = { + ...psd, + [FAMILY_PSD_KEY]: { ...untils, [family]: params.rateLimitedUntil }, + }; + await updateProviderConnection(params.connectionId, { providerSpecificData: nextPsd }); + return nextPsd; +} + +/** Fire-and-forget family PSD write. RPM/burst 429s must pass reason !== quota_exhausted. */ +export function persistAntigravityFamilyCooldownIfQuota(params: { + provider?: string | null; + connectionId: string; + model?: string | null; + cooldownMs: number; + reason?: string | null; +}): void { + if (!isAntigravityQuotaProvider(params.provider)) return; + if (!params.model?.trim() || params.cooldownMs <= 0) return; + if (params.reason != null && params.reason !== "quota_exhausted") return; + void persistAntigravityFamilyCooldown({ + connectionId: params.connectionId, + model: params.model, + rateLimitedUntil: new Date(Date.now() + params.cooldownMs).toISOString(), + }).catch(() => {}); +} + +export async function persistAntigravityPreflightFamilyLock(params: { + provider: string; + connectionId: string; + model: string; + unavailableUntil: string; +}): Promise { + const cooldownMs = Math.max(0, Date.parse(params.unavailableUntil) - Date.now()); + lockAntigravityFamilyModel(params.connectionId, params.model, "quota_exhausted", cooldownMs); + await persistAntigravityFamilyCooldown({ + connectionId: params.connectionId, + model: params.model, + rateLimitedUntil: params.unavailableUntil, + }); +} + +export function rehydrateAntigravityFamilyLocks( + provider: string, + connectionId: string, + providerSpecificData: JsonRecord | null | undefined +): void { + if (!isAntigravityQuotaProvider(provider)) return; + const untils = asRecord(asRecord(providerSpecificData)[FAMILY_PSD_KEY]); + const now = Date.now(); + for (const family of ["gemini", "claude"] as const) { + const untilMs = parseUntilMs(untils[family]); + if (!Number.isFinite(untilMs) || untilMs <= now) continue; + const model = dummyModelForFamily(family); + const remainingMs = untilMs - now; + lockAntigravityFamilyModel(connectionId, model, "quota_exhausted", remainingMs); + } +} + +export function rehydrateAntigravityFamilyLocksForConnections( + provider: string, + connections: Array<{ id: string; providerSpecificData?: unknown }> +): void { + if (!isAntigravityQuotaProvider(provider)) return; + for (const conn of connections) { + rehydrateAntigravityFamilyLocks( + provider, + conn.id, + conn.providerSpecificData as JsonRecord | null | undefined + ); + } +} + +/** Family lock for executor quota exhaustion. Returns false when model is absent. */ +export function markAntigravityModelQuotaExhausted( + connectionId: string, + retryAfterMs: number, + model?: string | null +): boolean { + if (!model) return false; + lockAntigravityFamilyModel(connectionId, model, "quota_exhausted", retryAfterMs); + persistAntigravityFamilyCooldownIfQuota({ + provider: "agy", + connectionId, + model, + cooldownMs: retryAfterMs, + reason: "quota_exhausted", + }); + return true; +} diff --git a/open-sse/services/antigravityQuotaFamily.ts b/open-sse/services/antigravityQuotaFamily.ts index e9ede18749..94016c1109 100644 --- a/open-sse/services/antigravityQuotaFamily.ts +++ b/open-sse/services/antigravityQuotaFamily.ts @@ -54,3 +54,77 @@ export function getQuotaScopeLabelForProvider( if (provider !== "antigravity" && provider !== "agy") return "model"; return getAntigravityQuotaFamily(model) === "other" ? "model" : "family"; } + +export function isAntigravityQuotaProvider(provider: string | null | undefined): boolean { + return provider === "antigravity" || provider === "agy"; +} + +export function quotaWindowNamesForScope( + names: string[], + scope?: { provider?: string | null; requestedModel?: string | null } +): string[] { + if (!scope?.requestedModel || !isAntigravityQuotaProvider(scope.provider)) return names; + const scoped = selectAntigravityQuotaWindowNames(names, scope.requestedModel); + return scoped.length > 0 ? scoped : names; +} + +/** Min remaining % across scoped windows, or 100 when an Antigravity family scope matched none. */ +export function remainingPercentFromQuotaWindows( + rawWindows: Record, + scope?: { provider?: string | null; requestedModel?: string | null } +): number | null { + const names = Object.keys(rawWindows); + const namesToScan = quotaWindowNamesForScope(names, scope); + let minRemaining: number | null = null; + for (const name of namesToScan) { + const windowInfo = rawWindows[name]; + if (!windowInfo || typeof windowInfo !== "object") continue; + const percentUsed = Number((windowInfo as Record).percentUsed); + if (!Number.isFinite(percentUsed)) continue; + const remaining = Math.max(0, Math.min(100, (1 - percentUsed) * 100)); + minRemaining = minRemaining === null ? remaining : Math.min(minRemaining, remaining); + } + if (minRemaining !== null) return minRemaining; + if (scope?.requestedModel && namesToScan !== names) return 100; + return null; +} + +/** + * Windows that belong to the requested Antigravity family. Claude weekly must + * not ride along on a Gemini request (and the reverse). + */ +export function selectAntigravityQuotaWindowNames( + quotaNames: string[], + requestedModel: string | null | undefined +): string[] { + if (!requestedModel) return quotaNames; + const requestedFamily = getAntigravityQuotaFamily(requestedModel); + const cleanRequestedModel = requestedModel.replace(/^(antigravity|agy)\//, ""); + const bareModel = cleanRequestedModel.includes("/") + ? cleanRequestedModel.slice(cleanRequestedModel.lastIndexOf("/") + 1) + : cleanRequestedModel; + + if (requestedFamily === "other") { + return quotaNames.filter((windowName) => { + const bare = windowName.replace(/^(antigravity|agy)\//, ""); + return bare === bareModel || bare === cleanRequestedModel; + }); + } + + const familyAggregates = + requestedFamily === "gemini" + ? ["gemini_weekly"] + : requestedFamily === "claude" + ? ["claude_gpt_weekly"] + : []; + + const exactWindows = quotaNames.filter((windowName) => { + const bare = windowName.replace(/^(antigravity|agy)\//, ""); + return bare === bareModel; + }); + const aggregateWindows = familyAggregates.filter((key) => quotaNames.includes(key)); + const scoped = [...exactWindows, ...aggregateWindows]; + if (scoped.length > 0) return scoped; + + return quotaNames.filter((windowName) => getAntigravityQuotaFamily(windowName) === requestedFamily); +} diff --git a/open-sse/services/combo.ts b/open-sse/services/combo.ts index 9c316e52c7..c18cd37333 100644 --- a/open-sse/services/combo.ts +++ b/open-sse/services/combo.ts @@ -613,12 +613,13 @@ export async function buildAutoCandidates( const quota = await quotaPromises.get(quotaKey)!; resetWindowAffinity = calculateResetWindowAffinity(quota, resetWindowConfig); if (!quotaCutoffBlocked) { - quotaRemaining = quotaRemainingPercentFromQuota(quota); + quotaRemaining = quotaRemainingPercentFromQuota(quota, { provider, requestedModel: modelStr }); } if (!quotaCutoffBlocked && quotaCutoffEnabled) { const cutoffDecision = evaluateQuotaCutoff( quota as QuotaInfo | null, - buildAutoQuotaThresholds(provider, connection, resilienceSettings) + buildAutoQuotaThresholds(provider, connection, resilienceSettings), + { provider, requestedModel: modelStr } ); if (!cutoffDecision.proceed) { quotaCutoffBlocked = true; @@ -1379,7 +1380,7 @@ async function handleComboChatInner({ resilienceSettings, quotaCutoffResetWindowConfig, combo.name, - log + log, modelStr ); if (quotaCutoff.blocked) { log.info( @@ -4015,8 +4016,5 @@ async function handleRoundRobinCombo({ } log.warn("COMBO-RR", `All models failed | ${msg}`); - return new Response(JSON.stringify({ error: { message: msg } }), { - status, - headers: { "Content-Type": "application/json" }, - }); + return new Response(JSON.stringify({ error: { message: msg } }), { status, headers: { "Content-Type": "application/json" } }); } diff --git a/open-sse/services/combo/comboPredicates.ts b/open-sse/services/combo/comboPredicates.ts index 2c6acb099e..972abc5765 100644 --- a/open-sse/services/combo/comboPredicates.ts +++ b/open-sse/services/combo/comboPredicates.ts @@ -7,6 +7,7 @@ */ import { EXECUTOR_CONTRACT_VIOLATION_CODE } from "../../config/constants.ts"; +import { remainingPercentFromQuotaWindows } from "../antigravityQuotaFamily.ts"; import { errorResponse } from "../../utils/error.ts"; import { parseModel } from "../model.ts"; import { isSelfInflictedUpstreamTimeout } from "../../handlers/chatCore/cooldownClassification.ts"; @@ -431,24 +432,24 @@ export function clampPercent(value: number): number { return Math.max(0, Math.min(100, value)); } -export function quotaRemainingPercentFromQuota(quota: unknown): number { +export function quotaRemainingPercentFromQuota( + quota: unknown, + scope?: { provider?: string | null; requestedModel?: string | null } +): number { if (!quota || typeof quota !== "object") return 100; const record = quota as Record; - if (record.limitReached === true) return 0; const windows = record.windows; if (windows && typeof windows === "object" && !Array.isArray(windows)) { - let minRemaining: number | null = null; - for (const windowInfo of Object.values(windows as Record)) { - if (!windowInfo || typeof windowInfo !== "object") continue; - const percentUsed = Number((windowInfo as Record).percentUsed); - if (!Number.isFinite(percentUsed)) continue; - const remaining = clampPercent((1 - percentUsed) * 100); - minRemaining = minRemaining === null ? remaining : Math.min(minRemaining, remaining); - } - if (minRemaining !== null) return minRemaining; + const fromWindows = remainingPercentFromQuotaWindows( + windows as Record, + scope + ); + if (fromWindows !== null) return fromWindows; } + if (record.limitReached === true) return 0; + const percentUsed = Number(record.percentUsed); if (Number.isFinite(percentUsed)) return clampPercent((1 - percentUsed) * 100); return 100; diff --git a/open-sse/services/combo/nativeCodexTurnPin.ts b/open-sse/services/combo/nativeCodexTurnPin.ts index 9910d8d993..2b6d94945f 100644 --- a/open-sse/services/combo/nativeCodexTurnPin.ts +++ b/open-sse/services/combo/nativeCodexTurnPin.ts @@ -245,7 +245,8 @@ export async function isPinnedTargetModelScopedUnusable(args: { resilienceSettings, quotaCutoffResetWindowConfig, comboName, - log ?? { info: () => {}, warn: () => {}, error: () => {}, debug: () => {} } + log ?? { info: () => {}, warn: () => {}, error: () => {}, debug: () => {} }, + target.modelStr ); if (cutoff.blocked) return true; } diff --git a/open-sse/services/combo/quotaExhaustionCutoff.ts b/open-sse/services/combo/quotaExhaustionCutoff.ts index 2dc78dbe1b..a73d3628f4 100644 --- a/open-sse/services/combo/quotaExhaustionCutoff.ts +++ b/open-sse/services/combo/quotaExhaustionCutoff.ts @@ -97,7 +97,8 @@ export async function resolveQuotaExhaustionCutoffForTarget( resilienceSettings: ResilienceSettings | null | undefined, resetWindowConfig: ResetWindowConfig, comboName: string, - log: { debug?: (...args: unknown[]) => void; warn?: (...args: unknown[]) => void } + log: { debug?: (...args: unknown[]) => void; warn?: (...args: unknown[]) => void }, + requestedModel?: string | null ): Promise<{ blocked: boolean; reason?: string }> { const quotaCutoffEnabled = (resilienceSettings ?? resolveResilienceSettings(null))?.quotaPreflight?.enabled === true; @@ -126,7 +127,8 @@ export async function resolveQuotaExhaustionCutoffForTarget( }); const cutoffDecision = evaluateQuotaCutoff( quota as QuotaInfo | null, - buildAutoQuotaThresholds(provider, connection, resilienceSettings) + buildAutoQuotaThresholds(provider, connection, resilienceSettings), + { provider, requestedModel: requestedModel ?? null } ); if (!cutoffDecision.proceed) { return { blocked: true, reason: cutoffDecision.reason || "quota_exhausted" }; diff --git a/open-sse/services/quotaPreflight.ts b/open-sse/services/quotaPreflight.ts index a6c7d99aef..e22d8597b6 100644 --- a/open-sse/services/quotaPreflight.ts +++ b/open-sse/services/quotaPreflight.ts @@ -21,12 +21,22 @@ import { isCompatibleProviderConnectionId } from "@/shared/utils/compatibleProviderId"; import { isFeatureFlagEnabled } from "@/shared/utils/featureFlags"; import { fetchNewApiAggregatorQuota } from "./newApiAggregatorQuotaFetcher.ts"; +import { + isAntigravityQuotaProvider, + selectAntigravityQuotaWindowNames, +} from "./antigravityQuotaFamily.ts"; export interface PreflightQuotaResult { proceed: boolean; reason?: string; quotaPercent?: number; resetAt?: string | null; + windowName?: string | null; +} + +export interface QuotaCutoffScope { + provider?: string | null; + requestedModel?: string | null; } export interface QuotaWindowInfo { @@ -156,15 +166,36 @@ function isRemainingAtOrBelowThreshold( return remainingPercent <= thresholdPercent + REMAINING_PERCENT_EPSILON; } -function exhaustedResult(quotaPercent: number, resetAt: string | null): PreflightQuotaResult { +function exhaustedResult( + quotaPercent: number, + resetAt: string | null, + windowName?: string | null +): PreflightQuotaResult { return { proceed: false, reason: "quota_exhausted", quotaPercent, resetAt, + windowName: windowName ?? null, }; } +function windowsForScope( + windows: NonNullable, + scope?: QuotaCutoffScope +): NonNullable { + if (!scope?.requestedModel || !isAntigravityQuotaProvider(scope.provider ?? null)) { + return windows; + } + const selected = selectAntigravityQuotaWindowNames(Object.keys(windows), scope.requestedModel); + if (selected.length === 0) return windows; + const scoped: NonNullable = {}; + for (const name of selected) { + if (windows[name]) scoped[name] = windows[name]; + } + return Object.keys(scoped).length > 0 ? scoped : windows; +} + function limitReachedResult(quota: QuotaInfo): PreflightQuotaResult { return exhaustedResult( Number.isFinite(quota.percentUsed) ? quota.percentUsed : 1, @@ -201,7 +232,9 @@ function quotaWindowCutoffResult( worstResetAt = windowInfo.resetAt ?? null; } - return worstWindow === null ? null : exhaustedResult(worstUsedPercent, worstResetAt); + return worstWindow === null + ? null + : exhaustedResult(worstUsedPercent, worstResetAt, worstWindow); } function quotaPercentCutoffResult( @@ -227,21 +260,27 @@ function quotaPercentCutoffResult( */ export function evaluateQuotaCutoff( quota: QuotaInfo | null | undefined, - thresholds?: PreflightQuotaThresholds + thresholds?: PreflightQuotaThresholds, + scope?: QuotaCutoffScope ): PreflightQuotaResult { if (!quota) return { proceed: true }; - if (quota.limitReached === true) return limitReachedResult(quota); const windows = quota.windows; if (windows && Object.keys(windows).length > 0) { - return ( - quotaWindowCutoffResult(windows, thresholds) ?? { - proceed: true, - quotaPercent: quota.percentUsed, - } - ); + const scopedWindows = windowsForScope(windows, scope); + const cutoff = quotaWindowCutoffResult(scopedWindows, thresholds); + if (cutoff) return cutoff; + if (isAntigravityQuotaProvider(scope?.provider ?? null) && scope?.requestedModel) { + return { proceed: true, quotaPercent: quota.percentUsed }; + } + if (quota.limitReached === true) return limitReachedResult(quota); + return { + proceed: true, + quotaPercent: quota.percentUsed, + }; } + if (quota.limitReached === true) return limitReachedResult(quota); return quotaPercentCutoffResult(quota, thresholds); } @@ -297,61 +336,40 @@ export async function preflightQuota( return { proceed: true }; } - if (quota.limitReached === true) { - return limitReachedResult(quota); - } - - // Per-window evaluation — only when the fetcher surfaces a windows map. - // We block as soon as ANY single window's remaining quota drops to its - // configured cutoff or below; warnings are logged independently per window. - if (quota.windows && Object.keys(quota.windows).length > 0) { - let worstUsedPercent = 0; - let worstWindow: string | null = null; - let worstResetAt: string | null = null; - for (const [windowName, windowInfo] of Object.entries(quota.windows)) { - const minRemainingPercent = resolveOrDefault( - thresholds?.resolveMinRemainingPercent, - windowName, - DEFAULT_MIN_REMAINING_PERCENT - ); + const requestedModel = + typeof connection.requestedModel === "string" ? connection.requestedModel : null; + const scope: QuotaCutoffScope = { provider, requestedModel }; + const windows = quota.windows; + if (windows && Object.keys(windows).length > 0) { + const scopedWindows = windowsForScope(windows, scope); + for (const [windowName, windowInfo] of Object.entries(scopedWindows)) { const warnRemainingPercent = resolveOrDefault( thresholds?.resolveWarnRemainingPercent, windowName, DEFAULT_WARN_REMAINING_PERCENT ); const remainingPercent = remainingPercentFrom(windowInfo.percentUsed); - - if (isRemainingAtOrBelowThreshold(remainingPercent, minRemainingPercent)) { - // Track the most-depleted blocking window so the response can name it. - if (windowInfo.percentUsed > worstUsedPercent) { - worstUsedPercent = windowInfo.percentUsed; - worstWindow = windowName; - worstResetAt = windowInfo.resetAt ?? null; - } else if (worstWindow === null) { - worstWindow = windowName; - worstResetAt = windowInfo.resetAt ?? null; - } - } else if (isRemainingAtOrBelowThreshold(remainingPercent, warnRemainingPercent)) { + if (isRemainingAtOrBelowThreshold(remainingPercent, warnRemainingPercent)) { console.warn( `[QuotaPreflight] ${provider}/${connectionId} ${windowName}: ${remainingPercent.toFixed(1)}% remaining — approaching cutoff` ); } } + } - if (worstWindow !== null) { - const worstRemaining = remainingPercentFrom(worstUsedPercent); - console.info( - `[QuotaPreflight] ${provider}/${connectionId} ${worstWindow}: ${worstRemaining.toFixed(1)}% remaining — switching` - ); - return { - proceed: false, - reason: "quota_exhausted", - quotaPercent: worstUsedPercent, - resetAt: worstResetAt, - }; - } - - return { proceed: true, quotaPercent: quota.percentUsed }; + const decision = evaluateQuotaCutoff(quota, thresholds, scope); + if (!decision.proceed) { + const windowLabel = decision.windowName ? ` ${decision.windowName}` : ""; + const remaining = Number.isFinite(decision.quotaPercent) + ? remainingPercentFrom(decision.quotaPercent as number).toFixed(1) + : "?"; + console.info( + `[QuotaPreflight] ${provider}/${connectionId}${windowLabel}: ${remaining}% remaining - switching` + ); + return decision; + } + if (windows && Object.keys(windows).length > 0) { + return decision; } // Legacy single-signal path for fetchers that don't expose per-window data. diff --git a/src/domain/quotaCache.ts b/src/domain/quotaCache.ts index 377512ee13..0065eb61ea 100644 --- a/src/domain/quotaCache.ts +++ b/src/domain/quotaCache.ts @@ -38,7 +38,7 @@ import { resolveCodexAccount, type CodexPersistedQuotaState, } from "@omniroute/open-sse/services/codexAccount/index.ts"; -import { getAntigravityQuotaFamily } from "@omniroute/open-sse/services/antigravityQuotaFamily.ts"; +import { selectAntigravityQuotaWindowNames } from "@omniroute/open-sse/services/antigravityQuotaFamily.ts"; // ─── Types ────────────────────────────────────────────────────────────────── @@ -273,37 +273,7 @@ function resolveAntigravityQuotaWindowsForModel( quotaNames: string[], requestedModel: string ): string[] { - const requestedFamily = getAntigravityQuotaFamily(requestedModel); - const cleanRequestedModel = requestedModel.replace(/^(antigravity|agy)\//, ""); - const bareModel = cleanRequestedModel.includes("/") - ? cleanRequestedModel.slice(cleanRequestedModel.lastIndexOf("/") + 1) - : cleanRequestedModel; - - if (requestedFamily === "other") { - return quotaNames.filter((windowName) => { - const bare = windowName.replace(/^(antigravity|agy)\//, ""); - return bare === bareModel || bare === cleanRequestedModel; - }); - } - - const familyAggregates = - requestedFamily === "gemini" - ? ["gemini_weekly"] - : requestedFamily === "claude" - ? ["claude_gpt_weekly"] - : []; - - const exactWindows = quotaNames.filter((windowName) => { - const bare = windowName.replace(/^(antigravity|agy)\//, ""); - return bare === bareModel; - }); - const aggregateWindows = familyAggregates.filter((key) => quotaNames.includes(key)); - const scoped = [...exactWindows, ...aggregateWindows]; - if (scoped.length > 0) return scoped; - - return quotaNames.filter( - (windowName) => getAntigravityQuotaFamily(windowName) === requestedFamily - ); + return selectAntigravityQuotaWindowNames(quotaNames, requestedModel); } function isAntigravityQuotaExhausted( diff --git a/src/sse/services/auth.ts b/src/sse/services/auth.ts index ec2b42a17d..697c8a036f 100644 --- a/src/sse/services/auth.ts +++ b/src/sse/services/auth.ts @@ -47,7 +47,12 @@ import { hydrateCodexQuotaCacheForRequest, isQuotaExhaustedForRequest, } from "@/domain/quotaCache"; -import { getQuotaScopeLabelForProvider } from "@omniroute/open-sse/services/antigravityQuotaFamily.ts"; +import { + getQuotaScopeLabelForProvider, + isAntigravityQuotaProvider, +} from "@omniroute/open-sse/services/antigravityQuotaFamily.ts"; +import { rehydrateAntigravityFamilyLocksForConnections, persistAntigravityFamilyCooldownIfQuota } from "@omniroute/open-sse/services/antigravityFamilyCooldown.ts"; +import { markQuotaPreflightAccountUnavailable } from "./quotaPreflightUnavailable.ts"; import { getCreditsMode } from "@omniroute/open-sse/services/antigravityCredits.ts"; import { preferAntigravityConnectionsWithStoredProject } from "@omniroute/open-sse/services/antigravityProjectPersistence.ts"; import { @@ -891,43 +896,6 @@ function buildQuotaPreflightRateLimitedResult( lastErrorCode: 429, }; } -function quotaPreflightUnavailableUntil(resetAt?: string | null): string { - const resetMs = parseFutureDateMs(resetAt ?? null); - return new Date(resetMs ?? Date.now() + 5 * 60 * 1000).toISOString(); -} -async function markQuotaPreflightAccountUnavailable( - provider: string, - connectionId: string, - preflight: { quotaPercent?: number; resetAt?: string | null }, - requestedModel: string | null -): Promise { - const unavailableUntil = quotaPreflightUnavailableUntil(preflight.resetAt ?? null); - if (provider === "codex" && requestedModel?.trim()) { - await persistCodexChildCooldown({ - connectionId, - model: requestedModel, - rateLimitedUntil: unavailableUntil, - }); - return unavailableUntil; - } - - const percentLabel = Number.isFinite(preflight.quotaPercent) - ? `${Math.round((preflight.quotaPercent as number) * 100)}%` - : "exhausted"; - const modelLabel = requestedModel ? ` for ${requestedModel}` : ""; - - await updateProviderConnection(connectionId, { - rateLimitedUntil: unavailableUntil, - testStatus: "unavailable", - lastError: `Quota preflight blocked${modelLabel}: ${percentLabel}`, - lastErrorType: "quota_exhausted", - lastErrorSource: "quota_preflight", - errorCode: 429, - lastErrorAt: new Date().toISOString(), - }); - - return unavailableUntil; -} // Provider-scoped mutexes prevent race conditions during account selection without // serializing unrelated providers behind a single global lock. @@ -1326,6 +1294,7 @@ export async function getProviderCredentials( ); } } + rehydrateAntigravityFamilyLocksForConnections(provider, connections); // allowedConnections: restrict to specific connection IDs (from API key policy, #363) if (allowedConnections && allowedConnections.length > 0) { connections = connections.filter((conn) => allowedConnections.includes(conn.id)); @@ -2405,7 +2374,9 @@ export async function getProviderCredentialsWithQuotaPreflight( return defaultThresholdPercent; }; // #6842: openrouter also needs requestedModel, for the :free-window check. - const modelAwarePreflight = provider === "codex" || provider === "openrouter"; + // agy/antigravity need it so Claude weekly cannot cool a Gemini request. + const modelAwarePreflight = + provider === "codex" || provider === "openrouter" || isAntigravityQuotaProvider(provider); const preflightCredentials = requestedModel && modelAwarePreflight ? { ...credentials, requestedModel } : credentials; let preflight; @@ -2967,6 +2938,7 @@ export async function markAccountUnavailable( "AUTH", `Model-only lockout for ${provider}:${model} — ${status} ${reason} ${Math.ceil(lockout.cooldownMs / 1000)}s (failureCount=${lockout.failureCount}, connection stays active)` ); + persistAntigravityFamilyCooldownIfQuota({ provider, connectionId, model, cooldownMs: lockout.cooldownMs, reason }); return { shouldFallback: true, cooldownMs: lockout.cooldownMs }; } const result = fallbackResult; diff --git a/src/sse/services/quotaPreflightUnavailable.ts b/src/sse/services/quotaPreflightUnavailable.ts new file mode 100644 index 0000000000..f1ac373a50 --- /dev/null +++ b/src/sse/services/quotaPreflightUnavailable.ts @@ -0,0 +1,61 @@ +import { persistCodexChildCooldown } from "@omniroute/open-sse/services/codexAccount/index.ts"; +import { persistAntigravityPreflightFamilyLock } from "@omniroute/open-sse/services/antigravityFamilyCooldown.ts"; +import { isAntigravityQuotaProvider } from "@omniroute/open-sse/services/antigravityQuotaFamily.ts"; +import { cooldownUntilMs } from "@omniroute/open-sse/services/accountFallback.ts"; +import { updateProviderConnection } from "@/lib/db/providers"; + +function parseFutureDateMs(value: string | null): number | null { + if (!value) return null; + const ms = cooldownUntilMs(value); + if (!Number.isFinite(ms) || ms <= Date.now()) return null; + return ms; +} + +function quotaPreflightUnavailableUntil(resetAt?: string | null): string { + const resetMs = parseFutureDateMs(resetAt ?? null); + return new Date(resetMs ?? Date.now() + 5 * 60 * 1000).toISOString(); +} + +export async function markQuotaPreflightAccountUnavailable( + provider: string, + connectionId: string, + preflight: { quotaPercent?: number; resetAt?: string | null }, + requestedModel: string | null +): Promise { + const unavailableUntil = quotaPreflightUnavailableUntil(preflight.resetAt ?? null); + if (provider === "codex" && requestedModel?.trim()) { + await persistCodexChildCooldown({ + connectionId, + model: requestedModel, + rateLimitedUntil: unavailableUntil, + }); + return unavailableUntil; + } + + if (isAntigravityQuotaProvider(provider) && requestedModel?.trim()) { + await persistAntigravityPreflightFamilyLock({ + provider, + connectionId, + model: requestedModel, + unavailableUntil, + }); + return unavailableUntil; + } + + const percentLabel = Number.isFinite(preflight.quotaPercent) + ? `${Math.round((preflight.quotaPercent as number) * 100)}%` + : "exhausted"; + const modelLabel = requestedModel ? ` for ${requestedModel}` : ""; + + await updateProviderConnection(connectionId, { + rateLimitedUntil: unavailableUntil, + testStatus: "unavailable", + lastError: `Quota preflight blocked${modelLabel}: ${percentLabel}`, + lastErrorType: "quota_exhausted", + lastErrorSource: "quota_preflight", + errorCode: 429, + lastErrorAt: new Date().toISOString(), + }); + + return unavailableUntil; +} diff --git a/stryker.conf.json b/stryker.conf.json index 23dd4fb2d5..58dd1f4d4f 100644 --- a/stryker.conf.json +++ b/stryker.conf.json @@ -67,6 +67,7 @@ "tests/unit/agentrouter-lock-scope-10334.test.ts", "tests/unit/alibaba-free-tier-exhaustion.test.ts", "tests/unit/anthropic-thinking-signature-recovery.test.ts", + "tests/unit/agy-family-not-connection-cooldown.test.ts", "tests/unit/antigravity-429-quota-cooldown.test.ts", "tests/unit/antigravity-429-quota-tdd.test.ts", "tests/unit/antigravity-prefer-stored-project.test.ts", diff --git a/tests/unit/agy-family-not-connection-cooldown.test.ts b/tests/unit/agy-family-not-connection-cooldown.test.ts new file mode 100644 index 0000000000..9c4499e8cb --- /dev/null +++ b/tests/unit/agy-family-not-connection-cooldown.test.ts @@ -0,0 +1,258 @@ +/** + * Claude weekly exhaustion must not cool the whole agy/antigravity connection. + * Gemini on the same account stays routable; only family:claude is locked. + */ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-agy-family-cd-")); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.API_KEY_SECRET = process.env.API_KEY_SECRET || "agy-family-test-secret"; + +const core = await import("../../src/lib/db/core.ts"); +const providersDb = await import("../../src/lib/db/providers.ts"); +const quotaPreflight = await import("../../open-sse/services/quotaPreflight.ts"); +const family = await import("../../open-sse/services/antigravityQuotaFamily.ts"); +const fallback = await import("../../open-sse/services/accountFallback.ts"); +const { markConnectionQuotaExhausted } = await import("../../open-sse/executors/antigravity.ts"); +const { quotaRemainingPercentFromQuota } = await import( + "../../open-sse/services/combo/comboPredicates.ts" +); + +const CLAUDE_RESET = "2026-09-06T17:38:10.000Z"; +const GEMINI_RESET = "2026-09-09T09:59:00.000Z"; + +function mixedWindows() { + return { + claude_gpt_weekly: { percentUsed: 1, resetAt: CLAUDE_RESET }, + gemini_weekly: { percentUsed: 0.022, resetAt: GEMINI_RESET }, + "gemini-3.1-flash-lite": { percentUsed: 0.1, resetAt: null }, + }; +} + +test.after(() => { + fallback.clearAllModelLockouts(); + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("selectAntigravityQuotaWindowNames keeps Claude weekly off a Gemini request", () => { + const names = family.selectAntigravityQuotaWindowNames(Object.keys(mixedWindows()), "gemini-3.1-flash-lite"); + assert.deepEqual(names.sort(), ["gemini-3.1-flash-lite", "gemini_weekly"].sort()); +}); + +test("preflightQuota proceeds on Gemini when only Claude weekly is exhausted", async () => { + quotaPreflight.registerQuotaFetcher("agy", async () => ({ + used: 0, + total: 0, + percentUsed: 1, + limitReached: true, + windows: mixedWindows(), + })); + + const result = await quotaPreflight.preflightQuota("agy", "conn-1", { + requestedModel: "agy/gemini-3.1-flash-lite", + }); + assert.equal(result.proceed, true, "Gemini must not inherit Claude weekly exhaustion"); +}); + +test("preflightQuota blocks Gemini when gemini_weekly is exhausted", async () => { + quotaPreflight.registerQuotaFetcher("agy-gemini-dead", async () => ({ + used: 0, + total: 0, + percentUsed: 0.99, + windows: { + claude_gpt_weekly: { percentUsed: 0.1, resetAt: CLAUDE_RESET }, + gemini_weekly: { percentUsed: 0.99, resetAt: GEMINI_RESET }, + }, + })); + + const result = await quotaPreflight.preflightQuota("agy-gemini-dead", "conn-2", { + requestedModel: "gemini-3.1-flash-lite", + }); + assert.equal(result.proceed, false); + assert.equal(result.windowName, "gemini_weekly"); + assert.equal(result.resetAt, GEMINI_RESET); +}); + +test("evaluateQuotaCutoff with requestedModel ignores the other family window", () => { + const quota = { + used: 0, + total: 0, + percentUsed: 1, + limitReached: true, + windows: mixedWindows(), + }; + const gemini = quotaPreflight.evaluateQuotaCutoff(quota, undefined, { + provider: "antigravity", + requestedModel: "gemini-3.1-flash-lite", + }); + assert.equal(gemini.proceed, true); + + const claude = quotaPreflight.evaluateQuotaCutoff(quota, undefined, { + provider: "agy", + requestedModel: "claude-opus-4-6-thinking", + }); + assert.equal(claude.proceed, false); + assert.equal(claude.windowName, "claude_gpt_weekly"); +}); + +test("quotaRemainingPercentFromQuota for Gemini uses Gemini windows, not Claude", () => { + const quota = { windows: mixedWindows(), percentUsed: 1, limitReached: true }; + const remaining = quotaRemainingPercentFromQuota(quota, { + provider: "agy", + requestedModel: "gemini-3.1-flash-lite", + }); + assert.ok(remaining > 50, `expected Gemini remaining, got ${remaining}`); +}); + +test("markConnectionQuotaExhausted with a Gemini model locks the family, not the row", async () => { + fallback.clearAllModelLockouts(); + const conn = await providersDb.createProviderConnection({ + provider: "agy", + authType: "oauth", + name: "agy-family-gemini", + }); + const connId = (conn as { id: string }).id; + + markConnectionQuotaExhausted(connId, 24 * 60 * 60 * 1000, "gemini-3.1-flash-lite"); + + assert.equal( + providersDb.isConnectionRateLimited(connId), + false, + "connection row must stay selectable for the other family" + ); + assert.equal(fallback.isModelLocked("agy", connId, "gemini-3.1-flash-lite"), true); + assert.equal(fallback.isModelLocked("agy", connId, "gemini-3.7-flash-high"), true); + assert.equal(fallback.isModelLocked("agy", connId, "claude-opus-4-6-thinking"), false); +}); + +test("Antigravity RPM 429 stays exact-model and does not persist a family cooldown", async () => { + fallback.clearAllModelLockouts(); + const auth = await import("../../src/sse/services/auth.ts"); + const conn = await providersDb.createProviderConnection({ + provider: "antigravity", + authType: "oauth", + email: "rpm@example.test", + accessToken: "tok-rpm", + isActive: true, + testStatus: "active", + }); + const connId = (conn as { id: string }).id; + + await auth.markAccountUnavailable( + connId, + 429, + "RESOURCE_EXHAUSTED: Resource has been exhausted (requests per minute / RPM limit was reached)", + "antigravity", + "gemini-3-pro" + ); + + const sibling = await auth.getProviderCredentials("antigravity", null, null, "gemini-2.5-pro"); + assert.ok(sibling && !("allRateLimited" in sibling && sibling.allRateLimited)); + assert.equal(sibling.connectionId, connId); + + const fresh = await providersDb.getProviderConnectionById(connId); + const psd = (fresh as { providerSpecificData?: Record }).providerSpecificData; + assert.equal( + psd && typeof psd === "object" ? psd.antigravityFamilyRateLimitedUntil : undefined, + undefined + ); + await providersDb.updateProviderConnection(connId, { isActive: false }); +}); + +test("Antigravity QPM 429 stays exact-model and does not persist a family cooldown", async () => { + fallback.clearAllModelLockouts(); + const auth = await import("../../src/sse/services/auth.ts"); + const conn = await providersDb.createProviderConnection({ + provider: "antigravity", + authType: "oauth", + email: "qpm@example.test", + accessToken: "tok-qpm", + isActive: true, + testStatus: "active", + }); + const connId = (conn as { id: string }).id; + + await auth.markAccountUnavailable( + connId, + 429, + "RESOURCE_EXHAUSTED: Resource has been exhausted (queries per minute limit was reached)", + "antigravity", + "gemini-3-pro" + ); + + const sibling = await auth.getProviderCredentials("antigravity", null, null, "gemini-2.5-pro"); + assert.ok(sibling && !("allRateLimited" in sibling && sibling.allRateLimited)); + assert.equal(sibling.connectionId, connId); + + const fresh = await providersDb.getProviderConnectionById(connId); + const psd = (fresh as { providerSpecificData?: Record }).providerSpecificData; + assert.equal( + psd && typeof psd === "object" ? psd.antigravityFamilyRateLimitedUntil : undefined, + undefined + ); + await providersDb.updateProviderConnection(connId, { isActive: false }); +}); + +test("persisted family cooldown rehydrates after a process-local lockout wipe", async () => { + fallback.clearAllModelLockouts(); + const conn = await providersDb.createProviderConnection({ + provider: "antigravity", + authType: "oauth", + name: "ag-family-persist", + }); + const connId = (conn as { id: string }).id; + const until = new Date(Date.now() + 60 * 60 * 1000).toISOString(); + + const { persistAntigravityFamilyCooldown, rehydrateAntigravityFamilyLocks } = await import( + "../../open-sse/services/antigravityFamilyCooldown.ts" + ); + await persistAntigravityFamilyCooldown({ + connectionId: connId, + model: "claude-sonnet-4", + rateLimitedUntil: until, + }); + + fallback.clearAllModelLockouts(); + assert.equal(fallback.isModelLocked("antigravity", connId, "claude-opus-4"), false); + + const fresh = await providersDb.getProviderConnectionById(connId); + rehydrateAntigravityFamilyLocks( + "antigravity", + connId, + (fresh as { providerSpecificData?: Record }).providerSpecificData + ); + assert.equal(fallback.isModelLocked("antigravity", connId, "claude-opus-4"), true); + assert.equal(fallback.isModelLocked("antigravity", connId, "gemini-3.1-flash-lite"), false); + assert.equal(providersDb.isConnectionRateLimited(connId), false); +}); + +test("preflight family lock covers both agy and antigravity spellings", async () => { + fallback.clearAllModelLockouts(); + const { persistAntigravityPreflightFamilyLock } = await import( + "../../open-sse/services/antigravityFamilyCooldown.ts" + ); + const conn = await providersDb.createProviderConnection({ + provider: "antigravity", + authType: "oauth", + name: "ag-preflight-alias", + }); + const connId = (conn as { id: string }).id; + const until = new Date(Date.now() + 60 * 60 * 1000).toISOString(); + + await persistAntigravityPreflightFamilyLock({ + provider: "antigravity", + connectionId: connId, + model: "claude-sonnet-4", + unavailableUntil: until, + }); + + assert.equal(fallback.isModelLocked("antigravity", connId, "claude-opus-4"), true); + assert.equal(fallback.isModelLocked("agy", connId, "claude-opus-4"), true); + assert.equal(fallback.isModelLocked("antigravity", connId, "gemini-3.1-flash-lite"), false); + assert.equal(fallback.isModelLocked("agy", connId, "gemini-3.1-flash-lite"), false); +}); diff --git a/tests/unit/hard-session-lease-bypass-inventory.test.ts b/tests/unit/hard-session-lease-bypass-inventory.test.ts index 22e6954a76..97e713a0c7 100644 --- a/tests/unit/hard-session-lease-bypass-inventory.test.ts +++ b/tests/unit/hard-session-lease-bypass-inventory.test.ts @@ -84,6 +84,8 @@ const EXPECTED: Record> = { "open-sse/handlers/cursorCliProxy.ts": 1, "open-sse/services/alibabaFreeTier.ts": 1, "open-sse/services/alibabaFreeTierQuotaFetcher.ts": 1, + // Family cooldown persist looks the row up to write PSD, not dispatch. + "open-sse/services/antigravityFamilyCooldown.ts": 1, // v3.8.50 back-merge additions (f95b03d7): combo routing infra and the // volcengine-plan binding/auto-sync services query connections the same // way as their classified siblings. From a47d2e521ea9345b9f772ea8dc9a58c13941c026 Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Thu, 3 Sep 2026 11:46:06 -0400 Subject: [PATCH 039/143] feat(providers): add SeekAi OpenAI-compatible New-API gateway (#12557) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check:provider-consistency` OK (273 entradas REGISTRY, **356** providers canônicos), `check-docs-counts-sync` exit 0 e **300/300** nos testes que a leva toca. O crescimento de arquivo que os PRs empilham uns sobre os outros foi rebaselinado num único registro datado (`_rebaseline_2026_09_03_houminxi_batch`), com a decomposição por arquivo: `providers/page.tsx` +18 (import CSV do #12504 + busca do #12495 no mesmo painel), `accountFallback.ts` +6 (o #12566 sobre o rebaseline que o #12590 já registrou — os dois tocam `checkFallbackError`) e `chatCore.ts` +3 (invalidez de cache de quota no 429 do #12325). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. --- AGENTS.md | 2 +- README.md | 6 +- changelog.d/features/11786-seekai-provider.md | 1 + config/quality/file-size-baseline.json | 3 +- docs/diagrams/cli-terminal.svg | 2 +- docs/diagrams/comparison-table.svg | 2 +- docs/diagrams/promise-pillars.svg | 6 +- docs/diagrams/readme-hero.svg | 4 +- docs/i18n/ar/llm.txt | 4 +- docs/i18n/az/llm.txt | 4 +- docs/i18n/bg/llm.txt | 4 +- docs/i18n/bn/llm.txt | 4 +- docs/i18n/cs/llm.txt | 4 +- docs/i18n/da/llm.txt | 4 +- docs/i18n/de/llm.txt | 4 +- docs/i18n/es/llm.txt | 4 +- docs/i18n/fa/llm.txt | 4 +- docs/i18n/fi/llm.txt | 4 +- docs/i18n/fr/llm.txt | 4 +- docs/i18n/gu/llm.txt | 4 +- docs/i18n/he/llm.txt | 4 +- docs/i18n/hi/llm.txt | 4 +- docs/i18n/hu/llm.txt | 4 +- docs/i18n/id/llm.txt | 4 +- docs/i18n/it/llm.txt | 4 +- docs/i18n/ja/llm.txt | 4 +- docs/i18n/ko/llm.txt | 4 +- docs/i18n/mr/llm.txt | 4 +- docs/i18n/ms/llm.txt | 4 +- docs/i18n/nl/llm.txt | 4 +- docs/i18n/no/llm.txt | 4 +- docs/i18n/phi/llm.txt | 4 +- docs/i18n/pl/llm.txt | 4 +- docs/i18n/pt-BR/llm.txt | 4 +- docs/i18n/pt/llm.txt | 4 +- docs/i18n/ro/llm.txt | 4 +- docs/i18n/ru/llm.txt | 4 +- docs/i18n/sk/llm.txt | 4 +- docs/i18n/sv/llm.txt | 4 +- docs/i18n/sw/llm.txt | 4 +- docs/i18n/ta/llm.txt | 4 +- docs/i18n/te/llm.txt | 4 +- docs/i18n/th/llm.txt | 4 +- docs/i18n/tr/llm.txt | 4 +- docs/i18n/uk-UA/llm.txt | 4 +- docs/i18n/ur/llm.txt | 4 +- docs/i18n/vi/llm.txt | 4 +- docs/i18n/zh-CN/llm.txt | 4 +- docs/i18n/zh-TW/llm.txt | 4 +- docs/reference/PROVIDER_REFERENCE.md | 9 +-- llm.txt | 4 +- open-sse/config/providers/index.ts | 2 + .../config/providers/registry/seekai/index.ts | 18 +++++ package.json | 2 +- public/images/tier-flow-dark.svg | 6 +- public/images/tier-flow-light.svg | 6 +- src/i18n/messages/ar.json | 1 + src/i18n/messages/az.json | 1 + src/i18n/messages/bg.json | 1 + src/i18n/messages/bn.json | 1 + src/i18n/messages/cs.json | 1 + src/i18n/messages/da.json | 1 + src/i18n/messages/de.json | 1 + src/i18n/messages/en.json | 1 + src/i18n/messages/es.json | 1 + src/i18n/messages/fa.json | 1 + src/i18n/messages/fi.json | 1 + src/i18n/messages/fr.json | 1 + src/i18n/messages/gu.json | 1 + src/i18n/messages/he.json | 1 + src/i18n/messages/hi.json | 1 + src/i18n/messages/hu.json | 1 + src/i18n/messages/id.json | 1 + src/i18n/messages/it.json | 1 + src/i18n/messages/ja.json | 1 + src/i18n/messages/ko.json | 1 + src/i18n/messages/mr.json | 1 + src/i18n/messages/ms.json | 1 + src/i18n/messages/nl.json | 1 + src/i18n/messages/no.json | 1 + src/i18n/messages/phi.json | 1 + src/i18n/messages/pl.json | 1 + src/i18n/messages/pt-BR.json | 1 + src/i18n/messages/pt.json | 1 + src/i18n/messages/ro.json | 1 + src/i18n/messages/ru.json | 1 + src/i18n/messages/sk.json | 1 + src/i18n/messages/sv.json | 1 + src/i18n/messages/sw.json | 1 + src/i18n/messages/ta.json | 1 + src/i18n/messages/te.json | 1 + src/i18n/messages/th.json | 1 + src/i18n/messages/tr.json | 1 + src/i18n/messages/uk-UA.json | 1 + src/i18n/messages/ur.json | 1 + src/i18n/messages/vi.json | 1 + src/i18n/messages/zh-CN.json | 1 + src/i18n/messages/zh-TW.json | 1 + src/shared/constants/config.ts | 1 + src/shared/constants/providers.ts | 1 + .../constants/providers/apikey/gateways.ts | 20 ++++++ tests/snapshots/provider/translate-path.json | 23 +++++++ .../provider-node-reserved-prefix.test.ts | 3 +- tests/unit/providers-constants-split.test.ts | 13 ++-- tests/unit/seekai-provider.test.ts | 67 +++++++++++++++++++ 105 files changed, 293 insertions(+), 114 deletions(-) create mode 100644 changelog.d/features/11786-seekai-provider.md create mode 100644 open-sse/config/providers/registry/seekai/index.ts create mode 100644 tests/unit/seekai-provider.test.ts diff --git a/AGENTS.md b/AGENTS.md index d0ac2952b2..1e90b6ceed 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -46,7 +46,7 @@ Repository map and Reference Documentation sections below. ## Project at a Glance -**OmniRoute** — unified AI proxy/router. One endpoint, 355 LLM providers, auto-fallback. +**OmniRoute** — unified AI proxy/router. One endpoint, 356 LLM providers, auto-fallback. | Layer | Location | Purpose | | ------------- | ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | diff --git a/README.md b/README.md index 055f71a88b..119a590cb3 100644 --- a/README.md +++ b/README.md @@ -7,7 +7,7 @@ # 🚀 OmniRoute — The Free AI Gateway -OmniRoute — Never stop coding. Every AI tool → 355 providers — 150+ free — through one endpoint. Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini with auto-fallback. RTK + Caveman stacked compression saves 15–95% tokens (~89% avg) — never hit limits. 355 AI providers · 150+ free tiers · ~1.51B free tokens/mo · 19 routing strategies · $0 to start. +OmniRoute — Never stop coding. Every AI tool → 356 providers — 150+ free — through one endpoint. Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini with auto-fallback. RTK + Caveman stacked compression saves 15–95% tokens (~89% avg) — never hit limits. 356 AI providers · 150+ free tiers · ~1.51B free tokens/mo · 19 routing strategies · $0 to start.
@@ -209,7 +209,7 @@ curl http://localhost:20128/v1/chat/completions \
-The Promise — One endpoint and 355 providers. Automatic fallback keeps routing while another healthy target is available. Six pillars: resilient fallback across 355 providers · up to 95% token savings on eligible workloads · $0 to start with 150+ free tiers and 53 recurring/keyless free-forever providers · 36 CLI/agent integrations through one config · OpenAI, Claude, Gemini and Responses API compatibility at /v1 · production controls including circuit breakers, TLS stealth, MCP 110 tools, A2A, memory, guardrails, evals and 39,000+ static test declarations across 5,100+ tracked test files. +The Promise — One endpoint and 356 providers. Automatic fallback keeps routing while another healthy target is available. Six pillars: resilient fallback across 356 providers · up to 95% token savings on eligible workloads · $0 to start with 150+ free tiers and 53 recurring/keyless free-forever providers · 36 CLI/agent integrations through one config · OpenAI, Claude, Gemini and Responses API compatibility at /v1 · production controls including circuit breakers, TLS stealth, MCP 110 tools, A2A, memory, guardrails, evals and 39,000+ static test declarations across 5,100+ tracked test files.

@@ -462,7 +462,7 @@ All **19** strategies — mix & match per combo step:
-What sets OmniRoute apart — a dated feature snapshot vs 9router, OpenRouter, CLIProxyAPI and LiteLLM across 13 capabilities. OmniRoute: 355 providers, 150+ free tiers built in, 19 routing strategies, 12-engine token compression, built-in MCP server with 110 tools, A2A agent protocol, persistent memory, guardrails, cloud agents, TLS fingerprint stealth, Desktop/Termux/PWA and 42 i18n UI locales. OmniRoute is MIT-licensed and self-hostable. Competitor capabilities and counts may change; see the linked methodology. +What sets OmniRoute apart — a dated feature snapshot vs 9router, OpenRouter, CLIProxyAPI and LiteLLM across 13 capabilities. OmniRoute: 356 providers, 150+ free tiers built in, 19 routing strategies, 12-engine token compression, built-in MCP server with 110 tools, A2A agent protocol, persistent memory, guardrails, cloud agents, TLS fingerprint stealth, Desktop/Termux/PWA and 42 i18n UI locales. OmniRoute is MIT-licensed and self-hostable. Competitor capabilities and counts may change; see the linked methodology. 📊 Full methodology & per-feature detail vs 9router, OpenRouter, CLIProxyAPI & LiteLLM → [`docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md`](docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md) diff --git a/changelog.d/features/11786-seekai-provider.md b/changelog.d/features/11786-seekai-provider.md new file mode 100644 index 0000000000..ad4e2dbd44 --- /dev/null +++ b/changelog.d/features/11786-seekai-provider.md @@ -0,0 +1 @@ +- **feat(providers):** add SeekAi (`seekai.cc`) as an OpenAI-compatible New-API gateway — catalog id `seekai` (alias `ska`), `https://seekai.cc/v1`, live `/v1/models` via `passthroughModels`, aggregator-list membership so New-API balance detection can opt in. No referral/aff codes. ([#11786](https://github.com/diegosouzapw/OmniRoute/issues/11786)) diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index ffed35fbc1..e5d0287a65 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -1,4 +1,5 @@ { + "_rebaseline_2026_09_02_11786_seekai_provider": "PR #11786 (feat/11786-seekai-provider, closes #11786) own growth: src/shared/constants/providers/apikey/gateways.ts 1438->1458 (check-file-size split-newline=1459; the seekai APIKEY_PROVIDERS_GATEWAYS catalog entry plus authHint, additive data at the existing registry chokepoint, same god-file no-split rationale as prior gateways.ts rebaselines: #10987 logfare, #10531 freebuff). Covered by tests/unit/seekai-provider.test.ts.", "_rebaseline_2026_09_02_12325_generic_429_invalidate": "PR #12325 own growth: open-sse/handlers/chatCore.ts 5946->5955 (+9 = the non-Codex 429 else-if that drops the generic quota wrapper and stamps force-refresh, plus a source-regex breadcrumb). Irreducible call-site wiring next to the existing Codex 429 invalidateCodexQuotaCache branch; not extractable without splitting handleChatCore mid-response. Covered by tests/unit/generic-quota-fetcher.test.ts (31/31) and tests/unit/antigravity-429-quota-cooldown.test.ts.", "_rebaseline_2026_09_02_12429_wreq_migration_suite": "PR #12429 (wreq-js web-cookie transport): new test file tests/unit/tls-client-wreq-migration.test.ts at 1374 lines, above the 1200 new-file testCap. Frozen rather than split: it is the single cohesive regression suite for the transport migration (31 cases covering streaming, fragmented EOF sentinels, proxy isolation, first-byte and hard deadlines, binary responses and cancellation), and the cases share the native-transport harness the file sets up once. Splitting it during a merge would duplicate that harness across files for no coverage gain. Entered at the exact LOC, so it can only ratchet down from here.", "_rebaseline_2026_09_02_12239_chatgpt_web_cleanroom": "PR #12239 (backryun, codex/restore-chatgpt-web-cleanroom) own growth at the two existing chat chokepoints for the clean-room ChatGPT Web transport: src/sse/handlers/chat.ts 2384->2424 (+40); open-sse/handlers/chatCore.ts 5946->5976 (+30). Additive dispatch wiring; the retirement guard is narrowed to the GPL-derived cgpt-web alias rather than removed, so #11754's provenance decision still holds for the old implementation. Same own-growth rationale as _rebaseline_2026_08_20_10531_freebuff_provider.", @@ -450,7 +451,7 @@ "src/lib/tailscaleTunnel.ts": 1208, "src/lib/tokenHealthCheck.ts": 1218, "src/shared/components/RequestLoggerV2.tsx": 1718, - "src/shared/constants/providers/apikey/gateways.ts": 1439, + "src/shared/constants/providers/apikey/gateways.ts": 1459, "src/shared/services/cliRuntime.ts": 1296, "src/sse/handlers/chat.ts": 2424, "src/sse/services/auth.ts": 3427, diff --git a/docs/diagrams/cli-terminal.svg b/docs/diagrams/cli-terminal.svg index 139a868898..a16f88351f 100644 --- a/docs/diagrams/cli-terminal.svg +++ b/docs/diagrams/cli-terminal.svg @@ -1,4 +1,4 @@ - + Compact animated terminal cycling three real OmniRoute CLI commands with a typewriter effect and a scrolling subcommand ticker; the first frame shows the completed providers-list screen. diff --git a/docs/diagrams/comparison-table.svg b/docs/diagrams/comparison-table.svg index 3bc3895f25..92fe11718c 100644 --- a/docs/diagrams/comparison-table.svg +++ b/docs/diagrams/comparison-table.svg @@ -1,4 +1,4 @@ - + Static-header comparison table where each capability row fades in top to bottom; the OmniRoute column is highlighted and shows a check or a leading value in every row, while competitors show a mix of checks, partials and crosses. diff --git a/docs/diagrams/promise-pillars.svg b/docs/diagrams/promise-pillars.svg index f32198f62f..41bcdf397c 100644 --- a/docs/diagrams/promise-pillars.svg +++ b/docs/diagrams/promise-pillars.svg @@ -1,4 +1,4 @@ - + Animated promise card: six pillar tiles fade in in reading order, then a soft colored border highlight sweeps from tile to tile in a continuous cycle. @@ -21,7 +21,7 @@ - One endpoint. 355 providers. Never stop building — OmniRoute picks the cheapest one that works. + One endpoint. 356 providers. Never stop building — OmniRoute picks the cheapest one that works. @@ -38,7 +38,7 @@ Never hit limits - Auto-fallback across 355 providers in + Auto-fallback across 356 providers in milliseconds. Quota out? The next provider takes over while a healthy target remains. diff --git a/docs/diagrams/readme-hero.svg b/docs/diagrams/readme-hero.svg index b758959878..2fc0a31918 100644 --- a/docs/diagrams/readme-hero.svg +++ b/docs/diagrams/readme-hero.svg @@ -1,4 +1,4 @@ - + Animated hero card: a pulse travels the divider line and a compression bar demo repeatedly shrinks a prompt by up to 95 percent; all headline content is static and readable on the first frame. @@ -28,7 +28,7 @@ Never stop coding. - Every AI tool → 355 providers150+ free — through one endpoint. + Every AI tool → 356 providers150+ free — through one endpoint. Claude Code · Codex · Cursor · Cline · Copilot · Antigravity  →  FREE Claude / GPT / Gemini · auto-fallback diff --git a/docs/i18n/ar/llm.txt b/docs/i18n/ar/llm.txt index abd2becf64..603773977a 100644 --- a/docs/i18n/ar/llm.txt +++ b/docs/i18n/ar/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/az/llm.txt b/docs/i18n/az/llm.txt index 83aae20b40..3af7361352 100644 --- a/docs/i18n/az/llm.txt +++ b/docs/i18n/az/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/bg/llm.txt b/docs/i18n/bg/llm.txt index 19b6eaa3bc..c510c85d7e 100644 --- a/docs/i18n/bg/llm.txt +++ b/docs/i18n/bg/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/bn/llm.txt b/docs/i18n/bn/llm.txt index ded0a119f8..54b9ca1d9a 100644 --- a/docs/i18n/bn/llm.txt +++ b/docs/i18n/bn/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/cs/llm.txt b/docs/i18n/cs/llm.txt index d7d495ce79..4dd9012d51 100644 --- a/docs/i18n/cs/llm.txt +++ b/docs/i18n/cs/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/da/llm.txt b/docs/i18n/da/llm.txt index fc0f3956f9..d29af5f81c 100644 --- a/docs/i18n/da/llm.txt +++ b/docs/i18n/da/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/de/llm.txt b/docs/i18n/de/llm.txt index 0a3d6f41df..0002d85510 100644 --- a/docs/i18n/de/llm.txt +++ b/docs/i18n/de/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/es/llm.txt b/docs/i18n/es/llm.txt index a14c3364ca..c423fdd245 100644 --- a/docs/i18n/es/llm.txt +++ b/docs/i18n/es/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/fa/llm.txt b/docs/i18n/fa/llm.txt index 2637cb93e6..d8b83418ea 100644 --- a/docs/i18n/fa/llm.txt +++ b/docs/i18n/fa/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/fi/llm.txt b/docs/i18n/fi/llm.txt index dcdbce4e9b..b91777ac80 100644 --- a/docs/i18n/fi/llm.txt +++ b/docs/i18n/fi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/fr/llm.txt b/docs/i18n/fr/llm.txt index 4ec20698a0..7e396ec1e9 100644 --- a/docs/i18n/fr/llm.txt +++ b/docs/i18n/fr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/gu/llm.txt b/docs/i18n/gu/llm.txt index bb96a9ca15..75561762bf 100644 --- a/docs/i18n/gu/llm.txt +++ b/docs/i18n/gu/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/he/llm.txt b/docs/i18n/he/llm.txt index 7b05f7da36..e8dab2b860 100644 --- a/docs/i18n/he/llm.txt +++ b/docs/i18n/he/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/hi/llm.txt b/docs/i18n/hi/llm.txt index d5abf815fd..3df2c700b6 100644 --- a/docs/i18n/hi/llm.txt +++ b/docs/i18n/hi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/hu/llm.txt b/docs/i18n/hu/llm.txt index 242f219733..bd0667630c 100644 --- a/docs/i18n/hu/llm.txt +++ b/docs/i18n/hu/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/id/llm.txt b/docs/i18n/id/llm.txt index 6ea1e4a22c..0cce4bc272 100644 --- a/docs/i18n/id/llm.txt +++ b/docs/i18n/id/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/it/llm.txt b/docs/i18n/it/llm.txt index d3d4caf7c0..0763f0b728 100644 --- a/docs/i18n/it/llm.txt +++ b/docs/i18n/it/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/ja/llm.txt b/docs/i18n/ja/llm.txt index fbd4266e07..3dbb124940 100644 --- a/docs/i18n/ja/llm.txt +++ b/docs/i18n/ja/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/ko/llm.txt b/docs/i18n/ko/llm.txt index d72d39d120..0d44a5b222 100644 --- a/docs/i18n/ko/llm.txt +++ b/docs/i18n/ko/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/mr/llm.txt b/docs/i18n/mr/llm.txt index d3d06a534b..9410d228f5 100644 --- a/docs/i18n/mr/llm.txt +++ b/docs/i18n/mr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/ms/llm.txt b/docs/i18n/ms/llm.txt index fb92e9f704..e610a66e29 100644 --- a/docs/i18n/ms/llm.txt +++ b/docs/i18n/ms/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/nl/llm.txt b/docs/i18n/nl/llm.txt index 108465e6ee..a4364f5ac2 100644 --- a/docs/i18n/nl/llm.txt +++ b/docs/i18n/nl/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/no/llm.txt b/docs/i18n/no/llm.txt index 92ef34cc68..7daa5caa07 100644 --- a/docs/i18n/no/llm.txt +++ b/docs/i18n/no/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/phi/llm.txt b/docs/i18n/phi/llm.txt index ed4c534f3b..208e4a1398 100644 --- a/docs/i18n/phi/llm.txt +++ b/docs/i18n/phi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/pl/llm.txt b/docs/i18n/pl/llm.txt index 9c2d022df6..79f3f32583 100644 --- a/docs/i18n/pl/llm.txt +++ b/docs/i18n/pl/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/pt-BR/llm.txt b/docs/i18n/pt-BR/llm.txt index 1d6c34ce26..e795080292 100644 --- a/docs/i18n/pt-BR/llm.txt +++ b/docs/i18n/pt-BR/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/pt/llm.txt b/docs/i18n/pt/llm.txt index 5be880273e..ffa62b14e6 100644 --- a/docs/i18n/pt/llm.txt +++ b/docs/i18n/pt/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/ro/llm.txt b/docs/i18n/ro/llm.txt index a9006002d2..7a6f5067a9 100644 --- a/docs/i18n/ro/llm.txt +++ b/docs/i18n/ro/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/ru/llm.txt b/docs/i18n/ru/llm.txt index fb3bd9901f..ed8415886d 100644 --- a/docs/i18n/ru/llm.txt +++ b/docs/i18n/ru/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/sk/llm.txt b/docs/i18n/sk/llm.txt index 30a439beb7..11385eade3 100644 --- a/docs/i18n/sk/llm.txt +++ b/docs/i18n/sk/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/sv/llm.txt b/docs/i18n/sv/llm.txt index a3e29a0308..5a0f5b5c80 100644 --- a/docs/i18n/sv/llm.txt +++ b/docs/i18n/sv/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/sw/llm.txt b/docs/i18n/sw/llm.txt index 3e50fc5415..ef792bc7e7 100644 --- a/docs/i18n/sw/llm.txt +++ b/docs/i18n/sw/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/ta/llm.txt b/docs/i18n/ta/llm.txt index 26cc42048a..2c0694b490 100644 --- a/docs/i18n/ta/llm.txt +++ b/docs/i18n/ta/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/te/llm.txt b/docs/i18n/te/llm.txt index 95b13b644a..29ff10874a 100644 --- a/docs/i18n/te/llm.txt +++ b/docs/i18n/te/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/th/llm.txt b/docs/i18n/th/llm.txt index 0c144efb4f..4096423b6d 100644 --- a/docs/i18n/th/llm.txt +++ b/docs/i18n/th/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/tr/llm.txt b/docs/i18n/tr/llm.txt index 9c5ce3e607..b83be6d908 100644 --- a/docs/i18n/tr/llm.txt +++ b/docs/i18n/tr/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/uk-UA/llm.txt b/docs/i18n/uk-UA/llm.txt index 407ee0fdfa..f3cf580495 100644 --- a/docs/i18n/uk-UA/llm.txt +++ b/docs/i18n/uk-UA/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/ur/llm.txt b/docs/i18n/ur/llm.txt index b80c1a22cc..9efbef2f56 100644 --- a/docs/i18n/ur/llm.txt +++ b/docs/i18n/ur/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/vi/llm.txt b/docs/i18n/vi/llm.txt index 8aaea70f4e..eef12beee4 100644 --- a/docs/i18n/vi/llm.txt +++ b/docs/i18n/vi/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/zh-CN/llm.txt b/docs/i18n/zh-CN/llm.txt index 2a9812d5f2..f1c4a2b574 100644 --- a/docs/i18n/zh-CN/llm.txt +++ b/docs/i18n/zh-CN/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/i18n/zh-TW/llm.txt b/docs/i18n/zh-TW/llm.txt index 2caf3c753c..b7b1342bf5 100644 --- a/docs/i18n/zh-TW/llm.txt +++ b/docs/i18n/zh-TW/llm.txt @@ -4,7 +4,7 @@ --- -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -281,7 +281,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/docs/reference/PROVIDER_REFERENCE.md b/docs/reference/PROVIDER_REFERENCE.md index 67046599ad..c7ba75f601 100644 --- a/docs/reference/PROVIDER_REFERENCE.md +++ b/docs/reference/PROVIDER_REFERENCE.md @@ -1,16 +1,16 @@ --- title: "Provider Reference" version: 3.8.51 -lastUpdated: 2026-09-02 +lastUpdated: 2026-09-03 --- # Provider Reference > **Auto-generated** from `src/shared/constants/providers.ts` — do not edit by hand. > Regenerate with: `npm run gen:provider-reference` -> **Last generated:** 2026-09-02 +> **Last generated:** 2026-09-03 -Total providers: **355**. See category breakdown below. +Total providers: **356**. See category breakdown below. ## Categories @@ -118,7 +118,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `zai-web` | `zw` | Z.ai Web | Web cookie | [link](https://chat.z.ai) | Copy the "token" value from chat.z.ai → DevTools → Application → Local Storage. Do not copy cookies; OmniRoute handles the per-request CAPTCHA through its browser transport. | — | | `zenmux-free` | `zmf` | ZenMux Free (Web) | Web cookie | [link](https://zenmux.ai) | Login at zenmux.ai, then export all cookies using EditThisCookie or Cookie-Editor and paste the full Cookie header string here. Refresh every ~30 days. | — | -## API Key Providers (paid / paid-with-free-credits) (237) +## API Key Providers (paid / paid-with-free-credits) (238) | ID | Alias | Name | Tags | Website | Notes | |----|-------|------|------|---------|-------| @@ -310,6 +310,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `sarvam` | `sarvam` | Sarvam AI | API key | [link](https://docs.sarvam.ai) | ₹1,000 in free signup credits — never expire | | `scaleway` | `scw` | Scaleway AI | API key | [link](https://www.scaleway.com/en/docs/ai-data/generative-apis/) | 1M free tokens for new accounts — EU/GDPR compliant (Paris), Qwen3 235B & Llama 70B | | `sealion` | `sealion` | SEA-LION | API key | [link](https://sea-lion.ai) | Sign in at sea-lion.ai with Google (no card, no region wall), create an API key, then paste it here. | +| `seekai` | `ska` | SeekAi | API key, aggregator | [link](https://seekai.cc) | Create an API key at https://seekai.cc, then paste it here as a Bearer token. | | `segmind` | `segmind` | Segmind | API key, image, video | [link](https://segmind.com) | Use your Segmind API key in the x-api-key header. OmniRoute targets https://api.segmind.com/v1/ and returns the generated image/video bytes directly. | | `sensenova` | `sensenova` | SenseNova | API key | [link](https://platform.sensenova.cn) | Get API key at platform.sensenova.cn | | `siliconflow` | `siliconflow` | SiliconFlow | API key | [link](https://cloud.siliconflow.com) | $1 free credits plus currently listed $0 models after identity verification; availability and limits may change | diff --git a/llm.txt b/llm.txt index 9789ad5c22..13d3c28e78 100644 --- a/llm.txt +++ b/llm.txt @@ -1,6 +1,6 @@ # OmniRoute -> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 355 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. +> OmniRoute is a free, open-source AI Gateway that acts as a universal API proxy for multi-provider LLMs. It provides smart routing, automatic fallback, load balancing, and format translation across 356 AI providers — all through a single OpenAI-compatible endpoint. Includes a built-in MCP Server (110 tools), A2A v0.3 protocol, Memory/Skills systems, Cloud Agents (codex, cursor, devin, jules), Guardrails framework, and an Electron desktop app. ## Overview @@ -277,7 +277,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo ## Key Features (v3.8.50) ### Core Proxy -- **355 AI providers** with automatic format translation +- **356 AI providers** with automatic format translation - **Provider categories**: Free (90+ free tiers), OAuth, API Key, Self-Hosted, Custom (OpenAI/Anthropic-compatible) - **19 routing strategies**: priority, weighted, round-robin, fill-first, p2c, random, least-used, cost-optimized, reset-aware, reset-window, headroom, strict-random, auto, lkgp, context-optimized, cache-optimized, context-relay, fusion, pipeline - **4-tier fallback**: Subscription → API Key → Cheap → Free diff --git a/open-sse/config/providers/index.ts b/open-sse/config/providers/index.ts index 1e6b542d13..e6ee28c65f 100644 --- a/open-sse/config/providers/index.ts +++ b/open-sse/config/providers/index.ts @@ -270,6 +270,7 @@ import { voidAiProvider } from "./registry/void-ai/index.ts"; import { helixmindProvider } from "./registry/helixmind/index.ts"; import { tabitokenProvider } from "./registry/tabitoken/index.ts"; import { logfareProvider } from "./registry/logfare/index.ts"; +import { seekaiProvider } from "./registry/seekai/index.ts"; export const REGISTRY: Record = { aimlapi: aimlapiProvider, @@ -544,4 +545,5 @@ export const REGISTRY: Record = { helixmind: helixmindProvider, tabitoken: tabitokenProvider, logfare: logfareProvider, + seekai: seekaiProvider, }; diff --git a/open-sse/config/providers/registry/seekai/index.ts b/open-sse/config/providers/registry/seekai/index.ts new file mode 100644 index 0000000000..e37550cc52 --- /dev/null +++ b/open-sse/config/providers/registry/seekai/index.ts @@ -0,0 +1,18 @@ +import { buildOpenAiCompatibleRegistryEntry } from "../../shared.ts"; + +/** + * SeekAi (https://seekai.cc) — QuantumNous New-API gateway. + * Live-verified 2026-09-02: GET /api/status → system_name=SeekAi, + * version=v1.0.0-rc.25, quota_display_type=USD. GET /v1/models is + * API-key gated (401 Invalid token without a key). Catalog is dynamic; + * no static seed. Referral/aff query params stay out of this entry + * (no-hardcoded-referral-codes). + */ +export const seekaiProvider = buildOpenAiCompatibleRegistryEntry({ + id: "seekai", + alias: "ska", + baseUrl: "https://seekai.cc/v1/chat/completions", + modelsUrl: "https://seekai.cc/v1/models", + models: [], + passthroughModels: true, +}); diff --git a/package.json b/package.json index bc63455358..4925d88658 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "omniroute", "version": "3.8.51", - "description": "Unified AI router with 355 providers, RTK+Caveman compression, auto fallback, MCP/A2A, desktop, PWA, and OpenAI-compatible APIs.", + "description": "Unified AI router with 356 providers, RTK+Caveman compression, auto fallback, MCP/A2A, desktop, PWA, and OpenAI-compatible APIs.", "type": "module", "bin": { "omniroute": "bin/omniroute.mjs", diff --git a/public/images/tier-flow-dark.svg b/public/images/tier-flow-dark.svg index 1cf2589812..8f7fcde48f 100644 --- a/public/images/tier-flow-dark.svg +++ b/public/images/tier-flow-dark.svg @@ -1,6 +1,6 @@ - + OmniRoute 4-tier fallback - OmniRoute 4-tier fallback: your IDE or CLI calls one local endpoint and the OmniRoute Smart Router fails over across 355 providers in 4 tiers — Tier 1 Subscription, Tier 2 API, Tier 3 Cheap, Tier 4 Free. + OmniRoute 4-tier fallback: your IDE or CLI calls one local endpoint and the OmniRoute Smart Router fails over across 356 providers in 4 tiers — Tier 1 Subscription, Tier 2 API, Tier 3 Cheap, Tier 4 Free. @@ -15,7 +15,7 @@ OmniRoute 4-tier fallback - Never stop building — automatic zero-config failover across 355 providers + Never stop building — automatic zero-config failover across 356 providers diff --git a/public/images/tier-flow-light.svg b/public/images/tier-flow-light.svg index cd79d47e3b..5ad3a108f7 100644 --- a/public/images/tier-flow-light.svg +++ b/public/images/tier-flow-light.svg @@ -1,6 +1,6 @@ - + OmniRoute 4-tier fallback - OmniRoute 4-tier fallback: your IDE or CLI calls one local endpoint and the OmniRoute Smart Router fails over across 355 providers in 4 tiers — Tier 1 Subscription, Tier 2 API, Tier 3 Cheap, Tier 4 Free. + OmniRoute 4-tier fallback: your IDE or CLI calls one local endpoint and the OmniRoute Smart Router fails over across 356 providers in 4 tiers — Tier 1 Subscription, Tier 2 API, Tier 3 Cheap, Tier 4 Free. @@ -15,7 +15,7 @@ OmniRoute 4-tier fallback - Never stop building — automatic zero-config failover across 355 providers + Never stop building — automatic zero-config failover across 356 providers diff --git a/src/i18n/messages/ar.json b/src/i18n/messages/ar.json index b987620b46..a2d2c4acf9 100644 --- a/src/i18n/messages/ar.json +++ b/src/i18n/messages/ar.json @@ -6239,6 +6239,7 @@ "requesty": "أنشئ مفتاح API على https://app.requesty.ai، ثم الصقه هنا كرمز Bearer. نقطة نهاية متوافقة مع OpenAI على https://router.requesty.ai/v1، مع كتالوج /v1/models مباشر.", "runwayml": "يعتمد توليد الفيديو في Runway على المهام. يرسل OmniRoute وظائف تحويل النص إلى فيديو أو الصورة إلى فيديو، ويستعلم من /v1/tasks/[id]، ويقوم بتطبيع مخرجات الفيديو النهائية مرة أخرى إلى استجابة /v1/videos/generations الشبيهة بـ OpenAI.", "sambanova": "رصيد مجاني بقيمة 5$ عند التسجيل (صلاحية 30 يومًا)، لا يتطلب بطاقة ائتمان", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "يستخدم اكتشاف النماذج /v2/lm/scenarios/foundation-models/models على AI_API_URL. تستخدم طلبات الدردشة deploymentUrl/chat/completions وتتطلب AI-Resource-Group.", "sarvam": "سارفام AI متوافق مع OpenAI على /v1. يقوم OmniRoute بفحص /v1/models ويوجه حركة الدردشة إلى /v1/chat/completions. تم ضبط النماذج للغات الهندية.", "scaleway": "1 مليون رمز مميز مجاني للحسابات الجديدة — متوافق مع الاتحاد الأوروبي/GDPR (باريس)، Qwen3 235B وLlama 70B", diff --git a/src/i18n/messages/az.json b/src/i18n/messages/az.json index 33217b0d2b..131a9b212b 100644 --- a/src/i18n/messages/az.json +++ b/src/i18n/messages/az.json @@ -6239,6 +6239,7 @@ "requesty": "https://app.requesty.ai ünvanında API açarı yaradın, sonra onu bura Bearer tokeni kimi yapışdırın. OpenAI ilə uyğun son nöqtə canlı /v1/models kataloqu ilə https://router.requesty.ai/v1 ünvanındadır.", "runwayml": "Runway video yaradılması tapşırıq əsaslıdır. OmniRoute mətndən-videoya və ya şəkildən-videoya tapşırıqlarını təqdim edir, /v1/tasks/[id] ünvanını sorğulayır və tamamlanmış video çıxışlarını yenidən OpenAI tipli /v1/videos/generations cavabına normallaşdırır.", "sambanova": "Qeydiyyatdan keçdikdə $5 pulsuz kredit (30 gün etibarlılıq müddəti), kredit kartı tələb olunmur", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Model kəşfi AI_API_URL üzərində /v2/lm/scenarios/foundation-models/models istifadə edir. Söhbət sorğuları deploymentUrl/chat/completions istifadə edir və AI-Resource-Group tələb edir.", "sarvam": "Sarvam AI OpenAI ilə uyğun gəlir /v1. OmniRoute /v1/models-i yoxlayır və söhbət trafikini /v1/chat/completions-a yönləndirir. Modellər Hind dilləri üçün tənzimlənmişdir.", "scaleway": "Yeni hesablar üçün 1M pulsuz token — Aİ/GDPR uyğun (Paris), Qwen3 235B və Llama 70B", diff --git a/src/i18n/messages/bg.json b/src/i18n/messages/bg.json index 3c3c942e6d..f51cea18c1 100644 --- a/src/i18n/messages/bg.json +++ b/src/i18n/messages/bg.json @@ -6239,6 +6239,7 @@ "requesty": "Създайте API ключ на https://app.requesty.ai, след което го поставете тук като Bearer токен. Съвместима с OpenAI крайна точка на https://router.requesty.ai/v1 с каталог на живо за /v1/models.", "runwayml": "Генерирането на видео в Runway е базирано на задачи. OmniRoute изпраща задачи за text-to-video или image-to-video, проверява периодично /v1/tasks/[id] и нормализира готовите видео резултати обратно в наподобяващ OpenAI отговор на /v1/videos/generations.", "sambanova": "$5 безплатни кредити при регистрация (валидност 30 дни), не се изисква кредитна карта", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Откриването на модели използва /v2/lm/scenarios/foundation-models/models на AI_API_URL. Заявките за чат използват deploymentUrl/chat/completions и изискват AI-Resource-Group.", "sarvam": "Sarvam AI е съвместим с OpenAI на /v1. OmniRoute проучва /v1/models и маршрутизира чат трафика към /v1/chat/completions. Моделите са настроени за индийски езици.", "scaleway": "1M безплатни токена за нови акаунти — съвместимо с EU/GDPR (Париж), Qwen3 235B и Llama 70B", diff --git a/src/i18n/messages/bn.json b/src/i18n/messages/bn.json index 36cee7c2d9..4dacb5dd56 100644 --- a/src/i18n/messages/bn.json +++ b/src/i18n/messages/bn.json @@ -6239,6 +6239,7 @@ "requesty": "https://app.requesty.ai-এ একটি API কী তৈরি করুন, তারপর এটি এখানে Bearer টোকেন হিসেবে পেস্ট করুন। https://router.requesty.ai/v1-এ OpenAI-সামঞ্জস্যপূর্ণ এন্ডপয়েন্ট, সাথে একটি লাইভ /v1/models ক্যাটালগ রয়েছে।", "runwayml": "Runway ভিডিও জেনারেশন টাস্ক-ভিত্তিক। OmniRoute টেক্সট-টু-ভিডিও বা ইমেজ-টু-ভিডিও জব সাবমিট করে, /v1/tasks/[id] পোল করে এবং সমাপ্ত ভিডিও আউটপুটগুলোকে আবার OpenAI-এর মতো /v1/videos/generations রেসপন্সে নরমালাইজ করে।", "sambanova": "সাইন আপ করার সময় $5 ফ্রি ক্রেডিট (৩০ দিনের মেয়াদ), কোনো ক্রেডিট কার্ডের প্রয়োজন নেই", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "মডেল ডিসকভারি AI_API_URL-এ /v2/lm/scenarios/foundation-models/models ব্যবহার করে। চ্যাট রিকোয়েস্টগুলো deploymentUrl/chat/completions ব্যবহার করে এবং এর জন্য AI-Resource-Group প্রয়োজন।", "sarvam": "Sarvam AI OpenAI-সঙ্গত /v1-এ। OmniRoute /v1/models-এ প্রোব করে এবং চ্যাট ট্রাফিককে /v1/chat/completions-এ রাউট করে। মডেলগুলি ইন্ডিক ভাষার জন্য টিউন করা হয়েছে।", "scaleway": "নতুন অ্যাকাউন্টের জন্য 1M ফ্রি টোকেন — EU/GDPR কমপ্লায়েন্ট (প্যারিস), Qwen3 235B এবং Llama 70B", diff --git a/src/i18n/messages/cs.json b/src/i18n/messages/cs.json index 568a126332..2450daa9ff 100644 --- a/src/i18n/messages/cs.json +++ b/src/i18n/messages/cs.json @@ -6239,6 +6239,7 @@ "requesty": "Vytvořte API klíč na adrese https://app.requesty.ai a poté jej vložte sem jako Bearer token. Koncový bod kompatibilní s OpenAI je na adrese https://router.requesty.ai/v1, s živým katalogem /v1/models.", "runwayml": "Generování videa v Runway je založeno na úlohách. OmniRoute odesílá úlohy typu text-na-video nebo obrázek-na-video, dotazuje se na /v1/tasks/[id] a normalizuje hotové video výstupy zpět do odpovědi typu /v1/videos/generations podobné OpenAI.", "sambanova": "Bezplatný kredit 5 $ při registraci (platnost 30 dní), není vyžadována platební karta", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Vyhledávání modelů používá /v2/lm/scenarios/foundation-models/models na AI_API_URL. Požadavky na chat používají deploymentUrl/chat/completions a vyžadují AI-Resource-Group.", "sarvam": "Sarvam AI je kompatibilní s OpenAI na /v1. OmniRoute prozkoumává /v1/models a směruje chatový provoz na /v1/chat/completions. Modely jsou laděny pro indické jazyky.", "scaleway": "1 milion bezplatných tokenů pro nové účty — v souladu s EU/GDPR (Paříž), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/da.json b/src/i18n/messages/da.json index 2f19b8d75a..4dd4699f43 100644 --- a/src/i18n/messages/da.json +++ b/src/i18n/messages/da.json @@ -6239,6 +6239,7 @@ "requesty": "Opret en API-nøgle på https://app.requesty.ai, og indsæt den derefter her som en Bearer-token. OpenAI-kompatibelt slutpunkt på https://router.requesty.ai/v1 med et live /v1/models-katalog.", "runwayml": "Runway-videogenerering er opgavebaseret. OmniRoute indsender tekst-til-video- eller billede-til-video-job, poller /v1/tasks/[id] og normaliserer de færdige videooutput tilbage til det OpenAI-lignende /v1/videos/generations-svar.", "sambanova": "$5 i gratis kredit ved tilmelding (30 dages gyldighed), intet kreditkort påkrævet", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Modelfindelse bruger /v2/lm/scenarios/foundation-models/models på AI_API_URL. Chatanmodninger bruger deploymentUrl/chat/completions og kræver AI-Resource-Group.", "sarvam": "Sarvam AI er OpenAI-kompatibel på /v1. OmniRoute undersøger /v1/models og dirigerer chattrafik til /v1/chat/completions. Modellerne er tilpasset til indiske sprog.", "scaleway": "1M gratis tokens til nye konti — EU/GDPR-kompatibel (Paris), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/de.json b/src/i18n/messages/de.json index b6c9eccefc..0fa6991704 100644 --- a/src/i18n/messages/de.json +++ b/src/i18n/messages/de.json @@ -6239,6 +6239,7 @@ "requesty": "Erstellen Sie einen API-Schlüssel unter https://app.requesty.ai und fügen Sie ihn hier als Bearer-Token ein. OpenAI-kompatibler Endpunkt unter https://router.requesty.ai/v1 mit einem Live-Katalog unter /v1/models.", "runwayml": "Die Runway-Videogenerierung ist aufgabenbasiert. OmniRoute übermittelt Text-to-Video- oder Image-to-Video-Jobs, fragt /v1/tasks/[id] ab und normalisiert die fertigen Videoausgaben zurück in die OpenAI-ähnliche Antwort von /v1/videos/generations.", "sambanova": "5 $ kostenloses Guthaben bei Registrierung (30 Tage Gültigkeit), keine Kreditkarte erforderlich", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Die Modellerkennung verwendet /v2/lm/scenarios/foundation-models/models auf AI_API_URL. Chat-Anfragen verwenden deploymentUrl/chat/completions und erfordern AI-Resource-Group.", "sarvam": "Sarvam AI ist OpenAI-kompatibel unter /v1. OmniRoute durchsucht /v1/models und leitet den Chat-Verkehr an /v1/chat/completions weiter. Die Modelle sind auf indische Sprachen abgestimmt.", "scaleway": "1 Mio. kostenlose Token für neue Konten — EU-DSGVO-konform (Paris), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index ff70e5c7bb..e1b0adcb11 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -6242,6 +6242,7 @@ "requesty": "Create an API key at https://app.requesty.ai, then paste it here as a Bearer token. OpenAI-compatible endpoint at https://router.requesty.ai/v1, with a live /v1/models catalog.", "runwayml": "Runway video generation is task-based. OmniRoute submits text-to-video or image-to-video jobs, polls /v1/tasks/[id], and normalizes the finished video outputs back into the OpenAI-like /v1/videos/generations response.", "sambanova": "$5 free credits on signup (30-day validity), no credit card required", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Model discovery uses /v2/lm/scenarios/foundation-models/models on AI_API_URL. Chat requests use deploymentUrl/chat/completions and require AI-Resource-Group.", "sarvam": "Sarvam AI is OpenAI-compatible on /v1. OmniRoute probes /v1/models and routes chat traffic to /v1/chat/completions. Models are tuned for Indic languages.", "scaleway": "1M free tokens for new accounts — EU/GDPR compliant (Paris), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/es.json b/src/i18n/messages/es.json index 89f8137d05..99f66790a4 100644 --- a/src/i18n/messages/es.json +++ b/src/i18n/messages/es.json @@ -6239,6 +6239,7 @@ "requesty": "Create an API key at https://app.requesty.ai, then paste it here as a Bearer token. OpenAI-compatible endpoint at https://router.requesty.ai/v1, with a live /v1/models catalog.", "runwayml": "Runway video generation is task-based. OmniRoute submits text-to-video or image-to-video jobs, polls /v1/tasks/[id], and normalizes the finished video outputs back into the OpenAI-like /v1/videos/generations response.", "sambanova": "$5 free credits on signup (30-day validity), no credit card required", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Model discovery uses /v2/lm/scenarios/foundation-models/models on AI_API_URL. Chat requests use deploymentUrl/chat/completions and require AI-Resource-Group.", "sarvam": "Sarvam AI is OpenAI-compatible on /v1. OmniRoute probes /v1/models and routes chat traffic to /v1/chat/completions. Models are tuned for Indic languages.", "scaleway": "1M free tokens for new accounts — EU/GDPR compliant (Paris), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/fa.json b/src/i18n/messages/fa.json index fb8fab5f8b..176e89c4d2 100644 --- a/src/i18n/messages/fa.json +++ b/src/i18n/messages/fa.json @@ -6239,6 +6239,7 @@ "requesty": "یک کلید API در https://app.requesty.ai بسازید، سپس آن را در اینجا به عنوان توکن Bearer جای‌گذاری کنید. نقطه پایانی سازگار با OpenAI در https://router.requesty.ai/v1، همراه با کاتالوگ زنده /v1/models.", "runwayml": "تولید ویدیو در Runway مبتنی بر وظیفه (task-based) است. OmniRoute کارهای تبدیل متن به ویدیو یا تصویر به ویدیو را ارسال می‌کند، وضعیت /v1/tasks/[id] را بررسی می‌کند و خروجی‌های ویدیوی نهایی را به پاسخ شبیه به OpenAI در /v1/videos/generations تبدیل می‌کند.", "sambanova": "۵ دلار اعتبار رایگان هنگام ثبت‌نام (با اعتبار ۳۰ روزه)، بدون نیاز به کارت اعتباری", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "کشف مدل از /v2/lm/scenarios/foundation-models/models در AI_API_URL استفاده می‌کند. درخواست‌های چت از deploymentUrl/chat/completions استفاده می‌کنند و به AI-Resource-Group نیاز دارند.", "sarvam": "Sarvam AI با OpenAI سازگار است در /v1. OmniRoute به /v1/models دسترسی پیدا می‌کند و ترافیک چت را به /v1/chat/completions هدایت می‌کند. مدل‌ها برای زبان‌های هندی تنظیم شده‌اند.", "scaleway": "۱ میلیون توکن رایگان برای حساب‌های جدید — سازگار با قوانین اتحادیه اروپا/GDPR (پاریس)، Qwen3 235B و Llama 70B", diff --git a/src/i18n/messages/fi.json b/src/i18n/messages/fi.json index 75173454c9..8c7c2c9c98 100644 --- a/src/i18n/messages/fi.json +++ b/src/i18n/messages/fi.json @@ -6239,6 +6239,7 @@ "requesty": "Luo API-avain osoitteessa https://app.requesty.ai ja liitä se sitten tähän Bearer-tokenina. OpenAI-yhteensopiva päätepiste osoitteessa https://router.requesty.ai/v1 reaaliaikaisella /v1/models-luettelolla.", "runwayml": "Runway-videonluonti on tehtäväpohjaista. OmniRoute lähettää teksti-videoksi- tai kuva-videoksi -töitä, kyselyttää polkua /v1/tasks/[id] ja normalisoi valmiit videotulosteet takaisin OpenAI-tyyliseen /v1/videos/generations-vastaukseen.", "sambanova": "$5 ilmaista saldoa rekisteröitymisen yhteydessä (voimassa 30 päivää), luottokorttia ei vaadita", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Mallien haku käyttää polkua /v2/lm/scenarios/foundation-models/models osoitteessa AI_API_URL. Chat-pyynnöt käyttävät polkua deploymentUrl/chat/completions ja vaativat AI-Resource-Group-otsakkeen.", "sarvam": "Sarvam AI on OpenAI-yhteensopiva /v1:ssä. OmniRoute tutkii /v1/malleja ja ohjaa keskusteluliikennettä /v1/chat/completions:iin. Mallit on säädetty indialaisille kielille.", "scaleway": "1M ilmaista tokenia uusille tileille — EU/GDPR-yhteensopiva (Pariisi), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/fr.json b/src/i18n/messages/fr.json index e3d9dc6bab..c17cb242a0 100644 --- a/src/i18n/messages/fr.json +++ b/src/i18n/messages/fr.json @@ -6239,6 +6239,7 @@ "requesty": "Créez une clé API sur https://app.requesty.ai, puis collez-la ici en tant que jeton Bearer. Point de terminaison compatible OpenAI sur https://router.requesty.ai/v1, avec un catalogue /v1/models en direct.", "runwayml": "La génération de vidéos Runway est basée sur des tâches. OmniRoute soumet des tâches text-to-video ou image-to-video, interroge /v1/tasks/[id] et normalise les sorties vidéo terminées dans la réponse de type OpenAI /v1/videos/generations.", "sambanova": "5 $ de crédits gratuits à l'inscription (validité de 30 jours), aucune carte de crédit requise", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "La découverte de modèles utilise /v2/lm/scenarios/foundation-models/models sur AI_API_URL. Les requêtes de chat utilisent deploymentUrl/chat/completions et nécessitent AI-Resource-Group.", "sarvam": "Sarvam AI est compatible avec OpenAI sur /v1. OmniRoute interroge /v1/models et achemine le trafic de chat vers /v1/chat/completions. Les modèles sont optimisés pour les langues indiennes.", "scaleway": "1M de tokens gratuits pour les nouveaux comptes — conforme UE/RGPD (Paris), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/gu.json b/src/i18n/messages/gu.json index 21a08458e5..00196e5ad8 100644 --- a/src/i18n/messages/gu.json +++ b/src/i18n/messages/gu.json @@ -6239,6 +6239,7 @@ "requesty": "https://app.requesty.ai પર API key બનાવો, પછી તેને અહીં Bearer token તરીકે પેસ્ટ કરો. લાઈવ /v1/models કેટલોગ સાથે https://router.requesty.ai/v1 પર OpenAI-સુસંગત એન્ડપોઇન્ટ.", "runwayml": "Runway વીડિયો જનરેશન ટાસ્ક-આધારિત છે. OmniRoute ટેક્સ્ટ-ટુ-વીડિયો અથવા ઇમેજ-ટુ-વીડિયો જોબ્સ સબમિટ કરે છે, /v1/tasks/[id] ને પોલ કરે છે, અને પૂર્ણ થયેલા વીડિયો આઉટપુટને ફરીથી OpenAI જેવા /v1/videos/generations રિસ્પોન્સમાં નોર્મલાઇઝ કરે છે.", "sambanova": "સાઇનઅપ પર $5 મફત ક્રેડિટ્સ (30-દિવસની માન્યતા), કોઈ ક્રેડિટ કાર્ડની જરૂર નથી", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "મોડલ ડિસ્કવરી AI_API_URL પર /v2/lm/scenarios/foundation-models/models નો ઉપયોગ કરે છે. ચેટ વિનંતીઓ deploymentUrl/chat/completions નો ઉપયોગ કરે છે અને તેના માટે AI-Resource-Group જરૂરી છે.", "sarvam": "Sarvam AI OpenAI-સંગત છે /v1. OmniRoute /v1/models ને તપાસે છે અને ચેટ ટ્રાફિકને /v1/chat/completions પર રુટ કરે છે. મોડલ્સ ઇન્ડિક ભાષાઓ માટે ટ્યુન કરવામાં આવ્યા છે.", "scaleway": "નવા એકાઉન્ટ્સ માટે 1M મફત ટોકન્સ — EU/GDPR સુસંગત (પેરિસ), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/he.json b/src/i18n/messages/he.json index 5cbdf78924..19d7e96e65 100644 --- a/src/i18n/messages/he.json +++ b/src/i18n/messages/he.json @@ -6239,6 +6239,7 @@ "requesty": "צור מפתח API בכתובת https://app.requesty.ai, ולאחר מכן הדבק אותו כאן כ-Bearer token. נקודת קצה תואמת OpenAI בכתובת https://router.requesty.ai/v1, עם קטלוג /v1/models חי.", "runwayml": "יצירת וידאו ב-Runway מבוססת משימות. OmniRoute שולח משימות text-to-video או image-to-video, דוגם את /v1/tasks/[id], ומנרמל את פלטי הווידאו המוגמרים בחזרה לתגובה דמוית OpenAI של /v1/videos/generations.", "sambanova": "קרדיט חינם בסך $5 בהרשמה (תוקף ל-30 יום), ללא צורך בכרטיס אשראי", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "גילוי מודלים משתמש ב-/v2/lm/scenarios/foundation-models/models ב-AI_API_URL. בקשות צ'אט משתמשות ב-deploymentUrl/chat/completions ודורשות את AI-Resource-Group.", "sarvam": "Sarvam AI תואם ל-OpenAI ב-/v1. OmniRoute סורק את /v1/models ומנתב את תנועת השיחה ל-/v1/chat/completions. המודלים מותאמים לשפות אינדיות.", "scaleway": "1M טוקנים בחינם לחשבונות חדשים — תואם EU/GDPR (פריז), Qwen3 235B ו-Llama 70B", diff --git a/src/i18n/messages/hi.json b/src/i18n/messages/hi.json index 844ea16376..2f9b11ace2 100644 --- a/src/i18n/messages/hi.json +++ b/src/i18n/messages/hi.json @@ -6239,6 +6239,7 @@ "requesty": "https://app.requesty.ai पर एक API कुंजी बनाएं, फिर इसे यहाँ Bearer टोकन के रूप में पेस्ट करें। https://router.requesty.ai/v1 पर OpenAI-संगत एंडपॉइंट, एक लाइव /v1/models कैटलॉग के साथ।", "runwayml": "Runway वीडियो जनरेशन टास्क-आधारित है। OmniRoute टेक्स्ट-टू-वीडियो या इमेज-टू-वीडियो जॉब सबमिट करता है, /v1/tasks/[id] को पोल करता है, और तैयार वीडियो आउटपुट को वापस OpenAI जैसे /v1/videos/generations रिस्पॉन्स में सामान्य (normalize) करता है।", "sambanova": "साइनअप पर $5 मुफ्त क्रेडिट (30 दिनों की वैधता), किसी क्रेडिट कार्ड की आवश्यकता नहीं है", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "मॉडल खोज AI_API_URL पर /v2/lm/scenarios/foundation-models/models का उपयोग करती है। चैट अनुरोध deploymentUrl/chat/completions का उपयोग करते हैं और इसके लिए AI-Resource-Group की आवश्यकता होती है।", "sarvam": "Sarvam AI OpenAI के साथ संगत है /v1. OmniRoute /v1/models को प्रॉब करता है और चैट ट्रैफिक को /v1/chat/completions पर रूट करता है। मॉडल्स को इंडिक भाषाओं के लिए ट्यून किया गया है।", "scaleway": "नए खातों के लिए 1M मुफ्त टोकन — EU/GDPR अनुपालन (पेरिस), Qwen3 235B और Llama 70B", diff --git a/src/i18n/messages/hu.json b/src/i18n/messages/hu.json index d5741d4b62..339507ca3c 100644 --- a/src/i18n/messages/hu.json +++ b/src/i18n/messages/hu.json @@ -6239,6 +6239,7 @@ "requesty": "Hozzon létre egy API-kulcsot a https://app.requesty.ai oldalon, majd illessze be ide Bearer tokenként. OpenAI-kompatibilis végpont a https://router.requesty.ai/v1 címen, élő /v1/models katalógussal.", "runwayml": "A Runway videógenerálás feladatalapú. Az OmniRoute elküldi a text-to-video vagy image-to-video feladatokat, lekérdezi a /v1/tasks/[id] állapotát, és a kész videókimeneteket visszaalakítja az OpenAI-szerű /v1/videos/generations válasszá.", "sambanova": "$5 ingyenes kredit regisztrációkor (30 napos érvényesség), bankkártya nem szükséges", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "A modellfelderítés a /v2/lm/scenarios/foundation-models/models végpontot használja az AI_API_URL címen. A chat kérések a deploymentUrl/chat/completions végpontot használják, és AI-Resource-Group szükséges hozzájuk.", "sarvam": "A Sarvam AI OpenAI-kompatibilis a /v1-en. Az OmniRoute a /v1/models-t vizsgálja és a chat forgalmat a /v1/chat/completions-re irányítja. A modellek az indiai nyelvekre vannak optimalizálva.", "scaleway": "1M ingyenes token új fiókoknak — EU/GDPR-megfelelő (Párizs), Qwen3 235B és Llama 70B", diff --git a/src/i18n/messages/id.json b/src/i18n/messages/id.json index 95f4ca816a..bf2f06a265 100644 --- a/src/i18n/messages/id.json +++ b/src/i18n/messages/id.json @@ -6239,6 +6239,7 @@ "requesty": "Buat kunci API di https://app.requesty.ai, lalu tempel di sini sebagai token Bearer. Endpoint yang kompatibel dengan OpenAI di https://router.requesty.ai/v1, dengan katalog /v1/models langsung.", "runwayml": "Pembuatan video Runway berbasis tugas. OmniRoute mengirimkan pekerjaan text-to-video atau image-to-video, melakukan polling pada /v1/tasks/[id], dan menormalisasi output video yang selesai kembali ke respons /v1/videos/generations yang mirip OpenAI.", "sambanova": "Kredit gratis $5 saat pendaftaran (validitas 30 hari), tidak memerlukan kartu kredit", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Penemuan model menggunakan /v2/lm/scenarios/foundation-models/models pada AI_API_URL. Permintaan chat menggunakan deploymentUrl/chat/completions dan memerlukan AI-Resource-Group.", "sarvam": "Sarvam AI kompatibel dengan OpenAI di /v1. OmniRoute memeriksa /v1/models dan mengarahkan lalu lintas obrolan ke /v1/chat/completions. Model disesuaikan untuk bahasa-bahasa India.", "scaleway": "1 juta token gratis untuk akun baru — patuh EU/GDPR (Paris), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/it.json b/src/i18n/messages/it.json index 4b77e9d30c..855d5b3ecb 100644 --- a/src/i18n/messages/it.json +++ b/src/i18n/messages/it.json @@ -6239,6 +6239,7 @@ "requesty": "Crea una chiave API su https://app.requesty.ai, quindi incollala qui come token Bearer. Endpoint compatibile con OpenAI su https://router.requesty.ai/v1, con un catalogo /v1/models in tempo reale.", "runwayml": "La generazione video di Runway è basata su task. OmniRoute invia lavori text-to-video o image-to-video, interroga /v1/tasks/[id] e normalizza gli output video completati nella risposta simile a OpenAI /v1/videos/generations.", "sambanova": "$5 di crediti gratuiti alla registrazione (validità 30 giorni), nessuna carta di credito richiesta", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "La scoperta dei modelli utilizza /v2/lm/scenarios/foundation-models/models su AI_API_URL. Le richieste di chat utilizzano deploymentUrl/chat/completions e richiedono AI-Resource-Group.", "sarvam": "Sarvam AI è compatibile con OpenAI su /v1. OmniRoute controlla /v1/models e instrada il traffico chat verso /v1/chat/completions. I modelli sono ottimizzati per le lingue indiane.", "scaleway": "1M di token gratuiti per i nuovi account — conforme a UE/GDPR (Parigi), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/ja.json b/src/i18n/messages/ja.json index ce2d30773a..6e734c226e 100644 --- a/src/i18n/messages/ja.json +++ b/src/i18n/messages/ja.json @@ -6239,6 +6239,7 @@ "requesty": "https://app.requesty.aiでAPIキーを作成し、ここにBearerトークンとして貼り付けます。https://router.requesty.ai/v1にあるOpenAI互換のエンドポイントは、有効な/v1/modelsカタログを提供します。", "runwayml": "Runwayの動画生成はタスクベースです。OmniRouteはtext-to-videoまたはimage-to-videoジョブを送信し、/v1/tasks/[id]をポーリングして、完了した動画出力をOpenAI風の/v1/videos/generationsレスポンスに正規化して戻します。", "sambanova": "新規登録時に$5分の無料クレジット(30日間有効)、クレジットカード不要", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "モデルの検出はAI_API_URL上の/v2/lm/scenarios/foundation-models/modelsを使用します。チャットリクエストはdeploymentUrl/chat/completionsを使用し、AI-Resource-Groupが必要です。", "sarvam": "Sarvam AIは/v1でOpenAI互換です。OmniRouteは/v1/modelsをプローブし、チャットトラフィックを/v1/chat/completionsにルーティングします。モデルはインド系言語に調整されています。", "scaleway": "新規アカウント向けに100万無料トークン — EU/GDPR準拠(パリ)、Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/ko.json b/src/i18n/messages/ko.json index c6a703ccac..fb12ad0c99 100644 --- a/src/i18n/messages/ko.json +++ b/src/i18n/messages/ko.json @@ -6239,6 +6239,7 @@ "requesty": "https://app.requesty.ai에서 API 키를 생성한 후 여기에 Bearer 토큰으로 붙여넣으세요. https://router.requesty.ai/v1의 OpenAI 호환 엔드포인트는 실시간 /v1/models 카탈로그를 제공합니다.", "runwayml": "Runway 비디오 생성은 작업 기반입니다. OmniRoute는 텍스트-비디오 또는 이미지-비디오 작업을 제출하고, /v1/tasks/[id]를 폴링하며, 완료된 비디오 출력을 OpenAI 스타일의 /v1/videos/generations 응답으로 정규화합니다.", "sambanova": "가입 시 $5 무료 크레딧 제공(유효기간 30일), 신용카드 불필요", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "모델 검색은 AI_API_URL의 /v2/lm/scenarios/foundation-models/models를 사용합니다. 채팅 요청은 deploymentUrl/chat/completions를 사용하며 AI-Resource-Group이 필요합니다.", "sarvam": "Sarvam AI는 /v1에서 OpenAI와 호환됩니다. OmniRoute는 /v1/models를 탐색하고 채팅 트래픽을 /v1/chat/completions로 라우팅합니다. 모델은 인도 언어에 맞게 조정되었습니다.", "scaleway": "신규 계정 대상 1M 무료 토큰 — EU/GDPR 준수(파리), Qwen3 235B 및 Llama 70B", diff --git a/src/i18n/messages/mr.json b/src/i18n/messages/mr.json index aeeb20b219..8817c1c87f 100644 --- a/src/i18n/messages/mr.json +++ b/src/i18n/messages/mr.json @@ -6239,6 +6239,7 @@ "requesty": "https://app.requesty.ai वर API की तयार करा, नंतर ती येथे Bearer टोकन म्हणून पेस्ट करा. https://router.requesty.ai/v1 वर OpenAI-सुसंगत एंडपॉइंट आहे, ज्यामध्ये थेट /v1/models कॅटलॉग उपलब्ध आहे.", "runwayml": "Runway व्हिडिओ निर्मिती ही टास्क-आधारित आहे. OmniRoute हे text-to-video किंवा image-to-video जॉब्स सबमिट करते, /v1/tasks/[id] पोल करते आणि पूर्ण झालेल्या व्हिडिओ आउटपुटला पुन्हा OpenAI-सारख्या /v1/videos/generations प्रतिसादात सामान्य करते.", "sambanova": "साइनअपवर $5 मोफत क्रेडिट्स (30 दिवसांची वैधता), क्रेडिट कार्डची आवश्यकता नाही", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "मॉडेल शोध AI_API_URL वरील /v2/lm/scenarios/foundation-models/models वापरतो. चॅट विनंत्या deploymentUrl/chat/completions वापरतात आणि त्यासाठी AI-Resource-Group आवश्यक आहे.", "sarvam": "Sarvam AI हे OpenAI-संगत आहे /v1. OmniRoute /v1/models चा शोध घेतो आणि चॅट ट्रॅफिक /v1/chat/completions कडे मार्गदर्शित करतो. मॉडेल्स भारतीय भाषांसाठी ट्यून केलेले आहेत.", "scaleway": "नवीन खात्यांसाठी 1M मोफत टोकन्स — EU/GDPR सुसंगत (पॅरिस), Qwen3 235B आणि Llama 70B", diff --git a/src/i18n/messages/ms.json b/src/i18n/messages/ms.json index b506128049..186db22e3a 100644 --- a/src/i18n/messages/ms.json +++ b/src/i18n/messages/ms.json @@ -6239,6 +6239,7 @@ "requesty": "Cipta kunci API di https://app.requesty.ai, kemudian tampalkannya di sini sebagai token Bearer. Titik akhir serasi OpenAI di https://router.requesty.ai/v1, dengan katalog /v1/models langsung.", "runwayml": "Penjanaan video Runway adalah berasaskan tugas. OmniRoute menyerahkan kerja teks-ke-video atau imej-ke-video, meninjau /v1/tasks/[id], dan menormalkan output video yang telah selesai kembali ke dalam respons /v1/videos/generations seperti OpenAI.", "sambanova": "Kredit percuma $5 semasa pendaftaran (tempoh sah 30 hari), tiada kad kredit diperlukan", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Penemuan model menggunakan /v2/lm/scenarios/foundation-models/models pada AI_API_URL. Permintaan sembang menggunakan deploymentUrl/chat/completions dan memerlukan AI-Resource-Group.", "sarvam": "Sarvam AI adalah serasi dengan OpenAI pada /v1. OmniRoute menyiasat /v1/models dan mengarahkan trafik sembang ke /v1/chat/completions. Model disesuaikan untuk bahasa-bahasa India.", "scaleway": "1M token percuma untuk akaun baharu — mematuhi EU/GDPR (Paris), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/nl.json b/src/i18n/messages/nl.json index d92b373142..edf66b80a1 100644 --- a/src/i18n/messages/nl.json +++ b/src/i18n/messages/nl.json @@ -6239,6 +6239,7 @@ "requesty": "Maak een API-sleutel aan op https://app.requesty.ai en plak deze hier als een Bearer-token. OpenAI-compatibel eindpunt op https://router.requesty.ai/v1, met een live /v1/models-catalogus.", "runwayml": "Runway-videogeneratie is taakgebaseerd. OmniRoute dient text-to-video- of image-to-video-taken in, peilt /v1/tasks/[id] en normaliseert de voltooide video-uitvoer terug naar het OpenAI-achtige /v1/videos/generations-antwoord.", "sambanova": "$5 gratis tegoed bij aanmelding (30 dagen geldig), geen creditcard vereist", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Modeldetectie gebruikt /v2/lm/scenarios/foundation-models/models op AI_API_URL. Chatverzoeken gebruiken deploymentUrl/chat/completions en vereisen AI-Resource-Group.", "sarvam": "Sarvam AI is OpenAI-compatibel op /v1. OmniRoute onderzoekt /v1/models en leidt chatverkeer naar /v1/chat/completions. Modellen zijn afgestemd op Indic-talen.", "scaleway": "1M gratis tokens voor nieuwe accounts — EU/AVG-conform (Parijs), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/no.json b/src/i18n/messages/no.json index 1e415be8aa..d4e45081b6 100644 --- a/src/i18n/messages/no.json +++ b/src/i18n/messages/no.json @@ -6239,6 +6239,7 @@ "requesty": "Opprett en API-nøkkel på https://app.requesty.ai, og lim den deretter inn her som et Bearer-token. OpenAI-kompatibelt endepunkt på https://router.requesty.ai/v1, med en live /v1/models-katalog.", "runwayml": "Runway-videogenerering er oppgavebasert. OmniRoute sender inn tekst-til-video- eller bilde-til-video-jobber, poller /v1/tasks/[id], og normaliserer de ferdige videoresultatene tilbake til den OpenAI-lignende /v1/videos/generations-responsen.", "sambanova": "$5 gratis kreditt ved registrering (30 dagers gyldighet), ikke krav om kredittkort", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Modellsøk bruker /v2/lm/scenarios/foundation-models/models på AI_API_URL. Chat-forespørsler bruker deploymentUrl/chat/completions og krever AI-Resource-Group.", "sarvam": "Sarvam AI er OpenAI-kompatibel på /v1. OmniRoute undersøker /v1/models og ruter chat-trafikk til /v1/chat/completions. Modeller er tilpasset for indiske språk.", "scaleway": "1M gratis tokens for nye kontoer — EU/GDPR-kompatibel (Paris), Qwen3 235B og Llama 70B", diff --git a/src/i18n/messages/phi.json b/src/i18n/messages/phi.json index da06b21c62..6f9f973ec1 100644 --- a/src/i18n/messages/phi.json +++ b/src/i18n/messages/phi.json @@ -6239,6 +6239,7 @@ "requesty": "Gumawa ng API key sa https://app.requesty.ai, pagkatapos ay i-paste ito rito bilang isang Bearer token. OpenAI-compatible na endpoint sa https://router.requesty.ai/v1, na may live na catalog ng /v1/models.", "runwayml": "Ang pagbuo ng video sa Runway ay task-based. Nagpapasa ang OmniRoute ng mga text-to-video o image-to-video na job, nagpo-poll sa /v1/tasks/[id], at nino-normalize ang mga natapos na video output pabalik sa OpenAI-like na tugon ng /v1/videos/generations.", "sambanova": "$5 na libreng credits sa pag-signup (may bisa sa loob ng 30 araw), walang kinakailangang credit card", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Gumagamit ang pagtuklas ng modelo ng /v2/lm/scenarios/foundation-models/models sa AI_API_URL. Gumagamit ang mga kahilingan sa chat ng deploymentUrl/chat/completions at nangangailangan ng AI-Resource-Group.", "sarvam": "Ang Sarvam AI ay katugma ng OpenAI sa /v1. Ang OmniRoute ay nag-uusisa sa /v1/models at nagruruta ng chat traffic sa /v1/chat/completions. Ang mga modelo ay na-tune para sa mga wikang Indic.", "scaleway": "1M libreng token para sa mga bagong account — sumusunod sa EU/GDPR (Paris), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/pl.json b/src/i18n/messages/pl.json index ee567fe371..002c11a98d 100644 --- a/src/i18n/messages/pl.json +++ b/src/i18n/messages/pl.json @@ -6239,6 +6239,7 @@ "requesty": "Utwórz klucz API na https://app.requesty.ai, a następnie wklej go tutaj jako token Bearer. Punkt końcowy zgodny z OpenAI pod adresem https://router.requesty.ai/v1, z aktywnym katalogiem /v1/models.", "runwayml": "Generowanie wideo w Runway opiera się na zadaniach. OmniRoute przesyła zadania text-to-video lub image-to-video, odpytuje /v1/tasks/[id] i normalizuje gotowe wyniki wideo z powrotem do odpowiedzi w stylu OpenAI /v1/videos/generations.", "sambanova": "$5 darmowych kredytów przy rejestracji (ważność 30 dni), karta kredytowa nie jest wymagana", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Wykrywanie modeli używa /v2/lm/scenarios/foundation-models/models na AI_API_URL. Żądania czatu używają deploymentUrl/chat/completions i wymagają AI-Resource-Group.", "sarvam": "Sarvam AI jest zgodny z OpenAI na /v1. OmniRoute bada /v1/models i kieruje ruch czatu do /v1/chat/completions. Modele są dostosowane do języków indyjskich.", "scaleway": "1M darmowych tokenów dla nowych kont — zgodność z UE/RODO (Paryż), Qwen3 235B i Llama 70B", diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index 3fbe27e908..5717bff5cc 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -6243,6 +6243,7 @@ "requesty": "Crie uma chave de API em https://app.requesty.ai e cole-a aqui como um token Bearer. Endpoint compatível com OpenAI em https://router.requesty.ai/v1, com um catálogo /v1/models ao vivo.", "runwayml": "A geração de vídeo do Runway é baseada em tarefas. O OmniRoute envia jobs de texto-para-vídeo ou imagem-para-vídeo, consulta /v1/tasks/[id] periodicamente e normaliza as saídas de vídeo concluídas de volta para a resposta /v1/videos/generations no estilo OpenAI.", "sambanova": "$5 em créditos gratuitos no cadastro (validade de 30 dias), sem necessidade de cartão de crédito", + "seekai": "Crie uma chave de API em https://seekai.cc e cole aqui como Bearer token. URL base compatível com OpenAI: https://seekai.cc/v1.", "sap": "A descoberta de modelos usa /v2/lm/scenarios/foundation-models/models em AI_API_URL. As solicitações de chat usam deploymentUrl/chat/completions e requerem AI-Resource-Group.", "sarvam": "O Sarvam AI é compatível com OpenAI em /v1. O OmniRoute sonda /v1/models e roteia o tráfego de chat para /v1/chat/completions. Os modelos são ajustados para línguas indianas.", "scaleway": "1M tokens gratuitos para novas contas — compatível com UE/GDPR (Paris), Qwen3 235B e Llama 70B", diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index 4919461ca0..4722209d02 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -6239,6 +6239,7 @@ "requesty": "Crie uma chave de API em https://app.requesty.ai e, em seguida, cole-a aqui como um token Bearer. Endpoint compatível com a OpenAI em https://router.requesty.ai/v1, com um catálogo /v1/models em tempo real.", "runwayml": "A geração de vídeo da Runway é baseada em tarefas. O OmniRoute submete tarefas de texto para vídeo ou imagem para vídeo, consulta /v1/tasks/[id] e normaliza as saídas de vídeo concluídas de volta para a resposta /v1/videos/generations semelhante à da OpenAI.", "sambanova": "$5 em créditos gratuitos no registo (validade de 30 dias), sem necessidade de cartão de crédito", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "A descoberta de modelos utiliza /v2/lm/scenarios/foundation-models/models em AI_API_URL. Os pedidos de chat utilizam deploymentUrl/chat/completions e requerem AI-Resource-Group.", "sarvam": "Sarvam AI é compatível com OpenAI em /v1. OmniRoute investiga /v1/models e direciona o tráfego de chat para /v1/chat/completions. Os modelos são ajustados para línguas índicas.", "scaleway": "1M de tokens gratuitos para novas contas — em conformidade com a UE/RGPD (Paris), Qwen3 235B e Llama 70B", diff --git a/src/i18n/messages/ro.json b/src/i18n/messages/ro.json index 2d012143f9..a0d82ca4d5 100644 --- a/src/i18n/messages/ro.json +++ b/src/i18n/messages/ro.json @@ -6239,6 +6239,7 @@ "requesty": "Creează o cheie API la https://app.requesty.ai, apoi lipește-o aici ca token Bearer. Endpoint compatibil cu OpenAI la https://router.requesty.ai/v1, cu un catalog /v1/models live.", "runwayml": "Generarea video Runway este bazată pe sarcini. OmniRoute trimite lucrări text-to-video sau image-to-video, interoghează periodic /v1/tasks/[id] și normalizează ieșirile video finalizate înapoi în răspunsul de tip OpenAI /v1/videos/generations.", "sambanova": "$5 credite gratuite la înregistrare (valabilitate 30 de zile), nu este necesar un card de credit", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Descoperirea modelelor folosește /v2/lm/scenarios/foundation-models/models pe AI_API_URL. Solicitările de chat folosesc deploymentUrl/chat/completions și necesită AI-Resource-Group.", "sarvam": "Sarvam AI este compatibil cu OpenAI pe /v1. OmniRoute probează /v1/models și direcționează traficul de chat către /v1/chat/completions. Modelele sunt ajustate pentru limbile indic.", "scaleway": "1M tokenuri gratuite pentru conturi noi — conformitate UE/GDPR (Paris), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/ru.json b/src/i18n/messages/ru.json index 9f6bd4762e..4a7646aaef 100644 --- a/src/i18n/messages/ru.json +++ b/src/i18n/messages/ru.json @@ -6239,6 +6239,7 @@ "requesty": "Создайте API-ключ на https://app.requesty.ai, затем вставьте его сюда в качестве Bearer-токена. Совместимая с OpenAI конечная точка находится по адресу https://router.requesty.ai/v1, с живым каталогом /v1/models.", "runwayml": "Генерация видео в Runway основана на задачах. OmniRoute отправляет задания text-to-video или image-to-video, опрашивает /v1/tasks/[id] и нормализует готовые видеовыходы обратно в ответ /v1/videos/generations, аналогичный OpenAI.", "sambanova": "$5 бесплатных кредитов при регистрации (срок действия 30 дней), кредитная карта не требуется", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Обнаружение моделей использует /v2/lm/scenarios/foundation-models/models на AI_API_URL. Запросы чата используют deploymentUrl/chat/completions и требуют AI-Resource-Group.", "sarvam": "Sarvam AI — (sarvam) — Индийские языковые AI модели", "scaleway": "1 млн бесплатных токенов для новых аккаунтов — соответствие EU/GDPR (Париж), Qwen3 235B и Llama 70B", diff --git a/src/i18n/messages/sk.json b/src/i18n/messages/sk.json index b5fc50f625..66a01e79e2 100644 --- a/src/i18n/messages/sk.json +++ b/src/i18n/messages/sk.json @@ -6239,6 +6239,7 @@ "requesty": "Vytvorte API kľúč na https://app.requesty.ai, potom ho sem vložte ako Bearer token. Koncový bod kompatibilný s OpenAI na https://router.requesty.ai/v1, so živým katalógom /v1/models.", "runwayml": "Generovanie videa v Runway je založené na úlohách. OmniRoute odosiela úlohy text-to-video alebo image-to-video, dopytuje sa na /v1/tasks/[id] a normalizuje hotové video výstupy späť do odpovede /v1/videos/generations podobnej OpenAI.", "sambanova": "Bezplatný kredit 5 $ pri registrácii (platnosť 30 dní), nevyžaduje sa žiadna kreditná karta", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Vyhľadávanie modelov používa /v2/lm/scenarios/foundation-models/models na AI_API_URL. Požiadavky na chat používajú deploymentUrl/chat/completions a vyžadujú AI-Resource-Group.", "sarvam": "Sarvam AI je kompatibilný s OpenAI na /v1. OmniRoute skúma /v1/models a smeruje chatový prenos na /v1/chat/completions. Modely sú optimalizované pre indické jazyky.", "scaleway": "1M bezplatných tokenov pre nové účty — v súlade s EÚ/GDPR (Paríž), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/sv.json b/src/i18n/messages/sv.json index eb8a2bd8f3..d8ca20a04d 100644 --- a/src/i18n/messages/sv.json +++ b/src/i18n/messages/sv.json @@ -6239,6 +6239,7 @@ "requesty": "Skapa en API-nyckel på https://app.requesty.ai, klistra sedan in den här som en Bearer-token. OpenAI-kompatibel slutpunkt på https://router.requesty.ai/v1, med en live /v1/models-katalog.", "runwayml": "Runway-videogenerering är uppgiftsbaserad. OmniRoute skickar in text-till-video- eller bild-till-video-jobb, pollar /v1/tasks/[id] och normaliserar de färdiga videoutdata tillbaka till det OpenAI-liknande /v1/videos/generations-svaret.", "sambanova": "$5 i gratis kredit vid registrering (30 dagars giltighet), inget kreditkort krävs", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Modellidentifiering använder /v2/lm/scenarios/foundation-models/models på AI_API_URL. Chatförfrågningar använder deploymentUrl/chat/completions och kräver AI-Resource-Group.", "sarvam": "Sarvam AI är OpenAI-kompatibel på /v1. OmniRoute undersöker /v1/models och dirigerar chatttrafik till /v1/chat/completions. Modellerna är anpassade för indiska språk.", "scaleway": "1M gratis tokens för nya konton — EU/GDPR-kompatibel (Paris), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/sw.json b/src/i18n/messages/sw.json index 31c55e5173..5507c02543 100644 --- a/src/i18n/messages/sw.json +++ b/src/i18n/messages/sw.json @@ -6239,6 +6239,7 @@ "requesty": "Unda ufunguo wa API kwenye https://app.requesty.ai, kisha ubandike hapa kama tokeni ya Bearer. Endpoint inayooana na OpenAI iko kwenye https://router.requesty.ai/v1, ikiwa na orodha ya moja kwa moja ya /v1/models.", "runwayml": "Uzalishaji wa video wa Runway unategemea kazi. OmniRoute huwasilisha kazi za maandishi-hadi-video au picha-hadi-video, huangalia mara kwa mara /v1/tasks/[id], na kurekebisha matokeo ya video yaliyokamilika kurudi kwenye jibu la /v1/videos/generations linalofanana na OpenAI.", "sambanova": "Salio la bure la $5 unapojisajili (uhalali wa siku 30), hakuna kadi ya mkopo inayohitajika", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Ugunduzi wa miundo hutumia /v2/lm/scenarios/foundation-models/models kwenye AI_API_URL. Maombi ya gumzo hutumia deploymentUrl/chat/completions na yanahitaji AI-Resource-Group.", "sarvam": "Sarvam AI inapatana na OpenAI kwenye /v1. OmniRoute inachunguza /v1/models na kuelekeza trafiki ya mazungumzo kwenye /v1/chat/completions. Mifano imeboreshwa kwa lugha za Kihindi.", "scaleway": "Tokeni 1M za bure kwa akaunti mpya — inatii EU/GDPR (Paris), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/ta.json b/src/i18n/messages/ta.json index 49679193d9..a5e601ae93 100644 --- a/src/i18n/messages/ta.json +++ b/src/i18n/messages/ta.json @@ -6239,6 +6239,7 @@ "requesty": "https://app.requesty.ai இல் ஒரு API கீயை உருவாக்கி, பின்னர் அதை இங்கே Bearer டோக்கனாக ஒட்டவும். OpenAI-உடன் இணக்கமான எண்ட்பாயிண்ட் https://router.requesty.ai/v1 இல் நேரடி /v1/models பட்டியலுடன் உள்ளது.", "runwayml": "Runway வீடியோ உருவாக்கம் என்பது பணி அடிப்படையிலானது. OmniRoute ஆனது உரை-க்கு-வீடியோ அல்லது படம்-க்கு-வீடியோ பணிகளைச் சமர்ப்பித்து, /v1/tasks/[id] ஐத் தொடர்ந்து சரிபார்த்து, முடிக்கப்பட்ட வீடியோ வெளியீடுகளை மீண்டும் OpenAI போன்ற /v1/videos/generations பதிலுக்கு இயல்பாக்குகிறது.", "sambanova": "பதிவு செய்யும் போது $5 இலவச கிரெடிட்கள் (30 நாட்கள் செல்லுபடியாகும்), கிரெடிட் கார்டு தேவையில்லை", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "மாதிரி கண்டறிதல் ஆனது AI_API_URL இல் /v2/lm/scenarios/foundation-models/models ஐப் பயன்படுத்துகிறது. அரட்டை கோரிக்கைகள் deploymentUrl/chat/completions ஐப் பயன்படுத்துகின்றன மற்றும் AI-Resource-Group தேவைப்படுகிறது.", "sarvam": "Sarvam AI OpenAI-க்கு இணக்கமானது /v1 இல். OmniRoute /v1/models ஐ ஆராய்ந்து /v1/chat/completions க்கு உரையாடல் போக்குவரத்தை வழிமொழிகிறது. மாதிரிகள் இந்திய மொழிகளுக்காக அமைக்கப்பட்டுள்ளது.", "scaleway": "புதிய கணக்குகளுக்கு 1M இலவச டோக்கன்கள் — EU/GDPR இணக்கமானது (பாரிஸ்), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/te.json b/src/i18n/messages/te.json index 1a5223b1c5..69a51f4d6a 100644 --- a/src/i18n/messages/te.json +++ b/src/i18n/messages/te.json @@ -6239,6 +6239,7 @@ "requesty": "https://app.requesty.ai వద్ద API కీని సృష్టించండి, ఆపై దానిని ఇక్కడ Bearer టోకెన్‌గా పేస్ట్ చేయండి. https://router.requesty.ai/v1 వద్ద లైవ్ /v1/models కేటలాగ్‌తో OpenAI-అనుకూల ఎండ్‌పాయింట్ ఉంది.", "runwayml": "Runway వీడియో జనరేషన్ టాస్క్-ఆధారితమైనది. OmniRoute అనేది టెక్స్ట్-టు-వీడియో లేదా ఇమేజ్-టు-వీడియో జాబ్‌లను సమర్పిస్తుంది, /v1/tasks/[id] ని పోల్ చేస్తుంది మరియు పూర్తయిన వీడియో అవుట్‌పుట్‌లను తిరిగి OpenAI-వంటి /v1/videos/generations ప్రతిస్పందనగా సాధారణీకరిస్తుంది.", "sambanova": "సైన్అప్ చేసినప్పుడు $5 ఉచిత క్రెడిట్‌లు (30 రోజుల చెల్లుబాటు), క్రెడిట్ కార్డ్ అవసరం లేదు", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "మోడల్ డిస్కవరీ AI_API_URL లో /v2/lm/scenarios/foundation-models/models ని ఉపయోగిస్తుంది. చాట్ అభ్యర్థనలు deploymentUrl/chat/completions ని ఉపయోగిస్తాయి మరియు AI-Resource-Group అవసరం.", "sarvam": "Sarvam AI OpenAI-తో అనుకూలంగా ఉంది /v1. OmniRoute /v1/modelsని ప్రోబ్ చేస్తుంది మరియు చాట్ ట్రాఫిక్‌ను /v1/chat/completionsకి రూట్ చేస్తుంది. మోడల్స్ ఇండిక్ భాషల కోసం ట్యూన్ చేయబడ్డాయి.", "scaleway": "కొత్త ఖాతాల కోసం 1M ఉచిత టోకెన్‌లు — EU/GDPR కంప్లైంట్ (పారిస్), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/th.json b/src/i18n/messages/th.json index a7093e6758..b43dda761c 100644 --- a/src/i18n/messages/th.json +++ b/src/i18n/messages/th.json @@ -6239,6 +6239,7 @@ "requesty": "สร้างคีย์ API ที่ https://app.requesty.ai จากนั้นวางที่นี่เป็นโทเค็น Bearer ปลายทางที่เข้ากันได้กับ OpenAI อยู่ที่ https://router.requesty.ai/v1 พร้อมแคตตาล็อก /v1/models แบบสด", "runwayml": "การสร้างวิดีโอของ Runway เป็นแบบอิงตามงาน OmniRoute จะส่งงาน text-to-video หรือ image-to-video ดึงข้อมูลสถานะ /v1/tasks/[id] เป็นระยะ และปรับเอาต์พุตวิดีโอที่เสร็จสมบูรณ์ให้อยู่ในรูปแบบการตอบกลับ /v1/videos/generations ที่คล้ายกับ OpenAI", "sambanova": "เครดิตฟรี $5 เมื่อลงทะเบียน (มีอายุ 30 วัน) ไม่ต้องใช้บัตรเครดิต", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "การค้นหาโมเดลใช้ /v2/lm/scenarios/foundation-models/models บน AI_API_URL คำขอแชทใช้ deploymentUrl/chat/completions และต้องระบุ AI-Resource-Group", "sarvam": "Sarvam AI เข้ากันได้กับ OpenAI ที่ /v1. OmniRoute ตรวจสอบ /v1/models และจัดเส้นทางการสนทนาไปยัง /v1/chat/completions. โมเดลได้รับการปรับแต่งสำหรับภาษาอินดิก.", "scaleway": "โทเค็นฟรี 1M สำหรับบัญชีใหม่ — สอดคล้องตาม EU/GDPR (ปารีส), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/tr.json b/src/i18n/messages/tr.json index c7ea532a2f..31e48fb100 100644 --- a/src/i18n/messages/tr.json +++ b/src/i18n/messages/tr.json @@ -6239,6 +6239,7 @@ "requesty": "https://app.requesty.ai adresinde bir API anahtarı oluşturun, ardından buraya Bearer token olarak yapıştırın. OpenAI uyumlu uç nokta, canlı bir /v1/models kataloğu ile birlikte https://router.requesty.ai/v1 adresindedir.", "runwayml": "Runway video üretimi görev tabanlıdır. OmniRoute, metinden videoya veya görselden videoya işleri gönderir, /v1/tasks/[id] uç noktasını sorgular ve tamamlanan video çıktılarını OpenAI benzeri /v1/videos/generations yanıtına normalleştirir.", "sambanova": "Kayıt olunduğunda $5 ücretsiz kredi (30 gün geçerli), kredi kartı gerekmez", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Model keşfi, AI_API_URL üzerindeki /v2/lm/scenarios/foundation-models/models yolunu kullanır. Sohbet istekleri deploymentUrl/chat/completions yolunu kullanır ve AI-Resource-Group gerektirir.", "sarvam": "Sarvam AI, OpenAI ile uyumludur ve /v1 üzerinde çalışır. OmniRoute, /v1/models'ı sorgular ve sohbet trafiğini /v1/chat/completions'a yönlendirir. Modeller, Hint dilleri için ayarlanmıştır.", "scaleway": "Yeni hesaplar için 1M ücretsiz token — AB/GDPR uyumlu (Paris), Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/uk-UA.json b/src/i18n/messages/uk-UA.json index f6a1e7f0b1..62d8db3226 100644 --- a/src/i18n/messages/uk-UA.json +++ b/src/i18n/messages/uk-UA.json @@ -6239,6 +6239,7 @@ "requesty": "Створіть API-ключ на https://app.requesty.ai, а потім вставте його сюди як Bearer-токен. OpenAI-сумісна кінцева точка на https://router.requesty.ai/v1, з живим каталогом /v1/models.", "runwayml": "Генерація відео в Runway базується на завданнях. OmniRoute надсилає завдання text-to-video або image-to-video, опитує /v1/tasks/[id] та нормалізує готові відеофайли назад у відповідь типу OpenAI /v1/videos/generations.", "sambanova": "$5 безкоштовних кредитів при реєстрації (дійсні 30 днів), кредитна картка не потрібна", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Виявлення моделей використовує /v2/lm/scenarios/foundation-models/models на AI_API_URL. Запити чату використовують deploymentUrl/chat/completions та потребують AI-Resource-Group.", "sarvam": "Sarvam AI сумісний з OpenAI на /v1. OmniRoute перевіряє /v1/models і маршрутизує чат-трафік на /v1/chat/completions. Моделі налаштовані для індійських мов.", "scaleway": "1 млн безкоштовних токенів для нових акаунтів — сумісно з EU/GDPR (Париж), Qwen3 235B та Llama 70B", diff --git a/src/i18n/messages/ur.json b/src/i18n/messages/ur.json index bf80a21de8..1a4f1de826 100644 --- a/src/i18n/messages/ur.json +++ b/src/i18n/messages/ur.json @@ -6239,6 +6239,7 @@ "requesty": "https://app.requesty.ai پر ایک API key بنائیں، پھر اسے یہاں Bearer ٹوکن کے طور پر پیسٹ کریں۔ OpenAI سے ہم آہنگ اینڈ پوائنٹ https://router.requesty.ai/v1 پر ہے، جس میں ایک لائیو /v1/models کیٹلاگ موجود ہے۔", "runwayml": "Runway ویڈیو جنریشن ٹاسک پر مبنی ہے۔ OmniRoute، text-to-video یا image-to-video جابز جمع کراتا ہے، /v1/tasks/[id] کو پول کرتا ہے، اور مکمل شدہ ویڈیو آؤٹ پٹس کو واپس OpenAI جیسے /v1/videos/generations ریسپانس میں نارملائز کرتا ہے۔", "sambanova": "سائن اپ کرنے پر $5 مفت کریڈٹس (30 دن کی میعاد)، کسی کریڈٹ کارڈ کی ضرورت نہیں ہے", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "ماڈل کی دریافت AI_API_URL پر /v2/lm/scenarios/foundation-models/models کا استعمال کرتی ہے۔ چیٹ کی درخواستیں deploymentUrl/chat/completions کا استعمال کرتی ہیں اور ان کے لیے AI-Resource-Group درکار ہوتا ہے۔", "sarvam": "Sarvam AI OpenAI کے ساتھ ہم آہنگ ہے /v1 پر۔ OmniRoute /v1/models کی جانچ کرتا ہے اور چیٹ ٹریفک کو /v1/chat/completions پر بھیجتا ہے۔ ماڈلز کو انڈک زبانوں کے لیے ترتیب دیا گیا ہے۔", "scaleway": "نئے اکاؤنٹس کے لیے 1M مفت ٹوکنز — EU/GDPR کے مطابق (پیرس)، Qwen3 235B اور Llama 70B", diff --git a/src/i18n/messages/vi.json b/src/i18n/messages/vi.json index f2e4a120bc..0a8428ecdf 100644 --- a/src/i18n/messages/vi.json +++ b/src/i18n/messages/vi.json @@ -6243,6 +6243,7 @@ "requesty": "Tạo khóa API tại https://app.requesty.ai, rồi dán dưới dạng token Bearer. Endpoint tương thích OpenAI tại https://router.requesty.ai/v1, kèm danh mục /v1/models trực tiếp.", "runwayml": "Tạo video Runway hoạt động theo tác vụ. OmniRoute gửi tác vụ chuyển văn bản hoặc hình ảnh thành video, thăm dò /v1/tasks/[id], rồi chuẩn hóa đầu ra hoàn tất về phản hồi /v1/videos/generations kiểu OpenAI.", "sambanova": "5 USD tín dụng miễn phí khi đăng ký (có hiệu lực 30 ngày), không cần thẻ tín dụng", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "Khám phá mô hình dùng /v2/lm/scenarios/foundation-models/models trên AI_API_URL. Yêu cầu trò chuyện dùng deploymentUrl/chat/completions và yêu cầu AI-Resource-Group.", "sarvam": "Sarvam AI tương thích OpenAI trên /v1. OmniRoute thăm dò /v1/models và định tuyến trò chuyện tới /v1/chat/completions. Các mô hình được tinh chỉnh cho ngôn ngữ Ấn Độ.", "scaleway": "1 triệu token miễn phí cho tài khoản mới — tuân thủ EU/GDPR (Paris), Qwen3 235B và Llama 70B", diff --git a/src/i18n/messages/zh-CN.json b/src/i18n/messages/zh-CN.json index 2301729df6..facdccc8c2 100644 --- a/src/i18n/messages/zh-CN.json +++ b/src/i18n/messages/zh-CN.json @@ -6239,6 +6239,7 @@ "requesty": "在 https://app.requesty.ai 创建 API 密钥,然后将其作为 Bearer 令牌粘贴在此处。兼容 OpenAI 的端点位于 https://router.requesty.ai/v1,并提供实时的 /v1/models 目录。", "runwayml": "Runway 视频生成基于任务。OmniRoute 提交文生视频或图生视频作业,轮询 /v1/tasks/[id],并将完成的视频输出规范化为类似 OpenAI 的 /v1/videos/generations 响应。", "sambanova": "注册即送 $5 免费额度(30 天有效期),无需信用卡", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "模型发现使用 AI_API_URL 上的 /v2/lm/scenarios/foundation-models/models。Chat 请求使用 deploymentUrl/chat/completions 并需要 AI-Resource-Group。", "sarvam": "Sarvam AI 在 /v1 上兼容 OpenAI。OmniRoute 探测 /v1/models 并将聊天流量路由到 /v1/chat/completions。模型针对印度语言进行了优化。", "scaleway": "新账户可获 1M 免费 Token — 符合 EU/GDPR 规范(巴黎),Qwen3 235B & Llama 70B", diff --git a/src/i18n/messages/zh-TW.json b/src/i18n/messages/zh-TW.json index f0dea4fe5c..3d10722987 100644 --- a/src/i18n/messages/zh-TW.json +++ b/src/i18n/messages/zh-TW.json @@ -6239,6 +6239,7 @@ "requesty": "在 https://app.requesty.ai 建立 API 金鑰,然後以 Bearer token 形式貼上。OpenAI 相容端點為 https://router.requesty.ai/v1,附即時 /v1/models 目錄。", "runwayml": "Runway 影片生成為任務導向。OmniRoute 提交文字轉影片或圖片轉影片作業,輪詢 /v1/tasks/[id],並將完成的影片輸出正規化為類似 OpenAI 的 /v1/videos/generations 回應。", "sambanova": "註冊即贈 $5 美元免費額度(30 天有效期),無需信用卡", + "seekai": "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", "sap": "模型探索使用 AI_API_URL 上的 /v2/lm/scenarios/foundation-models/models。聊天請求使用 deploymentUrl/chat/completions,並需要 AI-Resource-Group。", "sarvam": "使用API Key連接Sarvam AI。", "scaleway": "新帳戶贈送 100 萬免費 tokens — 符合歐盟/GDPR 規範(巴黎),Qwen3 235B 和 Llama 70B", diff --git a/src/shared/constants/config.ts b/src/shared/constants/config.ts index 0810042444..8e21b9c0dd 100644 --- a/src/shared/constants/config.ts +++ b/src/shared/constants/config.ts @@ -2,6 +2,7 @@ export { APP_CONFIG, THEME_CONFIG } from "./appConfig"; // Provider API endpoints (for display only) export const PROVIDER_ENDPOINTS = { + seekai: "https://seekai.cc/v1/chat/completions", agentrouter: "https://agentrouter.org/v1/chat/completions", openrouter: "https://openrouter.ai/api/v1/chat/completions", dgrid: "https://api.dgrid.ai/v1/chat/completions", diff --git a/src/shared/constants/providers.ts b/src/shared/constants/providers.ts index 7732808fbc..ef8d8ac94e 100644 --- a/src/shared/constants/providers.ts +++ b/src/shared/constants/providers.ts @@ -144,6 +144,7 @@ export const AGGREGATOR_PROVIDER_IDS = new Set([ "helixmind", "tabitoken", "logfare", + "seekai", ]); export const ENTERPRISE_CLOUD_PROVIDER_IDS = new Set([ diff --git a/src/shared/constants/providers/apikey/gateways.ts b/src/shared/constants/providers/apikey/gateways.ts index 1a37aab7f3..7f09204e3d 100644 --- a/src/shared/constants/providers/apikey/gateways.ts +++ b/src/shared/constants/providers/apikey/gateways.ts @@ -1435,4 +1435,24 @@ export const APIKEY_PROVIDERS_GATEWAYS = { apiHint: "Create an sk- key at https://tabitoken.com and use https://tabitoken.com. The Anthropic-compatible /v1/messages endpoint (default) takes x-api-key; /v1/chat/completions takes Bearer.", }, + // SeekAi (https://seekai.cc) — QuantumNous New-API aggregator. Live-verified + // 2026-09-02: GET /api/status → system_name=SeekAi, version=v1.0.0-rc.25, + // quota_display_type=USD. OpenAI-compatible /v1; models discovered live. + seekai: { + id: "seekai", + serviceKinds: ["llm"], + alias: "ska", + name: "SeekAi", + icon: "hub", + color: "#0D9488", + textIcon: "SK", + passthroughModels: true, + website: "https://seekai.cc", + hasFree: true, + freeNote: "Signup credit toward available models; amount and eligibility are set by SeekAi, not OmniRoute.", + authHint: + "Create an API key at https://seekai.cc, then paste it here as a Bearer token.", + apiHint: + "Create an API key at https://seekai.cc, then paste it here as a Bearer token. OpenAI-compatible base URL: https://seekai.cc/v1.", + }, }; diff --git a/tests/snapshots/provider/translate-path.json b/tests/snapshots/provider/translate-path.json index 64fd60cfa2..63db4728b1 100644 --- a/tests/snapshots/provider/translate-path.json +++ b/tests/snapshots/provider/translate-path.json @@ -5288,6 +5288,29 @@ "stream": "https://api.sea-lion.ai/v1/chat/completions" } }, + "seekai": { + "format": "openai", + "headers": { + "apiKey": { + "Accept": "text/event-stream", + "Authorization": "Bearer ", + "Content-Type": "application/json" + }, + "nonStream": { + "Authorization": "Bearer ", + "Content-Type": "application/json" + }, + "oauth": { + "Accept": "text/event-stream", + "Authorization": "Bearer ", + "Content-Type": "application/json" + } + }, + "url": { + "nonStream": "https://seekai.cc/v1/chat/completions", + "stream": "https://seekai.cc/v1/chat/completions" + } + }, "sensenova": { "format": "openai", "headers": { diff --git a/tests/unit/provider-node-reserved-prefix.test.ts b/tests/unit/provider-node-reserved-prefix.test.ts index 26c2250753..a20918edab 100644 --- a/tests/unit/provider-node-reserved-prefix.test.ts +++ b/tests/unit/provider-node-reserved-prefix.test.ts @@ -179,7 +179,8 @@ test("shared set size includes live REGISTRY and retired Designer + Felo + Qwen // alias "gembiz" to the REGISTRY walk (406 → 408). // 2026-09-02: a keyless provider was removed at its operator's request, taking its id and // alias out of the REGISTRY walk (408 → 406). - assert.equal(RESERVED_PREFIX_COUNT, 406); + // #11786: SeekAi adds id "seekai" + alias "ska" (406 → 408). + assert.equal(RESERVED_PREFIX_COUNT, 408); }); test("isReservedProviderPrefix rejects non-string input", () => { diff --git a/tests/unit/providers-constants-split.test.ts b/tests/unit/providers-constants-split.test.ts index d6196816ca..6d4ae62c89 100644 --- a/tests/unit/providers-constants-split.test.ts +++ b/tests/unit/providers-constants-split.test.ts @@ -32,7 +32,8 @@ // volcengine-coding-plan (regional family) — both land at 233. // release/v3.8.51 adds Opper (gateways, #11629) and 1min.ai (gateways, #11631) — lands at 235; // Perplexity Agent API (#12103) makes it 236; -// UC Direct (#11513, uncensored.com metered Developer API) adds one frontier-labs entry — 237. +// UC Direct (#11513, uncensored.com metered Developer API) adds one frontier-labs entry — 237; +// SeekAi (#11786, QuantumNous New-API gateway) adds one gateways entry — 238. import { test } from "node:test"; import assert from "node:assert/strict"; @@ -61,12 +62,12 @@ test("barrel still exports every catalog + key helpers", () => { } }); -test("APIKEY_PROVIDERS merges the 6 family files into 237 entries (no loss / no dup)", async () => { +test("APIKEY_PROVIDERS merges the 6 family files into 238 entries (no loss / no dup)", async () => { const keys = Object.keys((P as Record).APIKEY_PROVIDERS); - assert.equal(keys.length, 237); - assert.equal(new Set(keys).size, 237, "duplicate keys after spread-merge"); + assert.equal(keys.length, 238); + assert.equal(new Set(keys).size, 238, "duplicate keys after spread-merge"); // the merged object's entry-count equals the sum of the 6 semantic family files; families are a - // strict partition (every provider in exactly one), so the sum must be exactly 237. + // strict partition (every provider in exactly one), so the sum must be exactly 238. const families: [string, string][] = [ ["gateways", "APIKEY_PROVIDERS_GATEWAYS"], ["frontier-labs", "APIKEY_PROVIDERS_FRONTIER"], @@ -86,7 +87,7 @@ test("APIKEY_PROVIDERS merges the 6 family files into 237 entries (no loss / no seen.add(k); } } - assert.equal(famTotal, 237, "families must partition all 237 providers"); + assert.equal(famTotal, 238, "families must partition all 238 providers"); }); test("AI_PROVIDERS Proxy aggregates all sections; lookups resolve", () => { diff --git a/tests/unit/seekai-provider.test.ts b/tests/unit/seekai-provider.test.ts new file mode 100644 index 0000000000..13f3ca512a --- /dev/null +++ b/tests/unit/seekai-provider.test.ts @@ -0,0 +1,67 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +const { APIKEY_PROVIDERS, AGGREGATOR_PROVIDER_IDS } = await import( + "../../src/shared/constants/providers.ts" +); +const { PROVIDER_ENDPOINTS } = await import("../../src/shared/constants/config.ts"); +const { REGISTRY: providerRegistry } = await import("../../open-sse/config/providerRegistry.ts"); +const { isValidModel } = await import("../../src/shared/constants/models.ts"); +const { DefaultExecutor, getExecutor } = await import("../../open-sse/executors/index.ts"); + +const SEEKAI_CHAT_URL = "https://seekai.cc/v1/chat/completions"; +const SEEKAI_MODELS_URL = "https://seekai.cc/v1/models"; + +test("#11786 seekai is registered as an API-key gateway provider", () => { + const entry = APIKEY_PROVIDERS.seekai; + assert.ok(entry, "APIKEY_PROVIDERS.seekai must be defined"); + assert.equal(entry.id, "seekai"); + assert.equal(entry.alias, "ska"); + assert.equal(entry.name, "SeekAi"); + assert.equal(entry.website, "https://seekai.cc"); + assert.equal(entry.passthroughModels, true); + assert.equal(entry.hasFree, true); + assert.equal(typeof entry.authHint, "string"); + assert.ok((entry.authHint as string).length > 0); + assert.equal(typeof entry.apiHint, "string"); + assert.ok((entry.apiHint as string).length > 0); +}); + +test("#11786 seekai website and hints carry no referral/aff query", () => { + const entry = APIKEY_PROVIDERS.seekai; + const haystack = [entry.website, entry.apiHint, entry.authHint, entry.freeNote] + .filter((value): value is string => typeof value === "string") + .join("\n"); + assert.equal(/[?&]aff=/.test(haystack), false); + assert.equal(haystack.includes("qR5U"), false); +}); + +test("#11786 seekai registry entry uses OpenAI format with bearer API-key auth", () => { + const entry = providerRegistry.seekai; + assert.ok(entry, "providerRegistry.seekai must be defined"); + assert.equal(entry.id, "seekai"); + assert.equal(entry.alias, "ska"); + assert.equal(entry.format, "openai"); + assert.equal(entry.executor, "default"); + assert.equal(entry.authType, "apikey"); + assert.equal(entry.authHeader, "bearer"); + assert.equal(entry.baseUrl, SEEKAI_CHAT_URL); + assert.equal(entry.modelsUrl, SEEKAI_MODELS_URL); + assert.equal(entry.passthroughModels, true); +}); + +test("#11786 seekai discovers models live via passthrough (no static seed list)", () => { + assert.deepEqual(providerRegistry.seekai.models, []); + assert.equal(providerRegistry.seekai.passthroughModels, true); +}); + +test("#11786 seekai accepts any model id via passthrough", () => { + assert.equal(isValidModel("seekai", "claude-sonnet-5"), true); + assert.equal(isValidModel("ska", "gpt-5.6"), true); +}); + +test("#11786 seekai is on the aggregator list and display endpoint", async () => { + assert.equal(AGGREGATOR_PROVIDER_IDS.has("seekai"), true); + assert.equal(PROVIDER_ENDPOINTS.seekai, SEEKAI_CHAT_URL); + assert.ok((await getExecutor("seekai")) instanceof DefaultExecutor); +}); From 831ea040c3a45cc24ecffeba2bd3834f360b1f69 Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Thu, 3 Sep 2026 11:47:50 -0400 Subject: [PATCH 040/143] feat(quota): Moonshot Open Platform balance and TPD lock for custom nodes (#12590) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check:provider-consistency` OK (273 entradas REGISTRY, **356** providers canônicos), `check-docs-counts-sync` exit 0 e **300/300** nos testes que a leva toca. O crescimento de arquivo que os PRs empilham uns sobre os outros foi rebaselinado num único registro datado (`_rebaseline_2026_09_03_houminxi_batch`), com a decomposição por arquivo: `providers/page.tsx` +18 (import CSV do #12504 + busca do #12495 no mesmo painel), `accountFallback.ts` +6 (o #12566 sobre o rebaseline que o #12590 já registrou — os dois tocam `checkFallbackError`) e `chatCore.ts` +3 (invalidez de cache de quota no 429 do #12325). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. --- AGENTS.md | 2 +- README.md | 2 +- config/quality/file-size-baseline.json | 11 +- docs/i18n/ar/llm.txt | 8 +- docs/i18n/az/llm.txt | 8 +- docs/i18n/bg/llm.txt | 8 +- docs/i18n/bn/llm.txt | 8 +- docs/i18n/cs/llm.txt | 8 +- docs/i18n/da/llm.txt | 8 +- docs/i18n/de/llm.txt | 8 +- docs/i18n/es/llm.txt | 8 +- docs/i18n/fa/llm.txt | 8 +- docs/i18n/fi/llm.txt | 8 +- docs/i18n/fr/llm.txt | 8 +- docs/i18n/gu/llm.txt | 8 +- docs/i18n/he/llm.txt | 8 +- docs/i18n/hi/llm.txt | 8 +- docs/i18n/hu/llm.txt | 8 +- docs/i18n/id/llm.txt | 8 +- docs/i18n/it/llm.txt | 8 +- docs/i18n/ja/llm.txt | 8 +- docs/i18n/ko/llm.txt | 8 +- docs/i18n/mr/llm.txt | 8 +- docs/i18n/ms/llm.txt | 8 +- docs/i18n/nl/llm.txt | 8 +- docs/i18n/no/llm.txt | 8 +- docs/i18n/phi/llm.txt | 8 +- docs/i18n/pl/llm.txt | 8 +- docs/i18n/pt-BR/llm.txt | 8 +- docs/i18n/pt/llm.txt | 8 +- docs/i18n/ro/llm.txt | 8 +- docs/i18n/ru/llm.txt | 8 +- docs/i18n/sk/llm.txt | 8 +- docs/i18n/sv/llm.txt | 8 +- docs/i18n/sw/llm.txt | 8 +- docs/i18n/ta/llm.txt | 8 +- docs/i18n/te/llm.txt | 8 +- docs/i18n/th/llm.txt | 8 +- docs/i18n/tr/llm.txt | 8 +- docs/i18n/uk-UA/llm.txt | 8 +- docs/i18n/ur/llm.txt | 8 +- docs/i18n/vi/llm.txt | 8 +- docs/i18n/zh-CN/llm.txt | 8 +- docs/i18n/zh-TW/llm.txt | 8 +- llm.txt | 8 +- open-sse/services/accountFallback.ts | 67 +++-- open-sse/services/dailyQuotaReset.ts | 145 +++++++++++ open-sse/services/moonshotQuotaFetcher.ts | 229 ++++++++++++++++++ open-sse/services/usage.ts | 9 + open-sse/services/usage/fetcherProviders.ts | 2 + .../services/usage/moonshotOpenPlatform.ts | 76 ++++++ open-sse/services/usage/supportedProviders.ts | 2 + .../modals/EditCompatibleNodeModal.tsx | 29 +++ .../usage/components/ProviderLimits/index.tsx | 5 +- .../(dashboard)/home/ProviderQuotaWidget.tsx | 5 +- src/app/api/provider-nodes/[id]/route.ts | 20 +- src/app/api/provider-nodes/route.ts | 35 +++ src/i18n/messages/ar.json | 4 + src/i18n/messages/az.json | 4 + src/i18n/messages/bg.json | 4 + src/i18n/messages/bn.json | 4 + src/i18n/messages/cs.json | 4 + src/i18n/messages/da.json | 4 + src/i18n/messages/de.json | 4 + src/i18n/messages/en.json | 4 + src/i18n/messages/es.json | 4 + src/i18n/messages/fa.json | 4 + src/i18n/messages/fi.json | 4 + src/i18n/messages/fr.json | 4 + src/i18n/messages/gu.json | 4 + src/i18n/messages/he.json | 4 + src/i18n/messages/hi.json | 4 + src/i18n/messages/hu.json | 4 + src/i18n/messages/id.json | 4 + src/i18n/messages/it.json | 4 + src/i18n/messages/ja.json | 4 + src/i18n/messages/ko.json | 4 + src/i18n/messages/mr.json | 4 + src/i18n/messages/ms.json | 4 + src/i18n/messages/nl.json | 4 + src/i18n/messages/no.json | 4 + src/i18n/messages/phi.json | 4 + src/i18n/messages/pl.json | 4 + src/i18n/messages/pt-BR.json | 4 + src/i18n/messages/pt.json | 4 + src/i18n/messages/ro.json | 4 + src/i18n/messages/ru.json | 4 + src/i18n/messages/sk.json | 4 + src/i18n/messages/sv.json | 4 + src/i18n/messages/sw.json | 4 + src/i18n/messages/ta.json | 4 + src/i18n/messages/te.json | 4 + src/i18n/messages/th.json | 4 + src/i18n/messages/tr.json | 4 + src/i18n/messages/uk-UA.json | 4 + src/i18n/messages/ur.json | 4 + src/i18n/messages/vi.json | 4 + src/i18n/messages/zh-CN.json | 4 + src/i18n/messages/zh-TW.json | 4 + src/instrumentation-node.ts | 16 ++ src/lib/db/migrationRunner.ts | 5 + .../172_provider_node_daily_quota_reset.sql | 6 + src/lib/db/providers/nodes.ts | 59 ++--- src/lib/usage/apiKeySelfService.ts | 20 +- src/lib/usage/providerLimits.ts | 21 +- src/shared/utils/classify429.ts | 11 + src/shared/utils/providerQuotaVisibility.ts | 14 +- src/shared/validation/schemas/provider.ts | 28 ++- src/sse/handlers/chat.ts | 16 ++ src/sse/services/auth.ts | 25 +- stryker.conf.json | 1 + tests/unit/account-fallback-service.test.ts | 48 ++++ tests/unit/api-key-self-service.test.ts | 36 +++ tests/unit/classify429.test.ts | 15 ++ tests/unit/daily-quota-reset.test.ts | 54 +++++ tests/unit/moonshot-open-platform.test.ts | 72 ++++++ tests/unit/moonshot-quota-fetcher.test.ts | 158 ++++++++++++ tests/unit/moonshot-quota-writeback.test.ts | 50 ++++ .../provider-node-daily-reset-schema.test.ts | 54 +++++ tests/unit/provider-quota-visibility.test.ts | 12 + tests/unit/qoder-usage-quota.test.ts | 9 + 121 files changed, 1626 insertions(+), 247 deletions(-) create mode 100644 open-sse/services/dailyQuotaReset.ts create mode 100644 open-sse/services/moonshotQuotaFetcher.ts create mode 100644 open-sse/services/usage/moonshotOpenPlatform.ts create mode 100644 src/lib/db/migrations/172_provider_node_daily_quota_reset.sql create mode 100644 tests/unit/daily-quota-reset.test.ts create mode 100644 tests/unit/moonshot-open-platform.test.ts create mode 100644 tests/unit/moonshot-quota-fetcher.test.ts create mode 100644 tests/unit/moonshot-quota-writeback.test.ts create mode 100644 tests/unit/provider-node-daily-reset-schema.test.ts diff --git a/AGENTS.md b/AGENTS.md index 1e90b6ceed..6e7ad18f2f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -56,7 +56,7 @@ Repository map and Reference Documentation sections below. | Translators | `open-sse/translator/` | Format conversion (OpenAI↔Claude↔Gemini) | | Transformer | `open-sse/transformer/` | Responses API ↔ Chat Completions | | Services | `open-sse/services/` | Combo routing, rate limits, caching, etc | -| Database | `src/lib/db/` | SQLite domain modules (168 migrations) | +| Database | `src/lib/db/` | SQLite domain modules (169 migrations) | | Domain/Policy | `src/domain/` | Policy engine, cost rules, fallback logic | | MCP Server | `open-sse/mcp-server/` | 110 tools (45 canonical + memory/skill/GitHub/pool/gamification/plugin/Notion/Obsidian/local-corpus/RTK modules), 3 transports (stdio / SSE / Streamable HTTP), 33 scopes | | A2A Server | `src/lib/a2a/` | JSON-RPC 2.0 agent protocol | diff --git a/README.md b/README.md index 119a590cb3..9814b0e62d 100644 --- a/README.md +++ b/README.md @@ -1244,7 +1244,7 @@ Métricas canônicas em 2026-08-24: **1.029 vídeos únicos** · **11.132.922 vi
RuntimeNode.js 22.x / 24.x LTS — >=22.22.2 <23 || >=24.0.0 <27 LanguageTypeScript 6.0 — 100% TypeScript across src/ and open-sse/ (zero any in core since v2.0) FrameworkNext.js 16 + React 19 + Tailwind CSS 4 - Databasebetter-sqlite3 (SQLite, WAL journaling) + LowDB (JSON legacy) — 122 domain modules, 168 migrations + Databasebetter-sqlite3 (SQLite, WAL journaling) + LowDB (JSON legacy) — 122 domain modules, 169 migrations MemorySQLite FTS5 full-text + int8-quantized vector embeddings, typed decay SchemasZod 4 — MCP tool I/O validation + API contracts ProtocolsMCP (stdio / HTTP / SSE) + A2A v0.3 (JSON-RPC 2.0 + SSE) diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index e5d0287a65..cfa43566b4 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -1,4 +1,5 @@ { + "_rebaseline_2026_09_03_moonshot_native_quota": "PR feat/moonshot-native-quota own growth on release/v3.8.51: src/lib/db/migrationRunner.ts 1201->1206 (+5, case 172 retroactive guard for daily_quota_reset_* columns); src/sse/handlers/chat.ts 2434->2450 (+16, registerMoonshotQuotaFetcher + startup node scan at the existing quota-fetcher registration chokepoint); src/sse/services/auth.ts 3427->3450 (+23, resolveDailyResetForProvider + dailyReset arg on checkFallbackError); open-sse/services/accountFallback.ts 2422->2461 (+39, compatible-node credits_exhausted carve-out + TPD node-clock lock); tests/unit/account-fallback-service.test.ts 2008->2056 (+48, TPD/empty-wallet cases). Wiring at existing chokepoints; Moonshot host predicates, daily reset clock, and the balance fetcher live in new leaves under cap. Covered by tests/unit/moonshot-*.test.ts + account-fallback-service.test.ts (135/135 focused).", "_rebaseline_2026_09_02_11786_seekai_provider": "PR #11786 (feat/11786-seekai-provider, closes #11786) own growth: src/shared/constants/providers/apikey/gateways.ts 1438->1458 (check-file-size split-newline=1459; the seekai APIKEY_PROVIDERS_GATEWAYS catalog entry plus authHint, additive data at the existing registry chokepoint, same god-file no-split rationale as prior gateways.ts rebaselines: #10987 logfare, #10531 freebuff). Covered by tests/unit/seekai-provider.test.ts.", "_rebaseline_2026_09_02_12325_generic_429_invalidate": "PR #12325 own growth: open-sse/handlers/chatCore.ts 5946->5955 (+9 = the non-Codex 429 else-if that drops the generic quota wrapper and stamps force-refresh, plus a source-regex breadcrumb). Irreducible call-site wiring next to the existing Codex 429 invalidateCodexQuotaCache branch; not extractable without splitting handleChatCore mid-response. Covered by tests/unit/generic-quota-fetcher.test.ts (31/31) and tests/unit/antigravity-429-quota-cooldown.test.ts.", "_rebaseline_2026_09_02_12429_wreq_migration_suite": "PR #12429 (wreq-js web-cookie transport): new test file tests/unit/tls-client-wreq-migration.test.ts at 1374 lines, above the 1200 new-file testCap. Frozen rather than split: it is the single cohesive regression suite for the transport migration (31 cases covering streaming, fragmented EOF sentinels, proxy isolation, first-byte and hard deadlines, binary responses and cancellation), and the cases share the native-transport harness the file sets up once. Splitting it during a merge would duplicate that harness across files for no coverage gain. Entered at the exact LOC, so it can only ratchet down from here.", @@ -206,7 +207,7 @@ "_rebaseline_basered_codebuddy_cn": "Base-red fix (#4664 CodeBuddy CN): oauth-providers-config.test.ts 867->870 (+3) to align the EXPECTED provider list/config with the codebuddy-cn provider that #4664 added to the registry without updating this test (it asserts 'exactly once').", "_rebaseline_pr4613_compatible_provider_groups": "Reconcile #4613 already-merged growth: providers-page-utils.test.ts 1004->1052 (+48, buildCompatibleProviderGroups partition unit test). Fast-gate PR->release does not run check:file-size, so this surfaced post-merge.", "tests/integration/chat-pipeline.test.ts": 1644, - "tests/unit/account-fallback-service.test.ts": 2008, + "tests/unit/account-fallback-service.test.ts": 2056, "tests/unit/batch_api.test.ts": 1345, "tests/unit/cc-compatible-provider.test.ts": 1225, "tests/unit/chatcore-translation-paths.test.ts": 3447, @@ -419,7 +420,7 @@ "open-sse/handlers/search.ts": 1789, "open-sse/mcp-server/schemas/tools.ts": 1621, "open-sse/mcp-server/server.ts": 1572, - "open-sse/services/accountFallback.ts": 2422, + "open-sse/services/accountFallback.ts": 2461, "open-sse/services/adobeFireflyBrowserLogin.ts": 1401, "open-sse/services/combo.ts": 4023, "open-sse/translator/response/openai-responses.ts": 1466, @@ -447,14 +448,14 @@ "src/app/docs/lib/openapi.generated.ts": 1347, "src/lib/db/apiKeys.ts": 1610, "src/lib/db/core.ts": 1745, - "src/lib/db/migrationRunner.ts": 1201, + "src/lib/db/migrationRunner.ts": 1206, "src/lib/tailscaleTunnel.ts": 1208, "src/lib/tokenHealthCheck.ts": 1218, "src/shared/components/RequestLoggerV2.tsx": 1718, "src/shared/constants/providers/apikey/gateways.ts": 1459, "src/shared/services/cliRuntime.ts": 1296, - "src/sse/handlers/chat.ts": 2424, - "src/sse/services/auth.ts": 3427, + "src/sse/handlers/chat.ts": 2450, + "src/sse/services/auth.ts": 3450, "tests/unit/account-fallback-service.test.ts": 2453, "tests/unit/provider-validation-specialty.test.ts": 4656 }, diff --git a/docs/i18n/ar/llm.txt b/docs/i18n/ar/llm.txt index 603773977a..9f405eda91 100644 --- a/docs/i18n/ar/llm.txt +++ b/docs/i18n/ar/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/az/llm.txt b/docs/i18n/az/llm.txt index 3af7361352..74a600ba34 100644 --- a/docs/i18n/az/llm.txt +++ b/docs/i18n/az/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/bg/llm.txt b/docs/i18n/bg/llm.txt index c510c85d7e..0db6d8a4ce 100644 --- a/docs/i18n/bg/llm.txt +++ b/docs/i18n/bg/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/bn/llm.txt b/docs/i18n/bn/llm.txt index 54b9ca1d9a..a7417f642e 100644 --- a/docs/i18n/bn/llm.txt +++ b/docs/i18n/bn/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/cs/llm.txt b/docs/i18n/cs/llm.txt index 4dd9012d51..c8b5c31922 100644 --- a/docs/i18n/cs/llm.txt +++ b/docs/i18n/cs/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/da/llm.txt b/docs/i18n/da/llm.txt index d29af5f81c..3d9663c28f 100644 --- a/docs/i18n/da/llm.txt +++ b/docs/i18n/da/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/de/llm.txt b/docs/i18n/de/llm.txt index 0002d85510..0c8572e4f4 100644 --- a/docs/i18n/de/llm.txt +++ b/docs/i18n/de/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/es/llm.txt b/docs/i18n/es/llm.txt index c423fdd245..75a6488e37 100644 --- a/docs/i18n/es/llm.txt +++ b/docs/i18n/es/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/fa/llm.txt b/docs/i18n/fa/llm.txt index d8b83418ea..cdcccf2fc8 100644 --- a/docs/i18n/fa/llm.txt +++ b/docs/i18n/fa/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/fi/llm.txt b/docs/i18n/fi/llm.txt index b91777ac80..66d62309f4 100644 --- a/docs/i18n/fi/llm.txt +++ b/docs/i18n/fi/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/fr/llm.txt b/docs/i18n/fr/llm.txt index 7e396ec1e9..dabd29e87f 100644 --- a/docs/i18n/fr/llm.txt +++ b/docs/i18n/fr/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/gu/llm.txt b/docs/i18n/gu/llm.txt index 75561762bf..21dad6d8f9 100644 --- a/docs/i18n/gu/llm.txt +++ b/docs/i18n/gu/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/he/llm.txt b/docs/i18n/he/llm.txt index e8dab2b860..4174e2aa0f 100644 --- a/docs/i18n/he/llm.txt +++ b/docs/i18n/he/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/hi/llm.txt b/docs/i18n/hi/llm.txt index 3df2c700b6..b08baf5189 100644 --- a/docs/i18n/hi/llm.txt +++ b/docs/i18n/hi/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/hu/llm.txt b/docs/i18n/hu/llm.txt index bd0667630c..13109da6fc 100644 --- a/docs/i18n/hu/llm.txt +++ b/docs/i18n/hu/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/id/llm.txt b/docs/i18n/id/llm.txt index 0cce4bc272..77df7b4c16 100644 --- a/docs/i18n/id/llm.txt +++ b/docs/i18n/id/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/it/llm.txt b/docs/i18n/it/llm.txt index 0763f0b728..b479c853cd 100644 --- a/docs/i18n/it/llm.txt +++ b/docs/i18n/it/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/ja/llm.txt b/docs/i18n/ja/llm.txt index 3dbb124940..a1528c646b 100644 --- a/docs/i18n/ja/llm.txt +++ b/docs/i18n/ja/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/ko/llm.txt b/docs/i18n/ko/llm.txt index 0d44a5b222..4d8508fd12 100644 --- a/docs/i18n/ko/llm.txt +++ b/docs/i18n/ko/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/mr/llm.txt b/docs/i18n/mr/llm.txt index 9410d228f5..8ba71d98f5 100644 --- a/docs/i18n/mr/llm.txt +++ b/docs/i18n/mr/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/ms/llm.txt b/docs/i18n/ms/llm.txt index e610a66e29..3903f94cf1 100644 --- a/docs/i18n/ms/llm.txt +++ b/docs/i18n/ms/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/nl/llm.txt b/docs/i18n/nl/llm.txt index a4364f5ac2..8db8eb985b 100644 --- a/docs/i18n/nl/llm.txt +++ b/docs/i18n/nl/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/no/llm.txt b/docs/i18n/no/llm.txt index 7daa5caa07..e635301c86 100644 --- a/docs/i18n/no/llm.txt +++ b/docs/i18n/no/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/phi/llm.txt b/docs/i18n/phi/llm.txt index 208e4a1398..b760a56857 100644 --- a/docs/i18n/phi/llm.txt +++ b/docs/i18n/phi/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/pl/llm.txt b/docs/i18n/pl/llm.txt index 79f3f32583..e683cdb76c 100644 --- a/docs/i18n/pl/llm.txt +++ b/docs/i18n/pl/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/pt-BR/llm.txt b/docs/i18n/pt-BR/llm.txt index e795080292..49a62961e7 100644 --- a/docs/i18n/pt-BR/llm.txt +++ b/docs/i18n/pt-BR/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/pt/llm.txt b/docs/i18n/pt/llm.txt index ffa62b14e6..9362022f86 100644 --- a/docs/i18n/pt/llm.txt +++ b/docs/i18n/pt/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/ro/llm.txt b/docs/i18n/ro/llm.txt index 7a6f5067a9..fd21c2042b 100644 --- a/docs/i18n/ro/llm.txt +++ b/docs/i18n/ro/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/ru/llm.txt b/docs/i18n/ru/llm.txt index ed8415886d..d68314887e 100644 --- a/docs/i18n/ru/llm.txt +++ b/docs/i18n/ru/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/sk/llm.txt b/docs/i18n/sk/llm.txt index 11385eade3..666d1a570e 100644 --- a/docs/i18n/sk/llm.txt +++ b/docs/i18n/sk/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/sv/llm.txt b/docs/i18n/sv/llm.txt index 5a0f5b5c80..74238247ae 100644 --- a/docs/i18n/sv/llm.txt +++ b/docs/i18n/sv/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/sw/llm.txt b/docs/i18n/sw/llm.txt index ef792bc7e7..5f129a8263 100644 --- a/docs/i18n/sw/llm.txt +++ b/docs/i18n/sw/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/ta/llm.txt b/docs/i18n/ta/llm.txt index 2c0694b490..d42fa346a0 100644 --- a/docs/i18n/ta/llm.txt +++ b/docs/i18n/ta/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/te/llm.txt b/docs/i18n/te/llm.txt index 29ff10874a..6ca040018f 100644 --- a/docs/i18n/te/llm.txt +++ b/docs/i18n/te/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/th/llm.txt b/docs/i18n/th/llm.txt index 4096423b6d..9884ab3b21 100644 --- a/docs/i18n/th/llm.txt +++ b/docs/i18n/th/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/tr/llm.txt b/docs/i18n/tr/llm.txt index b83be6d908..77b9eadb1b 100644 --- a/docs/i18n/tr/llm.txt +++ b/docs/i18n/tr/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/uk-UA/llm.txt b/docs/i18n/uk-UA/llm.txt index f3cf580495..3a7bc01f36 100644 --- a/docs/i18n/uk-UA/llm.txt +++ b/docs/i18n/uk-UA/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/ur/llm.txt b/docs/i18n/ur/llm.txt index 9efbef2f56..1cdd0e6bb5 100644 --- a/docs/i18n/ur/llm.txt +++ b/docs/i18n/ur/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/vi/llm.txt b/docs/i18n/vi/llm.txt index eef12beee4..3e28435b0a 100644 --- a/docs/i18n/vi/llm.txt +++ b/docs/i18n/vi/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/zh-CN/llm.txt b/docs/i18n/zh-CN/llm.txt index f1c4a2b574..d11163a83a 100644 --- a/docs/i18n/zh-CN/llm.txt +++ b/docs/i18n/zh-CN/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/docs/i18n/zh-TW/llm.txt b/docs/i18n/zh-TW/llm.txt index b7b1342bf5..56256a53da 100644 --- a/docs/i18n/zh-TW/llm.txt +++ b/docs/i18n/zh-TW/llm.txt @@ -18,7 +18,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -128,7 +128,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -393,7 +393,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -437,7 +437,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/llm.txt b/llm.txt index 13d3c28e78..3748b08432 100644 --- a/llm.txt +++ b/llm.txt @@ -14,7 +14,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo - **Runtime:** Node.js `>=22.22.2 <23 || >=24.0.0 <27`, ES Modules (`"type": "module"`) - **Framework:** Next.js 16 (App Router) with TypeScript 6 -- **Database:** SQLite via better-sqlite3 (local, zero-config, 168 migrations) +- **Database:** SQLite via better-sqlite3 (local, zero-config, 169 migrations) - **State management:** Zustand (client), SQLite (server persistence) - **UI:** React 19, Tailwind CSS 4, Recharts for analytics, @lobehub/icons for 130+ provider SVG icons - **Auth:** OAuth 2.0 (PKCE) for providers, bcrypt for local user auth @@ -124,7 +124,7 @@ OmniRoute solves the problem of managing multiple AI provider subscriptions, quo │ │ │ ├── secrets.ts # Secrets management │ │ │ ├── stateReset.ts # State reset utilities │ │ │ ├── migrationRunner.ts # Schema migration runner -│ │ │ └── migrations/ # 168 versioned SQL migration files +│ │ │ └── migrations/ # 169 versioned SQL migration files │ │ ├── evals/ # Eval runner and scheduler │ │ ├── memory/ # Persistent conversational memory │ │ │ ├── extraction.ts # Memory extraction from conversations @@ -389,7 +389,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 8. **ProviderIcon component:** Unified icon system using `@lobehub/icons` (130+ SVG) with PNG fallback and generic icon fallback chain. Used on providers, dashboard, and agents pages. -9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 168 SQL migrations. +9. **DB architecture:** `localDb.ts` is a re-export layer only — real logic lives in 122 `src/lib/db/` modules with 169 SQL migrations. 10. **Upstream headers:** Custom headers merged in executors after default auth; same header name replaces executor value. Forbidden header names in `src/shared/constants/upstreamHeaders.ts`. @@ -433,7 +433,7 @@ diagnostics) plus **memory**, **skill**, **agentSkill**, **githubSkill**, **pool 4. **Environment variables:** All configuration is in `.env` (from `.env.example`). Key vars: `PORT`, `NEXT_PUBLIC_BASE_URL`, `API_KEY`, `ADMIN_PASSWORD`. -5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 168 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. +5. **Database layer:** Operations go through `src/lib/db/` modules (122 domain-specific files, 169 migrations). `localDb.ts` is re-exports only — add new functions to the proper `db/*.ts` module. 6. **Tests** use Node.js built-in test runner + Vitest. Run `npm test`. Vitest for MCP/autoCombo (`npm run test:vitest`). Playwright for E2E (`npm run test:e2e`). Coverage gate: ratchet vs `quality-baseline.json`, absolute floor 60% statements/lines/functions/branches. diff --git a/open-sse/services/accountFallback.ts b/open-sse/services/accountFallback.ts index f5bd87c1ed..c6465321f9 100644 --- a/open-sse/services/accountFallback.ts +++ b/open-sse/services/accountFallback.ts @@ -64,6 +64,8 @@ import { parseDelayString, MAX_SHORT_RETRY_HINT_MS, } from "./retryAfterJson.ts"; +import { isMoonshotAccountBalanceExhausted } from "./usage/moonshotOpenPlatform.ts"; +import { isTpdRateLimit, resolveTpdCooldownMs } from "./dailyQuotaReset.ts"; // Pre-compiled regex constants for hot-path retry parsing (avoid per-call compilation) const RETRY_AFTER_RE = /retry\s+after\s+(\d+)\s*s/i; @@ -1605,7 +1607,8 @@ export function isDailyQuotaExhausted(errorText: string): boolean { return ( lower.includes("today's quota") || lower.includes("daily quota") || - lower.includes("try again tomorrow") + lower.includes("try again tomorrow") || + lower.includes("tpd rate limit") ); } @@ -1651,7 +1654,12 @@ export function checkFallbackError( headers: Headers | Record | null = null, profileOverride: ProviderProfile | null = null, structuredError?: { code?: string | null; type?: string | null } | null, - rotation?: { account?: unknown } | null + rotation?: { account?: unknown } | null, + dailyReset?: { + timezone?: unknown; + hour?: unknown; + nowMs?: number; + } | null, ): { shouldFallback: boolean; cooldownMs: number; @@ -1934,8 +1942,13 @@ export function checkFallbackError( } } - // T10 (sub2api #1169) + #8247: credits/quota exhausted; *-compatible-* nicknames stay model-scoped. - if (shouldUseQuotaSignal && isCreditsExhausted(errorStr) && !isCompatibleProvider(provider)) { + // T10 (sub2api #1169) + #8247: credits/quota exhausted; *-compatible-* nicknames stay model-scoped + // unless the body is an account-level Open Platform empty wallet. + if ( + shouldUseQuotaSignal && + isCreditsExhausted(errorStr) && + (!isCompatibleProvider(provider) || isMoonshotAccountBalanceExhausted(errorStr)) + ) { return { shouldFallback: true, cooldownMs: COOLDOWN_MS.paymentRequired ?? 3600 * 1000, // 1h cooldown @@ -1944,17 +1957,43 @@ export function checkFallbackError( }; } - // Daily quota exhausted — lock model until tomorrow + // Daily quota exhausted. TPD uses the node clock / header; other daily + // quota text still uses getMsUntilTomorrow. TPD without either is not a + // host-midnight lock — fall through to short 429. if (shouldUseQuotaSignal && isDailyQuotaExhausted(errorStr)) { - const msUntilTomorrow = getMsUntilTomorrow(); - // Cap at 24 hours to handle timezone edge cases - const cooldownMs = Math.min(msUntilTomorrow, 24 * 60 * 60 * 1000); - return { - shouldFallback: true, - cooldownMs, - reason: RateLimitReason.QUOTA_EXHAUSTED, - dailyQuotaExhausted: true, - }; + if (isTpdRateLimit(errorStr)) { + const headerResetAtMs = parseResetFromHeaders(headers); + const tpdMs = resolveTpdCooldownMs(errorStr, { + timezone: dailyReset?.timezone, + hour: dailyReset?.hour, + nowMs: dailyReset?.nowMs, + headerResetAtMs, + }); + if (tpdMs == null) { + // no clock, no header — short 429, do not guess midnight + console.warn( + "[accountFallback] TPD 429 without node daily-reset clock or Reset header; using short cooldown", + { provider }, + ); + } else { + return { + shouldFallback: true, + cooldownMs: tpdMs, + reason: RateLimitReason.QUOTA_EXHAUSTED, + dailyQuotaExhausted: true, + }; + } + } else { + const msUntilTomorrow = getMsUntilTomorrow(); + // Cap at 24 hours to handle timezone edge cases + const cooldownMs = Math.min(msUntilTomorrow, 24 * 60 * 60 * 1000); + return { + shouldFallback: true, + cooldownMs, + reason: RateLimitReason.QUOTA_EXHAUSTED, + dailyQuotaExhausted: true, + }; + } } // Issue #2321 (5h subscription quota) + Issue #3709 (ollama-cloud weekly diff --git a/open-sse/services/dailyQuotaReset.ts b/open-sse/services/dailyQuotaReset.ts new file mode 100644 index 0000000000..a5108d132f --- /dev/null +++ b/open-sse/services/dailyQuotaReset.ts @@ -0,0 +1,145 @@ +/** + * Node-level daily quota reset clock. + * + * TPD cooldown endpoint: operator-configured IANA timezone + local hour. + * No default timezone. Do not call getMsUntilTomorrow() from here. + */ + +export function isValidIanaTimeZone(tz: string): boolean { + if (typeof tz !== "string" || tz.trim() === "") return false; + try { + new Intl.DateTimeFormat("en-US", { timeZone: tz.trim() }).format(); + return true; + } catch { + return false; + } +} + +export function isValidResetHour(hour: unknown): hour is number { + return typeof hour === "number" && Number.isInteger(hour) && hour >= 0 && hour <= 23; +} + +export function nodeDailyResetConfigured(timezone: unknown, hour: unknown): boolean { + return typeof timezone === "string" && isValidIanaTimeZone(timezone) && isValidResetHour(hour); +} + +type ZonedParts = { + year: number; + month: number; + day: number; + hour: number; + minute: number; + second: number; +}; + +function zonedParts(ms: number, timeZone: string): ZonedParts { + const fmt = new Intl.DateTimeFormat("en-US", { + timeZone, + hourCycle: "h23", + year: "numeric", + month: "2-digit", + day: "2-digit", + hour: "2-digit", + minute: "2-digit", + second: "2-digit", + }); + const bag: Record = {}; + for (const part of fmt.formatToParts(new Date(ms))) { + if (part.type !== "literal") bag[part.type] = part.value; + } + return { + year: Number(bag.year), + month: Number(bag.month), + day: Number(bag.day), + hour: Number(bag.hour), + minute: Number(bag.minute), + second: Number(bag.second), + }; +} + +function addCalendarDay(year: number, month: number, day: number): { + year: number; + month: number; + day: number; +} { + const utc = Date.UTC(year, month - 1, day + 1); + const dt = new Date(utc); + return { year: dt.getUTCFullYear(), month: dt.getUTCMonth() + 1, day: dt.getUTCDate() }; +} + +/** Convert wall-clock time in `timeZone` to epoch ms. */ +function zonedLocalToUtc( + year: number, + month: number, + day: number, + hour: number, + minute: number, + second: number, + timeZone: string, +): number { + const wanted = Date.UTC(year, month - 1, day, hour, minute, second); + let guess = wanted; + for (let i = 0; i < 4; i++) { + const p = zonedParts(guess, timeZone); + const asIfUtc = Date.UTC(p.year, p.month - 1, p.day, p.hour, p.minute, p.second); + const delta = asIfUtc - wanted; + if (delta === 0) return guess; + guess -= delta; + } + return guess; +} + +/** + * Next local `hour:00:00` in `timezone` strictly after `nowMs`. + * If now lands exactly on that instant, return the following cycle. + */ +export function nextDailyResetAtMs(timezone: string, hour: number, nowMs: number): number { + const now = zonedParts(nowMs, timezone); + let date = { year: now.year, month: now.month, day: now.day }; + let next = zonedLocalToUtc(date.year, date.month, date.day, hour, 0, 0, timezone); + if (next <= nowMs) { + date = addCalendarDay(date.year, date.month, date.day); + next = zonedLocalToUtc(date.year, date.month, date.day, hour, 0, 0, timezone); + } + return next; +} + +export function parseTpdLimitFromText(text: string): number | null { + const m = /limit:\s*(\d+)/i.exec(text); + if (!m) return null; + const n = Number(m[1]); + return Number.isFinite(n) ? n : null; +} + +export function isTpdRateLimit(errorText: string | null | undefined): boolean { + return String(errorText || "") + .toLowerCase() + .includes("tpd rate limit"); +} + +export type TpdCooldownOptions = { + timezone?: unknown; + hour?: unknown; + nowMs?: number; + headerResetAtMs?: number | null; +}; + +/** + * Cooldown for a TPD 429. Header reset wins; else the node clock. + * Both missing → null (caller uses short 429, does not guess midnight). + */ +export function resolveTpdCooldownMs( + errorText: string | null | undefined, + options: TpdCooldownOptions = {}, +): number | null { + if (!isTpdRateLimit(errorText)) return null; + const now = options.nowMs ?? Date.now(); + if (typeof options.headerResetAtMs === "number" && options.headerResetAtMs > now) { + return options.headerResetAtMs - now; + } + if (typeof options.timezone === "string" && isValidResetHour(options.hour)) { + if (!nodeDailyResetConfigured(options.timezone, options.hour)) return null; + return nextDailyResetAtMs(options.timezone, options.hour, now) - now; + } + return null; +} diff --git a/open-sse/services/moonshotQuotaFetcher.ts b/open-sse/services/moonshotQuotaFetcher.ts new file mode 100644 index 0000000000..df80fe11b3 --- /dev/null +++ b/open-sse/services/moonshotQuotaFetcher.ts @@ -0,0 +1,229 @@ +/** + * moonshotQuotaFetcher.ts — Moonshot Open Platform balance quota fetcher + * + * GET {origin}/v1/users/me/balance + * { code: 0, data: { available_balance, voucher_balance, cash_balance } } + * + * Origin comes from the connection baseUrl (api.moonshot.cn or api.moonshot.ai). + * Do not hardcode .ai as a fallback for .cn keys. + * + * Cache: 60s in-memory. Registration: registerMoonshotQuotaFetcher() at startup. + */ + +import { toNumber } from "@/shared/utils/numeric"; +import { registerQuotaFetcher, type QuotaInfo } from "./quotaPreflight.ts"; +import { registerMonitorFetcher } from "./quotaMonitor.ts"; +import { throttleQuotaFetch } from "./quotaFetchThrottle.ts"; +import { + isMoonshotOpenPlatformConnection, + moonshotBalanceUrl, + resolveMoonshotOrigin, +} from "./usage/moonshotOpenPlatform.ts"; +import type { UsageQuota } from "./usage/quota.ts"; + +const CACHE_TTL_MS = 60_000; + +export interface MoonshotQuota extends QuotaInfo { + availableBalance: number; + voucherBalance: number; + cashBalance: number; + origin: string; + limitReached: boolean; +} + +interface CacheEntry { + quota: MoonshotQuota; + fetchedAt: number; +} + +const quotaCache = new Map(); + +const _cacheCleanup = setInterval(() => { + const now = Date.now(); + for (const [key, entry] of quotaCache) { + if (now - entry.fetchedAt > CACHE_TTL_MS * 5) { + quotaCache.delete(key); + } + } +}, 5 * 60_000); + +if (typeof _cacheCleanup === "object" && "unref" in _cacheCleanup) { + (_cacheCleanup as { unref?: () => void }).unref?.(); +} + +function toRecord(value: unknown): Record { + return value && typeof value === "object" && !Array.isArray(value) + ? (value as Record) + : {}; +} + +function parseMoonshotQuotaResponse(data: unknown, origin: string): MoonshotQuota | null { + const obj = toRecord(data); + const code = obj.code; + if (code !== 0 && code !== undefined) return null; + const payload = toRecord(obj.data); + if (!("available_balance" in payload) && !("availableBalance" in payload)) return null; + const availableBalance = toNumber(payload.available_balance ?? payload.availableBalance, 0); + const voucherBalance = toNumber(payload.voucher_balance ?? payload.voucherBalance, 0); + const cashBalance = toNumber(payload.cash_balance ?? payload.cashBalance, 0); + const limitReached = availableBalance <= 0; + const percentUsed = limitReached ? 1 : 0; + return { + used: percentUsed * 100, + total: 100, + percentUsed, + resetAt: null, + availableBalance, + voucherBalance, + cashBalance, + origin, + limitReached, + windows: { balance: { percentUsed, resetAt: null } }, + }; +} + +function connectionApiKey(connection?: Record): string | null { + const apiKey = connection?.apiKey; + return typeof apiKey === "string" && apiKey.trim().length > 0 ? apiKey : null; +} + +export async function fetchMoonshotQuota( + connectionId: string, + connection?: Record +): Promise { + const cached = quotaCache.get(connectionId); + if (cached && Date.now() - cached.fetchedAt < CACHE_TTL_MS) { + return cached.quota; + } + + const apiKey = connectionApiKey(connection); + if (!apiKey) return null; + + const origin = resolveMoonshotOrigin({ + provider: typeof connection?.provider === "string" ? connection.provider : undefined, + providerSpecificData: connection?.providerSpecificData, + }); + if (!origin) return null; + + const url = moonshotBalanceUrl(origin); + const authHeader = ["Bearer", apiKey].join(" "); + + try { + await throttleQuotaFetch(); + const response = await fetch(url, { + method: "GET", + headers: { + Authorization: authHeader, + "Content-Type": "application/json", + Accept: "application/json", + }, + signal: AbortSignal.timeout(8_000), + }); + + if (response.status === 401 || response.status === 403) { + quotaCache.delete(connectionId); + return null; + } + if (!response.ok) return null; + + const data = await response.json(); + const quota = parseMoonshotQuotaResponse(data, origin); + if (!quota) return null; + quotaCache.set(connectionId, { quota, fetchedAt: Date.now() }); + return quota; + } catch { + return null; + } +} + +export function invalidateMoonshotQuotaCache(connectionId: string): void { + quotaCache.delete(connectionId); +} + +export type MoonshotUsageConnection = { + id?: string; + provider?: string; + apiKey?: string; + providerSpecificData?: unknown; +}; + +export async function getMoonshotOpenPlatformUsage( + connection: MoonshotUsageConnection +): Promise<{ + plan?: string; + quotas?: Record; + message?: string; + limitReached?: boolean; +}> { + const origin = resolveMoonshotOrigin(connection); + if (!origin) { + return { message: "Not a Moonshot Open Platform connection." }; + } + const quota = (await fetchMoonshotQuota(connection.id || "moonshot", { + apiKey: connection.apiKey, + provider: connection.provider, + providerSpecificData: connection.providerSpecificData, + })) as MoonshotQuota | null; + if (!quota) { + return { message: "Moonshot API key not available. Add a key to view usage." }; + } + const domestic = origin.includes("moonshot.cn"); + return { + plan: domestic ? "Kimi 开放平台(国内)" : "Kimi Open Platform", + quotas: buildMoonshotBalanceQuotas(quota, domestic ? "CNY" : "USD"), + limitReached: quota.limitReached, + }; +} + +function balanceQuota( + remaining: number, + remainingPercentage: number, + currency: string +): UsageQuota { + return { + used: 0, + total: 0, + remaining, + remainingPercentage, + resetAt: null, + unlimited: true, + currency, + }; +} + +function buildMoonshotBalanceQuotas( + quota: MoonshotQuota, + currency: string +): Record { + return { + available: balanceQuota(quota.availableBalance, quota.limitReached ? 0 : 100, currency), + voucher: balanceQuota(quota.voucherBalance, 100, currency), + cash: balanceQuota(quota.cashBalance, 100, currency), + }; +} + +export function registerMoonshotQuotaFetcher(): void { + registerQuotaFetcher("moonshot", fetchMoonshotQuota); + registerQuotaFetcher("kimi", fetchMoonshotQuota); + registerMonitorFetcher("moonshot", fetchMoonshotQuota); + registerMonitorFetcher("kimi", fetchMoonshotQuota); +} + +export function registerMoonshotFetchersForNodes( + nodes: Array<{ id?: string | null; prefix?: string | null; baseUrl?: string | null }> +): void { + for (const node of nodes) { + const origin = resolveMoonshotOrigin({}, node.baseUrl); + if (!origin) continue; + if (typeof node.id === "string" && node.id) { + registerQuotaFetcher(node.id, fetchMoonshotQuota); + registerMonitorFetcher(node.id, fetchMoonshotQuota); + } + if (typeof node.prefix === "string" && node.prefix) { + registerQuotaFetcher(node.prefix, fetchMoonshotQuota); + registerMonitorFetcher(node.prefix, fetchMoonshotQuota); + } + } +} + +export { isMoonshotOpenPlatformConnection }; diff --git a/open-sse/services/usage.ts b/open-sse/services/usage.ts index ea240340ea..62e96ce1bc 100644 --- a/open-sse/services/usage.ts +++ b/open-sse/services/usage.ts @@ -61,6 +61,8 @@ import { getQoderUsage, parseQoderUserStatusUsage } from "./usage/qoder.ts"; export { parseQoderUserStatusUsage } from "./usage/qoder.ts"; import { getOpencodeUsage } from "./usage/opencode.ts"; import { getDeepseekUsage } from "./usage/deepseek.ts"; +import { getMoonshotOpenPlatformUsage } from "./moonshotQuotaFetcher.ts"; +import { isMoonshotOpenPlatformConnection } from "./usage/moonshotOpenPlatform.ts"; import { getDevinCliUsage } from "./usage/devinCli.ts"; import { getBailianCodingPlanUsage } from "./usage/bailian.ts"; import { getVertexUsage } from "./usage/vertex.ts"; @@ -111,6 +113,10 @@ export async function getUsageForProvider( ) { const { id, provider, accessToken, apiKey, providerSpecificData, projectId, email } = connection; + if (isMoonshotOpenPlatformConnection(connection)) { + return await getMoonshotOpenPlatformUsage(connection); + } + switch (provider) { case "github": return await getGitHubUsage(accessToken, providerSpecificData); @@ -168,6 +174,9 @@ export async function getUsageForProvider( return await getNanoGptUsage(apiKey || ""); case "deepseek": return await getDeepseekUsage(id || "", apiKey || ""); + case "moonshot": + case "kimi": + return await getMoonshotOpenPlatformUsage(connection); case "openrouter": return await getOpenrouterUsage(id || "", apiKey || "", providerSpecificData); case "opencode": diff --git a/open-sse/services/usage/fetcherProviders.ts b/open-sse/services/usage/fetcherProviders.ts index bfcc5da710..05a3b47225 100644 --- a/open-sse/services/usage/fetcherProviders.ts +++ b/open-sse/services/usage/fetcherProviders.ts @@ -45,6 +45,8 @@ export const USAGE_FETCHER_PROVIDERS = [ "qwen-cloud-token-plan", "nanogpt", "deepseek", + "moonshot", + "kimi", "opencode", "opencode-zen", "xiaomi-mimo", diff --git a/open-sse/services/usage/moonshotOpenPlatform.ts b/open-sse/services/usage/moonshotOpenPlatform.ts new file mode 100644 index 0000000000..a652aacb36 --- /dev/null +++ b/open-sse/services/usage/moonshotOpenPlatform.ts @@ -0,0 +1,76 @@ +/** + * Moonshot Open Platform host recognition. + * + * Distinguishes prepaid Open Platform keys (api.moonshot.cn / api.moonshot.ai) + * from Kimi Coding Plan (api.kimi.com/coding). Custom compatible nodes are + * identified by baseUrl host, not by provider id (those ids are uuids). + */ + +import { moonshotProvider } from "../../config/providers/registry/moonshot/index.ts"; +import { kimiProvider } from "../../config/providers/registry/kimi/index.ts"; + +export const MOONSHOT_OPEN_PLATFORM_HOSTS: ReadonlySet = new Set([ + "api.moonshot.cn", + "api.moonshot.ai", +]); + +export type MoonshotOriginConnection = { + provider?: string; + providerSpecificData?: unknown; +}; + +function asRecord(value: unknown): Record { + return value && typeof value === "object" && !Array.isArray(value) + ? (value as Record) + : {}; +} + +export function parseMoonshotOrigin(baseUrl: string | null | undefined): string | null { + if (typeof baseUrl !== "string" || baseUrl.trim() === "") return null; + let url: URL; + try { + url = new URL(baseUrl.trim()); + } catch { + return null; + } + if (url.protocol !== "https:" && url.protocol !== "http:") return null; + const host = url.hostname.toLowerCase(); + if (!MOONSHOT_OPEN_PLATFORM_HOSTS.has(host)) return null; + const port = url.port ? `:${url.port}` : ""; + return `${url.protocol}//${host}${port}`; +} + +export function moonshotBalanceUrl(origin: string): string { + return `${origin}/v1/users/me/balance`; +} + +function registryDefaultOrigin(provider: string | undefined): string | null { + if (provider === "moonshot") return parseMoonshotOrigin(moonshotProvider.baseUrl); + if (provider === "kimi") return parseMoonshotOrigin(kimiProvider.baseUrl); + return null; +} + +export function resolveMoonshotOrigin( + connection: MoonshotOriginConnection, + nodeBaseUrl?: string | null, +): string | null { + const psd = asRecord(connection.providerSpecificData); + const fromPsd = typeof psd.baseUrl === "string" ? parseMoonshotOrigin(psd.baseUrl) : null; + if (fromPsd) return fromPsd; + const fromNode = parseMoonshotOrigin(nodeBaseUrl); + if (fromNode) return fromNode; + return registryDefaultOrigin(connection.provider); +} + +export function isMoonshotOpenPlatformConnection( + connection: MoonshotOriginConnection, + nodeBaseUrl?: string | null, +): boolean { + return resolveMoonshotOrigin(connection, nodeBaseUrl) !== null; +} + +/** Account-level empty wallet on Open Platform. Narrower than any compatible 429. */ +export function isMoonshotAccountBalanceExhausted(errorText: string | null | undefined): boolean { + const lower = String(errorText || "").toLowerCase(); + return lower.includes("insufficient balance") || lower.includes("exceeded_current_quota"); +} diff --git a/open-sse/services/usage/supportedProviders.ts b/open-sse/services/usage/supportedProviders.ts index dc088fa1c4..b8e0481b86 100644 --- a/open-sse/services/usage/supportedProviders.ts +++ b/open-sse/services/usage/supportedProviders.ts @@ -41,6 +41,8 @@ export const USAGE_SUPPORTED_PROVIDERS: readonly string[] = [ "crof", "nanogpt", "deepseek", + "moonshot", + "kimi", "xiaomi-mimo", "xiaomi-mimo-token-plan", "vertex", diff --git a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditCompatibleNodeModal.tsx b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditCompatibleNodeModal.tsx index 026aa324fa..dc1966677f 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditCompatibleNodeModal.tsx +++ b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditCompatibleNodeModal.tsx @@ -15,6 +15,8 @@ interface EditCompatibleNodeModalNode { chatPath?: string; modelsPath?: string; iconUrl?: string; + dailyQuotaResetTimezone?: string | null; + dailyQuotaResetHour?: number | null; providerSpecificData?: Record; } @@ -48,6 +50,8 @@ export default function EditCompatibleNodeModal({ consoleApiKey: "", newApiUserId: "", quotaPerUnit: "", + dailyQuotaResetTimezone: "", + dailyQuotaResetHour: "", }); const [saving, setSaving] = useState(false); const [checkKey, setCheckKey] = useState(""); @@ -98,6 +102,11 @@ export default function EditCompatibleNodeModal({ consoleApiKey: typeof psd.consoleApiKey === "string" ? psd.consoleApiKey : "", newApiUserId: typeof psd.newApiUserId === "string" ? psd.newApiUserId : "", quotaPerUnit: typeof psd.quotaPerUnit === "number" ? String(psd.quotaPerUnit) : "", + dailyQuotaResetTimezone: node.dailyQuotaResetTimezone || "", + dailyQuotaResetHour: + node.dailyQuotaResetHour === 0 || node.dailyQuotaResetHour + ? String(node.dailyQuotaResetHour) + : "", }); setSaveError(null); setIconUrlError(null); @@ -141,6 +150,10 @@ export default function EditCompatibleNodeModal({ modelsPath: isCcCompatible ? "" : formData.modelsPath, iconUrl: formData.iconUrl.trim(), }; + const tz = formData.dailyQuotaResetTimezone.trim(); + payload.dailyQuotaResetTimezone = tz || null; + const hourRaw = formData.dailyQuotaResetHour.trim(); + payload.dailyQuotaResetHour = hourRaw === "" ? null : Number(hourRaw); if (!isAnthropic) { payload.apiType = formData.apiType; } @@ -345,6 +358,22 @@ export default function EditCompatibleNodeModal({ hint={t("modelsPathHint")} /> )} + + setFormData({ ...formData, dailyQuotaResetTimezone: e.target.value }) + } + placeholder="Asia/Shanghai" + hint={t("dailyQuotaResetTimezoneHint")} + /> + setFormData({ ...formData, dailyQuotaResetHour: e.target.value })} + placeholder="0" + hint={t("dailyQuotaResetHourHint")} + /> )}
diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.tsx b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.tsx index d5f7fa0de0..eda5308eb4 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.tsx +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.tsx @@ -16,8 +16,8 @@ import { } from "./utils"; import Card from "@/shared/components/Card"; import { CardSkeleton } from "@/shared/components/Loading"; -import { USAGE_SUPPORTED_PROVIDERS } from "@/shared/constants/providers"; import { pickDisplayValue } from "@/shared/utils/maskEmail"; +import { supportsProviderQuota, isProviderQuotaVisible } from "@/shared/utils/providerQuotaVisibility"; import useEmailPrivacyStore from "@/store/emailPrivacyStore"; import { useNotificationStore } from "@/store/notificationStore"; @@ -32,7 +32,6 @@ import { formatAutoRefreshCountdown } from "./formatters"; import { translateUsageOrFallback, type UsageTranslationValues } from "./i18nFallback"; import { compareTr } from "@/shared/utils/turkishText"; import { fetchWithTimeout } from "@/shared/utils/fetchTimeout"; -import { isProviderQuotaVisible } from "@/shared/utils/providerQuotaVisibility"; // Bound the two first-paint requests so a stalled connection cannot wedge // `initialLoading` on `true` and freeze the quota page on its skeleton forever @@ -529,7 +528,7 @@ export default function ProviderLimits({ connections.filter( (conn) => isProviderQuotaVisible(conn) && - USAGE_SUPPORTED_PROVIDERS.includes(conn.provider) && + supportsProviderQuota(conn.provider, conn) && (conn.authType === "oauth" || conn.authType === "apikey") ), [connections] diff --git a/src/app/(dashboard)/home/ProviderQuotaWidget.tsx b/src/app/(dashboard)/home/ProviderQuotaWidget.tsx index 37b11ff286..3e91bd3526 100644 --- a/src/app/(dashboard)/home/ProviderQuotaWidget.tsx +++ b/src/app/(dashboard)/home/ProviderQuotaWidget.tsx @@ -4,7 +4,7 @@ import { useCallback, useEffect, useMemo, useRef, useState } from "react"; import { useTranslations } from "next-intl"; import Card from "@/shared/components/Card"; import ProviderIcon from "@/shared/components/ProviderIcon"; -import { USAGE_SUPPORTED_PROVIDERS } from "@/shared/constants/providers"; +import { supportsProviderQuota } from "@/shared/utils/providerQuotaVisibility"; import QuotaMiniBar from "../dashboard/usage/components/ProviderLimits/QuotaMiniBar"; import { PROVIDER_LABEL } from "../dashboard/usage/components/ProviderLimits/constants"; import { translateUsageOrFallback } from "../dashboard/usage/components/ProviderLimits/i18nFallback"; @@ -25,6 +25,7 @@ type Connection = { name?: string; displayName?: string; email?: string; + providerSpecificData?: unknown; }; type QuotaData = Record; @@ -178,7 +179,7 @@ export default function ProviderQuotaWidget({ const quotaResponseData = quotasResponse.ok ? await quotasResponse.json() : {}; const relevant = ((connectionData.connections || []) as Connection[]).filter( (connection) => - USAGE_SUPPORTED_PROVIDERS.includes(connection.provider) && + supportsProviderQuota(connection.provider, connection) && (connection.authType === "oauth" || connection.authType === "apikey") ); setConnections(relevant); diff --git a/src/app/api/provider-nodes/[id]/route.ts b/src/app/api/provider-nodes/[id]/route.ts index eef4803372..19bd524ef8 100644 --- a/src/app/api/provider-nodes/[id]/route.ts +++ b/src/app/api/provider-nodes/[id]/route.ts @@ -56,7 +56,7 @@ export async function PUT(request: Request, { params }: { params: Promise<{ id: if (isValidationFailure(validation)) { return NextResponse.json({ error: validation.error }, { status: 400 }); } - const { name, prefix, apiType, baseUrl, chatPath, modelsPath, customHeaders, iconUrl } = + const { name, prefix, apiType, baseUrl, chatPath, modelsPath, customHeaders, iconUrl, dailyQuotaResetTimezone, dailyQuotaResetHour } = validation.data; const node: any = await getProviderNodeById(id); @@ -98,6 +98,9 @@ export async function PUT(request: Request, { params }: { params: Promise<{ id: // previously stored custom icon. iconUrl: iconUrl?.trim() || null, customHeaders: customHeaders || null, + dailyQuotaResetTimezone: dailyQuotaResetTimezone?.trim() || null, + dailyQuotaResetHour: + dailyQuotaResetHour === 0 || dailyQuotaResetHour != null ? dailyQuotaResetHour : null, }; if (node.type === "openai-compatible") { @@ -106,6 +109,21 @@ export async function PUT(request: Request, { params }: { params: Promise<{ id: const updated = await updateProviderNode(id, updates); + try { + const { registerMoonshotFetchersForNodes } = await import( + "@omniroute/open-sse/services/moonshotQuotaFetcher.ts" + ); + registerMoonshotFetchersForNodes([ + { + id: typeof updated?.id === "string" ? updated.id : id, + prefix: prefix.trim(), + baseUrl: sanitizedBaseUrl, + }, + ]); + } catch (error) { + console.warn("Moonshot fetcher re-register after node update skipped:", error); + } + const connections = await getProviderConnections({ provider: id }); await Promise.all( connections.flatMap((connectionRaw) => { diff --git a/src/app/api/provider-nodes/route.ts b/src/app/api/provider-nodes/route.ts index a11180f505..d60e3b299a 100644 --- a/src/app/api/provider-nodes/route.ts +++ b/src/app/api/provider-nodes/route.ts @@ -48,6 +48,27 @@ function sanitizeVibeProxyBaseUrl(baseUrl: string) { return `${base}/v1`; } +async function registerMoonshotFetchersForCreatedNode(node: { + id?: unknown; + prefix?: unknown; + baseUrl?: unknown; +}): Promise { + try { + const { registerMoonshotFetchersForNodes } = await import( + "@omniroute/open-sse/services/moonshotQuotaFetcher.ts" + ); + registerMoonshotFetchersForNodes([ + { + id: typeof node.id === "string" ? node.id : null, + prefix: typeof node.prefix === "string" ? node.prefix : null, + baseUrl: typeof node.baseUrl === "string" ? node.baseUrl : null, + }, + ]); + } catch (error) { + console.warn("Moonshot fetcher register after node create skipped:", error); + } +} + function sanitizeAnthropicBaseUrl(baseUrl: string) { return (baseUrl || "") .trim() @@ -126,6 +147,8 @@ export async function POST(request) { modelsPath, customHeaders, iconUrl, + dailyQuotaResetTimezone, + dailyQuotaResetHour, } = validation.data; if (preset === "vibeproxy-openai") { @@ -145,7 +168,11 @@ export async function POST(request) { modelsPath: modelsPath || null, iconUrl: iconUrl?.trim() || null, customHeaders: customHeaders || null, + dailyQuotaResetTimezone: dailyQuotaResetTimezone?.trim() || null, + dailyQuotaResetHour: + dailyQuotaResetHour === 0 || dailyQuotaResetHour != null ? dailyQuotaResetHour : null, }); + await registerMoonshotFetchersForCreatedNode(node); return NextResponse.json({ node }, { status: 201 }); } @@ -170,7 +197,11 @@ export async function POST(request) { modelsPath: modelsPath || null, iconUrl: iconUrl?.trim() || null, customHeaders: customHeaders || null, + dailyQuotaResetTimezone: dailyQuotaResetTimezone?.trim() || null, + dailyQuotaResetHour: + dailyQuotaResetHour === 0 || dailyQuotaResetHour != null ? dailyQuotaResetHour : null, }); + await registerMoonshotFetchersForCreatedNode(node); return NextResponse.json({ node }, { status: 201 }); } @@ -200,7 +231,11 @@ export async function POST(request) { modelsPath: compatMode === "cc" ? null : modelsPath || null, iconUrl: iconUrl?.trim() || null, customHeaders: customHeaders || null, + dailyQuotaResetTimezone: dailyQuotaResetTimezone?.trim() || null, + dailyQuotaResetHour: + dailyQuotaResetHour === 0 || dailyQuotaResetHour != null ? dailyQuotaResetHour : null, }); + await registerMoonshotFetchersForCreatedNode(node); return NextResponse.json({ node }, { status: 201 }); } diff --git a/src/i18n/messages/ar.json b/src/i18n/messages/ar.json index a2d2c4acf9..bdf4b7a4c7 100644 --- a/src/i18n/messages/ar.json +++ b/src/i18n/messages/ar.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "عنوان URL للأيقونة", "iconUrlHint": "اختياري. عنوان URL للصورة المعروضة كأيقونة لهذا المزود.", "iconUrlInvalid": "رابط الأيقونة غير صالح. استخدم http(s):// أو data:image/*;base64 رابط.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/az.json b/src/i18n/messages/az.json index 131a9b212b..543b6b60b0 100644 --- a/src/i18n/messages/az.json +++ b/src/i18n/messages/az.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "İkon URL-i", "iconUrlHint": "İstəyə bağlı. Bu provayderin ikonu kimi göstərilən şəkil URL-i.", "iconUrlInvalid": "Yanlış ikon URL-si. http(s):// və ya data:image/*;base64 URL istifadə edin.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/bg.json b/src/i18n/messages/bg.json index f51cea18c1..3944642aff 100644 --- a/src/i18n/messages/bg.json +++ b/src/i18n/messages/bg.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL адрес на икона", "iconUrlHint": "По избор. URL адрес на изображение, показвано като икона на този доставчик.", "iconUrlInvalid": "Невалиден URL на иконата. Използвайте http(s):// или data:image/*;base64 URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/bn.json b/src/i18n/messages/bn.json index 4dacb5dd56..8f6afadff3 100644 --- a/src/i18n/messages/bn.json +++ b/src/i18n/messages/bn.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "আইকন URL", "iconUrlHint": "ঐচ্ছিক। এই প্রদানকারীর আইকন হিসেবে দেখানোর জন্য ছবির URL।", "iconUrlInvalid": "অবৈধ আইকন URL। একটি http(s):// অথবা data:image/*;base64 URL ব্যবহার করুন।", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/cs.json b/src/i18n/messages/cs.json index 2450daa9ff..29af05425a 100644 --- a/src/i18n/messages/cs.json +++ b/src/i18n/messages/cs.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL ikony", "iconUrlHint": "Volitelné. URL obrázku zobrazeného jako ikona tohoto poskytovatele.", "iconUrlInvalid": "Neplatná URL ikony. Použijte http(s):// nebo data:image/*;base64 URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/da.json b/src/i18n/messages/da.json index 4dd4699f43..bea97fe340 100644 --- a/src/i18n/messages/da.json +++ b/src/i18n/messages/da.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "Ikon-URL", "iconUrlHint": "Valgfrit. Billed-URL, der vises som denne udbyders ikon.", "iconUrlInvalid": "Ugyldig ikon-URL. Brug en http(s):// eller data:image/*;base64 URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/de.json b/src/i18n/messages/de.json index 0fa6991704..d0c0322ac1 100644 --- a/src/i18n/messages/de.json +++ b/src/i18n/messages/de.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "Icon-URL", "iconUrlHint": "Optional. Bild-URL, die als Icon dieses Anbieters angezeigt wird.", "iconUrlInvalid": "Ungültige Icon-URL. Verwenden Sie eine http(s):// oder data:image/*;base64 URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "AC-Prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index e1b0adcb11..e9643f3511 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -5251,6 +5251,10 @@ "iconUrlLabel": "Icon URL", "iconUrlHint": "Optional. Image URL shown as this provider's icon.", "iconUrlInvalid": "Invalid icon URL. Use an http(s):// or data:image/*;base64 URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/es.json b/src/i18n/messages/es.json index 99f66790a4..ba61b59daa 100644 --- a/src/i18n/messages/es.json +++ b/src/i18n/messages/es.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "Icon URL", "iconUrlHint": "Optional. Image URL shown as this provider's icon.", "iconUrlInvalid": "URL de icono no válida. Utilice una URL http(s):// o data:image/*;base64.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "producto ac", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/fa.json b/src/i18n/messages/fa.json index 176e89c4d2..79c0ec653f 100644 --- a/src/i18n/messages/fa.json +++ b/src/i18n/messages/fa.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL آیکون", "iconUrlHint": "اختیاری. URL تصویری که به عنوان آیکون این ارائه‌دهنده نمایش داده می‌شود.", "iconUrlInvalid": "آدرس آیکون نامعتبر است. از آدرس http(s):// یا data:image/*;base64 استفاده کنید.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/fi.json b/src/i18n/messages/fi.json index 8c7c2c9c98..c83a56cb2c 100644 --- a/src/i18n/messages/fi.json +++ b/src/i18n/messages/fi.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "Kuvakkeen URL-osoite", "iconUrlHint": "Valinnainen. Kuvan URL-osoite, joka näytetään tämän tarjoajan kuvakkeena.", "iconUrlInvalid": "Virheellinen kuvakkeen URL. Käytä http(s):// tai data:image/*;base64 URL:ia.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/fr.json b/src/i18n/messages/fr.json index c17cb242a0..1bb74a6c87 100644 --- a/src/i18n/messages/fr.json +++ b/src/i18n/messages/fr.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL de l'icône", "iconUrlHint": "Facultatif. URL de l'image affichée comme icône de ce fournisseur.", "iconUrlInvalid": "URL d'icône invalide. Utilisez une URL http(s):// ou data:image/*;base64.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/gu.json b/src/i18n/messages/gu.json index 00196e5ad8..7fbc8ac29b 100644 --- a/src/i18n/messages/gu.json +++ b/src/i18n/messages/gu.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "આઇકન URL", "iconUrlHint": "વૈકલ્પિક. આ પ્રદાતાના આઇકન તરીકે દર્શાવેલ છબી URL.", "iconUrlInvalid": "અમાન્ય આઇકન URL. http(s):// અથવા data:image/*;base64 URL નો ઉપયોગ કરો.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/he.json b/src/i18n/messages/he.json index 19d7e96e65..66bea20bb8 100644 --- a/src/i18n/messages/he.json +++ b/src/i18n/messages/he.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "כתובת URL של סמל", "iconUrlHint": "אופציונלי. כתובת URL של תמונה שתוצג כסמל של ספק זה.", "iconUrlInvalid": "כתובת ה-URL של האייקון אינה חוקית. השתמש ב-http(s):// או ב-data:image/*;base64 URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/hi.json b/src/i18n/messages/hi.json index 2f9b11ace2..f45c0fad90 100644 --- a/src/i18n/messages/hi.json +++ b/src/i18n/messages/hi.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "आइकन URL", "iconUrlHint": "वैकल्पिक। इस प्रदाता के आइकन के रूप में दिखाया जाने वाला छवि URL।", "iconUrlInvalid": "अमान्य आइकन URL। http(s):// या data:image/*;base64 URL का उपयोग करें।", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "एसी-उत्पाद", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/hu.json b/src/i18n/messages/hu.json index 339507ca3c..b1275355b9 100644 --- a/src/i18n/messages/hu.json +++ b/src/i18n/messages/hu.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "Ikon URL-címe", "iconUrlHint": "Opcionális. A szolgáltató ikonjaként megjelenő kép URL-címe.", "iconUrlInvalid": "Érvénytelen ikon URL. Használjon http(s):// vagy data:image/*;base64 URL-t.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/id.json b/src/i18n/messages/id.json index bf2f06a265..474a26973d 100644 --- a/src/i18n/messages/id.json +++ b/src/i18n/messages/id.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL Ikon", "iconUrlHint": "Opsional. URL gambar yang ditampilkan sebagai ikon penyedia ini.", "iconUrlInvalid": "URL ikon tidak valid. Gunakan URL http(s):// atau data:image/*;base64.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/it.json b/src/i18n/messages/it.json index 855d5b3ecb..86e1226241 100644 --- a/src/i18n/messages/it.json +++ b/src/i18n/messages/it.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL Icona", "iconUrlHint": "Opzionale. URL dell'immagine mostrata come icona di questo provider.", "iconUrlInvalid": "URL dell'icona non valida. Usa un URL http(s):// o data:image/*;base64.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/ja.json b/src/i18n/messages/ja.json index 6e734c226e..ab2f52c289 100644 --- a/src/i18n/messages/ja.json +++ b/src/i18n/messages/ja.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "アイコンURL", "iconUrlHint": "任意。このプロバイダーのアイコンとして表示される画像URL。", "iconUrlInvalid": "無効なアイコンURLです。http(s)://またはdata:image/*;base64 URLを使用してください。", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/ko.json b/src/i18n/messages/ko.json index fb12ad0c99..ad26677619 100644 --- a/src/i18n/messages/ko.json +++ b/src/i18n/messages/ko.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "아이콘 URL", "iconUrlHint": "선택 사항. 이 제공자의 아이콘으로 표시될 이미지 URL입니다.", "iconUrlInvalid": "잘못된 아이콘 URL입니다. http(s):// 또는 data:image/*;base64 URL을 사용하세요.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/mr.json b/src/i18n/messages/mr.json index 8817c1c87f..7f5ee425c9 100644 --- a/src/i18n/messages/mr.json +++ b/src/i18n/messages/mr.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "आयकॉन URL", "iconUrlHint": "पर्यायी. या प्रदात्याचा आयकॉन म्हणून दर्शविलेली इमेज URL.", "iconUrlInvalid": "अवैध आयकॉन URL. http(s):// किंवा data:image/*;base64 URL वापरा.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/ms.json b/src/i18n/messages/ms.json index 186db22e3a..a780d09bcb 100644 --- a/src/i18n/messages/ms.json +++ b/src/i18n/messages/ms.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL Ikon", "iconUrlHint": "Pilihan. URL imej yang ditunjukkan sebagai ikon penyedia ini.", "iconUrlInvalid": "URL ikon tidak sah. Gunakan URL http(s):// atau data:image/*;base64.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/nl.json b/src/i18n/messages/nl.json index edf66b80a1..706a5ccd35 100644 --- a/src/i18n/messages/nl.json +++ b/src/i18n/messages/nl.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "Pictogram-URL", "iconUrlHint": "Optioneel. Afbeeldings-URL die wordt getoond als het pictogram van deze provider.", "iconUrlInvalid": "Ongeldige pictogram-URL. Gebruik een http(s):// of data:image/*;base64 URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/no.json b/src/i18n/messages/no.json index d4e45081b6..9576e4b6a6 100644 --- a/src/i18n/messages/no.json +++ b/src/i18n/messages/no.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "Ikon-URL", "iconUrlHint": "Valgfritt. Bilde-URL som vises som denne leverandørens ikon.", "iconUrlInvalid": "Ugyldig ikon-URL. Bruk en http(s):// eller data:image/*;base64 URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/phi.json b/src/i18n/messages/phi.json index 6f9f973ec1..856b906cfb 100644 --- a/src/i18n/messages/phi.json +++ b/src/i18n/messages/phi.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL ng Icon", "iconUrlHint": "Opsyonal. URL ng larawan na ipinapakita bilang icon ng provider na ito.", "iconUrlInvalid": "Hindi wastong URL ng icon. Gumamit ng http(s):// o data:image/*;base64 na URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/pl.json b/src/i18n/messages/pl.json index 002c11a98d..b9ecc290db 100644 --- a/src/i18n/messages/pl.json +++ b/src/i18n/messages/pl.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL ikony", "iconUrlHint": "Opcjonalnie. URL obrazu wyświetlany jako ikona tego provider.", "iconUrlInvalid": "Nieprawidłowy adres URL ikony. Użyj adresu http(s):// lub data:image/*;base64.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index 5717bff5cc..e65916a54d 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -5252,6 +5252,10 @@ "iconUrlLabel": "URL do ícone", "iconUrlHint": "Opcional. URL da imagem exibida como ícone deste provedor.", "iconUrlInvalid": "URL do ícone inválido. Use um URL http(s):// ou data:image/*;base64.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index 4722209d02..8e518e2e0e 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL do ícone", "iconUrlHint": "Opcional. URL da imagem mostrada como ícone deste provedor.", "iconUrlInvalid": "URL do ícone inválido. Utilize uma URL http(s):// ou data:image/*;base64.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/ro.json b/src/i18n/messages/ro.json index a0d82ca4d5..0bfdea697e 100644 --- a/src/i18n/messages/ro.json +++ b/src/i18n/messages/ro.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL pictogramă", "iconUrlHint": "Opțional. URL-ul imaginii afișate ca pictogramă a acestui furnizor.", "iconUrlInvalid": "URL-ul iconului este invalid. Folosiți un URL http(s):// sau data:image/*;base64.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/ru.json b/src/i18n/messages/ru.json index 4a7646aaef..c63a9fab99 100644 --- a/src/i18n/messages/ru.json +++ b/src/i18n/messages/ru.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL-адрес иконки", "iconUrlHint": "Необязательно. URL-адрес изображения, используемого в качестве иконки этого провайдера.", "iconUrlInvalid": "Неверный URL значка. Используйте URL-адрес http(s):// или data:image/*;base64.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-прод", "openaiPrefixPlaceholder": "oc-прод", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/sk.json b/src/i18n/messages/sk.json index 66a01e79e2..de5778a2cd 100644 --- a/src/i18n/messages/sk.json +++ b/src/i18n/messages/sk.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL ikony", "iconUrlHint": "Voliteľné. URL obrázka zobrazeného ako ikona tohto poskytovateľa.", "iconUrlInvalid": "Neplatná URL ikony. Použite http(s):// alebo data:image/*;base64 URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/sv.json b/src/i18n/messages/sv.json index d8ca20a04d..0408b0806b 100644 --- a/src/i18n/messages/sv.json +++ b/src/i18n/messages/sv.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "Ikon-URL", "iconUrlHint": "Valfritt. Bild-URL som visas som denna leverantörs ikon.", "iconUrlInvalid": "Ogiltig ikon-URL. Använd en http(s):// eller data:image/*;base64-URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/sw.json b/src/i18n/messages/sw.json index 5507c02543..88b9200348 100644 --- a/src/i18n/messages/sw.json +++ b/src/i18n/messages/sw.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL ya Aikoni", "iconUrlHint": "Si lazima. URL ya picha inayoonyeshwa kama aikoni ya mtoa huduma huyu.", "iconUrlInvalid": "URL ya ikoni si sahihi. Tumia http(s):// au data:image/*;base64 URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/ta.json b/src/i18n/messages/ta.json index a5e601ae93..95da2289ce 100644 --- a/src/i18n/messages/ta.json +++ b/src/i18n/messages/ta.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "சின்னத்தின் URL", "iconUrlHint": "விருப்பத்திற்குரியது. இந்த வழங்குநரின் சின்னமாக காட்டப்படும் பட URL.", "iconUrlInvalid": "தவறான ஐகான் URL. http(s):// அல்லது data:image/*;base64 URL ஐப் பயன்படுத்தவும்.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/te.json b/src/i18n/messages/te.json index 69a51f4d6a..3526a50b41 100644 --- a/src/i18n/messages/te.json +++ b/src/i18n/messages/te.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "ఐకాన్ URL", "iconUrlHint": "ఐచ్ఛికం. ఈ ప్రొవైడర్ ఐకాన్‌గా చూపబడే చిత్రం URL.", "iconUrlInvalid": "చెల్లని ఐకాన్ URL. http(s):// లేదా data:image/*;base64 URL ఉపయోగించండి.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/th.json b/src/i18n/messages/th.json index b43dda761c..e19eb88c20 100644 --- a/src/i18n/messages/th.json +++ b/src/i18n/messages/th.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL ไอคอน", "iconUrlHint": "ไม่บังคับ URL รูปภาพที่จะแสดงเป็นไอคอนของผู้ให้บริการรายนี้", "iconUrlInvalid": "URL ไอคอนไม่ถูกต้อง ใช้ http(s):// หรือ data:image/*;base64 URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-ผลิตภัณฑ์", "openaiPrefixPlaceholder": "oc-ผลิตภัณฑ์", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/tr.json b/src/i18n/messages/tr.json index 31e48fb100..12b3973933 100644 --- a/src/i18n/messages/tr.json +++ b/src/i18n/messages/tr.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "Simge URL'si", "iconUrlHint": "İsteğe bağlı. Bu sağlayıcının simgesi olarak gösterilen görsel URL'si.", "iconUrlInvalid": "Geçersiz simge URL'si. http(s):// veya data:image/*;base64 URL'si kullanın.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/uk-UA.json b/src/i18n/messages/uk-UA.json index 62d8db3226..fd9a0732d3 100644 --- a/src/i18n/messages/uk-UA.json +++ b/src/i18n/messages/uk-UA.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "URL-адреса іконки", "iconUrlHint": "Необов'язково. URL-адреса зображення, що відображається як іконка цього провайдера.", "iconUrlInvalid": "Недійсне URL-адреса значка. Використовуйте http(s):// або data:image/*;base64 URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/ur.json b/src/i18n/messages/ur.json index 1a4f1de826..02dba16c45 100644 --- a/src/i18n/messages/ur.json +++ b/src/i18n/messages/ur.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "آئیکن URL", "iconUrlHint": "اختیاری۔ تصویر کا URL جو اس فراہم کنندہ کے آئیکن کے طور پر دکھایا گیا ہے۔", "iconUrlInvalid": "غلط آئیکن یو آر ایل۔ http(s):// یا data:image/*;base64 یو آر ایل استعمال کریں۔", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/vi.json b/src/i18n/messages/vi.json index 0a8428ecdf..cf7235f61b 100644 --- a/src/i18n/messages/vi.json +++ b/src/i18n/messages/vi.json @@ -5252,6 +5252,10 @@ "iconUrlLabel": "URL biểu tượng", "iconUrlHint": "Tùy chọn. URL hình ảnh được hiển thị làm biểu tượng của nhà cung cấp này.", "iconUrlInvalid": "URL biểu tượng không hợp lệ. Sử dụng http(s):// hoặc data:image/*;base64 URL.", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/zh-CN.json b/src/i18n/messages/zh-CN.json index facdccc8c2..e53f191cf9 100644 --- a/src/i18n/messages/zh-CN.json +++ b/src/i18n/messages/zh-CN.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "图标 URL", "iconUrlHint": "可选。显示为此服务商图标的图片 URL。", "iconUrlInvalid": "无效的图标 URL。请使用 http(s):// 或 data:image/*;base64 URL。", + "dailyQuotaResetTimezoneLabel": "每日额度重置时区", + "dailyQuotaResetTimezoneHint": "可选。上游不返回 X-RateLimit-Reset 时使用的 IANA 时区。留空则只做短冷却。", + "dailyQuotaResetHourLabel": "每日额度重置小时", + "dailyQuotaResetHourHint": "上述时区的本地小时,0-23。留空表示不配置节点级时钟。", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/i18n/messages/zh-TW.json b/src/i18n/messages/zh-TW.json index 3d10722987..2cddce94b2 100644 --- a/src/i18n/messages/zh-TW.json +++ b/src/i18n/messages/zh-TW.json @@ -5248,6 +5248,10 @@ "iconUrlLabel": "圖示網址", "iconUrlHint": "選用。顯示為此提供者圖示的圖片網址。", "iconUrlInvalid": "無效的圖示網址。請使用 http(s):// 或 data:image/*;base64 URL。", + "dailyQuotaResetTimezoneLabel": "Daily quota reset timezone", + "dailyQuotaResetTimezoneHint": "Optional IANA timezone used when the upstream omits X-RateLimit-Reset. Leave empty to keep a short cooldown.", + "dailyQuotaResetHourLabel": "Daily quota reset hour", + "dailyQuotaResetHourHint": "Local hour 0-23 in the timezone above. Empty means no node-level clock.", "anthropicPrefixPlaceholder": "ac-prod", "openaiPrefixPlaceholder": "oc-prod", "anthropicBaseUrlPlaceholder": "https://api.anthropic.com/v1", diff --git a/src/instrumentation-node.ts b/src/instrumentation-node.ts index 069b8740b8..e3ae959118 100755 --- a/src/instrumentation-node.ts +++ b/src/instrumentation-node.ts @@ -282,6 +282,7 @@ export async function registerQuotaFetchers(): Promise { { registerQwenTokenPlanQuotaFetcher }, { registerCrofUsageFetcher }, { registerDeepseekQuotaFetcher }, + { registerMoonshotQuotaFetcher, registerMoonshotFetchersForNodes }, { registerOpenrouterQuotaFetcher }, { registerOpencodeQuotaFetcher }, { registerGrokWebQuotaFetcher }, @@ -292,6 +293,7 @@ export async function registerQuotaFetchers(): Promise { import("@omniroute/open-sse/services/qwenTokenPlanQuotaFetcher"), import("@omniroute/open-sse/services/crofUsageFetcher"), import("@omniroute/open-sse/services/deepseekQuotaFetcher"), + import("@omniroute/open-sse/services/moonshotQuotaFetcher"), import("@omniroute/open-sse/services/openrouterQuotaFetcher"), import("@omniroute/open-sse/services/opencodeQuotaFetcher"), import("@omniroute/open-sse/services/grokQuotaFetcher"), @@ -303,6 +305,20 @@ export async function registerQuotaFetchers(): Promise { registerQwenTokenPlanQuotaFetcher(); registerCrofUsageFetcher(); registerDeepseekQuotaFetcher(); + registerMoonshotQuotaFetcher(); + try { + const { getProviderNodes } = await import("@/lib/db/providers"); + const nodes = await getProviderNodes(); + registerMoonshotFetchersForNodes( + (Array.isArray(nodes) ? nodes : []).map((node) => ({ + id: typeof node.id === "string" ? node.id : null, + prefix: typeof node.prefix === "string" ? node.prefix : null, + baseUrl: typeof node.baseUrl === "string" ? node.baseUrl : null, + })), + ); + } catch (error) { + console.warn("[STARTUP] Moonshot custom-node fetcher scan skipped:", error); + } registerOpenrouterQuotaFetcher(); registerOpencodeQuotaFetcher(); registerGrokWebQuotaFetcher(); diff --git a/src/lib/db/migrationRunner.ts b/src/lib/db/migrationRunner.ts index 498be0f9bc..f53cd58070 100644 --- a/src/lib/db/migrationRunner.ts +++ b/src/lib/db/migrationRunner.ts @@ -516,6 +516,11 @@ function isSchemaAlreadyApplied( db.prepare("SELECT 1 FROM provider_connections WHERE provider = 'freepik' LIMIT 1").get() == null ); + case "172": + return ( + hasColumn(db, "provider_nodes", "daily_quota_reset_timezone") && + hasColumn(db, "provider_nodes", "daily_quota_reset_hour") + ); default: return false; } diff --git a/src/lib/db/migrations/172_provider_node_daily_quota_reset.sql b/src/lib/db/migrations/172_provider_node_daily_quota_reset.sql new file mode 100644 index 0000000000..969df2e0fb --- /dev/null +++ b/src/lib/db/migrations/172_provider_node_daily_quota_reset.sql @@ -0,0 +1,6 @@ +-- 172: per-node daily quota reset clock (IANA timezone + local hour). +-- Used by TPD cooldown when upstream omits X-RateLimit-Reset. +-- Both columns nullable: empty = operator has not configured a clock. + +ALTER TABLE provider_nodes ADD COLUMN daily_quota_reset_timezone TEXT; +ALTER TABLE provider_nodes ADD COLUMN daily_quota_reset_hour INTEGER; diff --git a/src/lib/db/providers/nodes.ts b/src/lib/db/providers/nodes.ts index 7c9a251a78..3bc975db92 100644 --- a/src/lib/db/providers/nodes.ts +++ b/src/lib/db/providers/nodes.ts @@ -9,6 +9,25 @@ import { backupDbFile } from "../backup"; import { invalidateDbCache } from "../readCache"; import { toRecord, type JsonRecord } from "./columns"; +function normalizeDailyQuotaResetHour(value: unknown): number | null { + return value === 0 || value ? Number(value) : null; +} + +function withParsedCustomHeaders(node: JsonRecord, storedJson: string | null): JsonRecord { + const result: JsonRecord = { ...node }; + if (storedJson) { + try { + result.customHeaders = JSON.parse(storedJson); + } catch { + result.customHeaders = null; + } + } else { + result.customHeaders = null; + } + delete result.customHeadersJson; + return result; +} + interface StatementLike { all: (...params: unknown[]) => TRow[]; get: (...params: unknown[]) => TRow | undefined; @@ -88,32 +107,23 @@ export async function createProviderNode(data: JsonRecord) { // Optional operator-supplied remote icon URL (#2166) — plain TEXT, no JSON parsing needed. iconUrl: data.iconUrl || null, customHeadersJson, + dailyQuotaResetTimezone: data.dailyQuotaResetTimezone || null, + dailyQuotaResetHour: normalizeDailyQuotaResetHour(data.dailyQuotaResetHour), createdAt: now, updatedAt: now, }; db.prepare( ` - INSERT INTO provider_nodes (id, type, name, prefix, api_type, base_url, chat_path, models_path, icon_url, custom_headers_json, created_at, updated_at) - VALUES (@id, @type, @name, @prefix, @apiType, @baseUrl, @chatPath, @modelsPath, @iconUrl, @customHeadersJson, @createdAt, @updatedAt) + INSERT INTO provider_nodes (id, type, name, prefix, api_type, base_url, chat_path, models_path, icon_url, custom_headers_json, daily_quota_reset_timezone, daily_quota_reset_hour, created_at, updated_at) + VALUES (@id, @type, @name, @prefix, @apiType, @baseUrl, @chatPath, @modelsPath, @iconUrl, @customHeadersJson, @dailyQuotaResetTimezone, @dailyQuotaResetHour, @createdAt, @updatedAt) ` ).run(node); backupDbFile("pre-write"); invalidateDbCache("nodes"); - const result: JsonRecord = { ...node }; - if (customHeadersJson) { - try { - result.customHeaders = JSON.parse(customHeadersJson); - } catch { - result.customHeaders = null; - } - } else { - result.customHeaders = null; - } - delete result.customHeadersJson; - return result; + return withParsedCustomHeaders({ ...node }, customHeadersJson); } export async function updateProviderNode(id: string, data: JsonRecord) { @@ -144,7 +154,10 @@ export async function updateProviderNode(id: string, data: JsonRecord) { UPDATE provider_nodes SET type = @type, name = @name, prefix = @prefix, api_type = @apiType, base_url = @baseUrl, chat_path = @chatPath, models_path = @modelsPath, icon_url = @iconUrl, - custom_headers_json = @customHeadersJson, updated_at = @updatedAt + custom_headers_json = @customHeadersJson, + daily_quota_reset_timezone = @dailyQuotaResetTimezone, + daily_quota_reset_hour = @dailyQuotaResetHour, + updated_at = @updatedAt WHERE id = @id ` ).run({ @@ -160,25 +173,15 @@ export async function updateProviderNode(id: string, data: JsonRecord) { // stored custom icon when the caller submits an empty value. iconUrl: merged["iconUrl"] || null, customHeadersJson: merged["customHeadersJson"] || null, + dailyQuotaResetTimezone: merged["dailyQuotaResetTimezone"] || null, + dailyQuotaResetHour: normalizeDailyQuotaResetHour(merged["dailyQuotaResetHour"]), updatedAt: merged["updatedAt"], }); backupDbFile("pre-write"); invalidateDbCache("nodes"); - const result: JsonRecord = { ...merged }; - const storedJson = merged["customHeadersJson"] as string | null; - if (storedJson) { - try { - result.customHeaders = JSON.parse(storedJson); - } catch { - result.customHeaders = null; - } - } else { - result.customHeaders = null; - } - delete result.customHeadersJson; - return result; + return withParsedCustomHeaders(merged, (merged["customHeadersJson"] as string | null) ?? null); } export async function deleteProviderNode(id: string) { diff --git a/src/lib/usage/apiKeySelfService.ts b/src/lib/usage/apiKeySelfService.ts index 2e9dc9ffa7..7bc2d440ca 100644 --- a/src/lib/usage/apiKeySelfService.ts +++ b/src/lib/usage/apiKeySelfService.ts @@ -1,5 +1,5 @@ import { hasSelfAccountQuotaScope, hasSelfUsageScope } from "@/shared/constants/selfServiceScopes"; -import { USAGE_SUPPORTED_PROVIDERS } from "@/shared/constants/providers"; +import { supportsProviderQuota } from "@/shared/utils/providerQuotaVisibility"; type JsonRecord = Record; type DateLike = number | string | Date | null | undefined; @@ -64,6 +64,7 @@ interface AccountQuotaConnection { id: string; provider: string; lookupFailed?: boolean; + providerSpecificData?: unknown; } function toNumber(value: unknown, fallback = 0): number { @@ -208,11 +209,14 @@ function normalizePlan(value: unknown): unknown { return undefined; } -function isSupportedProvider(provider: string): boolean { - return USAGE_SUPPORTED_PROVIDERS.includes(provider as (typeof USAGE_SUPPORTED_PROVIDERS)[number]); +function isSupportedProvider( + provider: string, + connection?: { provider?: string; providerSpecificData?: unknown }, +): boolean { + return supportsProviderQuota(provider, connection); } -function getConnectionIdentity(value: unknown): { id: string; provider: string } | null { +function getConnectionIdentity(value: unknown): AccountQuotaConnection | null { if (!value || typeof value !== "object" || Array.isArray(value)) return null; const record = value as JsonRecord; if (record.isActive === false) return null; @@ -221,7 +225,11 @@ function getConnectionIdentity(value: unknown): { id: string; provider: string } const provider = typeof record.provider === "string" ? record.provider : ""; if (!id || !provider) return null; - return { id, provider }; + return { + id, + provider, + providerSpecificData: record.providerSpecificData, + }; } async function listAccountQuotaConnections( @@ -297,7 +305,7 @@ async function resolveConnectionAccountQuota( }; } - if (!isSupportedProvider(connection.provider)) { + if (!isSupportedProvider(connection.provider, connection)) { return { provider: connection.provider, connectionId: connection.id, diff --git a/src/lib/usage/providerLimits.ts b/src/lib/usage/providerLimits.ts index 9a36d5e491..2e0b8895da 100644 --- a/src/lib/usage/providerLimits.ts +++ b/src/lib/usage/providerLimits.ts @@ -16,7 +16,7 @@ import { setQuotaCache } from "@/domain/quotaCache"; import { buildClaudeExtraUsageConnectionUpdate } from "@/lib/providers/claudeExtraUsage"; import { clearRecoveredProviderState } from "@/sse/services/auth"; import { getMachineId } from "@/shared/utils/machine"; -import { USAGE_SUPPORTED_PROVIDERS } from "@/shared/constants/providers"; +import { supportsProviderQuota } from "@/shared/utils/providerQuotaVisibility"; import { mergeProviderLimitsCacheEntry, toProviderLimitsCacheEntry } from "./providerLimitsCache"; import { getCredentialRefreshExecutor } from "@omniroute/open-sse/executors/credential.ts"; import { getUsageForProvider } from "@omniroute/open-sse/services/usage.ts"; @@ -174,19 +174,14 @@ function shouldRefreshProviderLimitsCache( } export function isSupportedUsageConnection(connection: ProviderConnectionLike | null): boolean { - if ( - !connection || - !connection.provider || - !USAGE_SUPPORTED_PROVIDERS.includes(connection.provider) - ) { - return false; - } + if (!connection?.provider) return false; - if (connection.authType === "oauth") return true; - return ( - (connection.authType === "apikey" || connection.authType === "api_key") && - PROVIDER_LIMITS_APIKEY_PROVIDERS.has(connection.provider) - ); + if (connection.authType === "oauth") { + return supportsProviderQuota(connection.provider, connection); + } + if (connection.authType !== "apikey" && connection.authType !== "api_key") return false; + if (PROVIDER_LIMITS_APIKEY_PROVIDERS.has(connection.provider)) return true; + return supportsProviderQuota(connection.provider, connection); } function withStatus(error: Error, status: number): Error & { status: number } { diff --git a/src/shared/utils/classify429.ts b/src/shared/utils/classify429.ts index 03b6e41658..d482334778 100644 --- a/src/shared/utils/classify429.ts +++ b/src/shared/utils/classify429.ts @@ -91,6 +91,14 @@ const QUOTA_PATTERNS: ReadonlyArray = [ // Trailing punctuation/whitespace before the closing quote is tolerated // because real API responses may include a period or trailing space. /"error"\s*:\s*"usage limit reached[.\s]*"/i, + + // Moonshot Open Platform organization TPD (tokens-per-day). Live body: + // "request reached organization TPD rate limit, current: N, limit: M". + // Do not use a bare /TPD/ — too wide. Limit is read from the body, never + // hardcoded (Tier0=1.5M, Tier1+=unlimited). + /organization TPD rate limit/i, + /\bTPD rate limit\b/i, + /insufficient balance/i, ]; /** @@ -155,6 +163,9 @@ const TERMINAL_QUOTA_PATTERNS: ReadonlyArray = [ /individual quota reached/i, /enable overages/i, /daily free allocation/i, + /organization TPD rate limit/i, + /\bTPD rate limit\b/i, + /insufficient balance/i, ]; /** diff --git a/src/shared/utils/providerQuotaVisibility.ts b/src/shared/utils/providerQuotaVisibility.ts index 3968005204..81edcdc450 100644 --- a/src/shared/utils/providerQuotaVisibility.ts +++ b/src/shared/utils/providerQuotaVisibility.ts @@ -1,13 +1,23 @@ import { USAGE_SUPPORTED_PROVIDERS } from "@/shared/constants/providers"; +import { isMoonshotOpenPlatformConnection } from "@omniroute/open-sse/services/usage/moonshotOpenPlatform.ts"; export interface ProviderQuotaVisibilityConnection { quotaVisible?: boolean; + provider?: string; + providerSpecificData?: unknown; } export function isProviderQuotaVisible(connection: ProviderQuotaVisibilityConnection): boolean { return connection.quotaVisible !== false; } -export function supportsProviderQuota(providerId: string): boolean { - return USAGE_SUPPORTED_PROVIDERS.includes(providerId); +export function supportsProviderQuota( + providerId: string, + connection?: { provider?: string; providerSpecificData?: unknown }, +): boolean { + if (USAGE_SUPPORTED_PROVIDERS.includes(providerId)) return true; + return isMoonshotOpenPlatformConnection({ + provider: providerId, + providerSpecificData: connection?.providerSpecificData, + }); } diff --git a/src/shared/validation/schemas/provider.ts b/src/shared/validation/schemas/provider.ts index fa6f0ef055..c93e563ab7 100644 --- a/src/shared/validation/schemas/provider.ts +++ b/src/shared/validation/schemas/provider.ts @@ -22,16 +22,36 @@ import { isReservedProviderPrefix, reservedProviderPrefixMessage, } from "@/shared/constants/reservedProviderPrefixes"; - +import { + isValidIanaTimeZone, + isValidResetHour, +} from "@omniroute/open-sse/services/dailyQuotaReset.ts"; import { upstreamHeadersRecordSchema, modelCompatPerProtocolSchema, customHeadersSchema, } from "./misc.ts"; +import { isValidProviderIconUrl } from "@/shared/validation/iconUrl"; export { validateProviderSpecificData }; -import { isValidProviderIconUrl } from "@/shared/validation/iconUrl"; +const dailyQuotaResetTimezoneSchema = z + .string() + .trim() + .optional() + .or(z.literal("")) + .refine((value) => !value || isValidIanaTimeZone(value), { + message: "Unknown IANA timezone", + }); + +const dailyQuotaResetHourSchema = z + .number() + .int() + .optional() + .nullable() + .refine((value) => value == null || isValidResetHour(value), { + message: "Hour must be 0-23", + }); // ──── Provider Schemas ──── @@ -337,6 +357,8 @@ export const createProviderNodeSchema = z // isValidProviderIconUrl (2000 chars for http(s), 256 KiB for data:image). iconUrl: providerNodeIconUrlSchema, customHeaders: customHeadersSchema, + dailyQuotaResetTimezone: dailyQuotaResetTimezoneSchema, + dailyQuotaResetHour: dailyQuotaResetHourSchema, }) .superRefine((value, ctx) => { const nodeType = value.type || "openai-compatible"; @@ -419,6 +441,8 @@ export const updateProviderNodeSchema = z // clears a previously stored custom icon. iconUrl: providerNodeIconUrlSchema, customHeaders: customHeadersSchema, + dailyQuotaResetTimezone: dailyQuotaResetTimezoneSchema, + dailyQuotaResetHour: dailyQuotaResetHourSchema, }) .superRefine((value, ctx) => { // Reserved-prefix guard (tokenrouter bug) — same rationale as the guard in diff --git a/src/sse/handlers/chat.ts b/src/sse/handlers/chat.ts index 6c426bcfce..7ad2131076 100644 --- a/src/sse/handlers/chat.ts +++ b/src/sse/handlers/chat.ts @@ -174,6 +174,7 @@ import { registerBailianCodingPlanQuotaFetcher } from "@omniroute/open-sse/servi import { registerQwenTokenPlanQuotaFetcher } from "@omniroute/open-sse/services/qwenTokenPlanQuotaFetcher.ts"; import { registerCrofUsageFetcher } from "@omniroute/open-sse/services/crofUsageFetcher.ts"; import { registerDeepseekQuotaFetcher } from "@omniroute/open-sse/services/deepseekQuotaFetcher.ts"; +import { registerMoonshotQuotaFetcher, registerMoonshotFetchersForNodes } from "@omniroute/open-sse/services/moonshotQuotaFetcher.ts"; import { registerOpenrouterQuotaFetcher } from "@omniroute/open-sse/services/openrouterQuotaFetcher.ts"; import { registerOpencodeQuotaFetcher } from "@omniroute/open-sse/services/opencodeQuotaFetcher.ts"; import { registerGrokWebQuotaFetcher } from "@omniroute/open-sse/services/grokQuotaFetcher.ts"; @@ -221,6 +222,21 @@ registerCrofUsageFetcher(); // Register DeepSeek balance quota fetcher. // Hooks into quotaPreflight + quotaMonitor so combos can switch accounts before balance is exhausted. registerDeepseekQuotaFetcher(); +registerMoonshotQuotaFetcher(); +void import("@/lib/db/providers") + .then(({ getProviderNodes }) => getProviderNodes()) + .then((nodes) => { + registerMoonshotFetchersForNodes( + (Array.isArray(nodes) ? nodes : []).map((node) => ({ + id: typeof node.id === "string" ? node.id : null, + prefix: typeof node.prefix === "string" ? node.prefix : null, + baseUrl: typeof node.baseUrl === "string" ? node.baseUrl : null, + })), + ); + }) + .catch((error) => { + console.warn("[STARTUP] Moonshot custom-node fetcher scan skipped:", error); + }); registerOpenrouterQuotaFetcher(); // Register OpenCode quota fetcher (opencode-go / opencode / opencode-zen). diff --git a/src/sse/services/auth.ts b/src/sse/services/auth.ts index 697c8a036f..b121876376 100644 --- a/src/sse/services/auth.ts +++ b/src/sse/services/auth.ts @@ -2467,6 +2467,26 @@ export function isAgentrouterConnectionQuotaScope( ); } +async function resolveDailyResetForProvider( + provider: string | null, +): Promise<{ timezone?: unknown; hour?: unknown } | null> { + if (!provider) return null; + try { + const nodes = await getCachedProviderNodes(); + const node = nodes.find((candidate) => { + if (!candidate) return false; + return candidate.id === provider || candidate.prefix === provider; + }); + if (!node) return null; + return { + timezone: node.dailyQuotaResetTimezone, + hour: node.dailyQuotaResetHour, + }; + } catch { + return null; + } +} + /** * #10880 — cools down every connection sharing the failing connection's last * known egress IP. Best-effort and side-effect-safe by design: @@ -2683,7 +2703,10 @@ export async function markAccountUnavailable( model, provider, options.headers ?? null, - effectiveProviderProfile + effectiveProviderProfile, + null, + null, + await resolveDailyResetForProvider(provider), ); // T-PROBE: probe-origin failures (model test-all) must never remove the diff --git a/stryker.conf.json b/stryker.conf.json index 58dd1f4d4f..96f421f624 100644 --- a/stryker.conf.json +++ b/stryker.conf.json @@ -58,6 +58,7 @@ "tests/unit/account-fallback-retry-after-json.test.ts", "tests/unit/account-fallback-route-restriction-403.test.ts", "tests/unit/account-fallback-service.test.ts", + "tests/unit/moonshot-quota-writeback.test.ts", "tests/unit/accountfallback-ratelimit-400-4976.test.ts", "tests/unit/adaptive-admission-route-matrix.test.ts", "tests/unit/adaptive-admission-runtime.test.ts", diff --git a/tests/unit/account-fallback-service.test.ts b/tests/unit/account-fallback-service.test.ts index d28075545a..d2693a5acf 100644 --- a/tests/unit/account-fallback-service.test.ts +++ b/tests/unit/account-fallback-service.test.ts @@ -2005,3 +2005,51 @@ test("#10460 acceptance: unambiguous model-unsupported 400 makes exactly ONE ups ); } }); + +const MOONSHOT_COMPAT = "openai-compatible-chat-e2971611-bc02-4c37-8fc5-39b8e3906fdf"; +const MOONSHOT_TPD = + "[429]: Your account org-x / proj-y request reached organization TPD rate limit, current: 1537190, limit: 1500000"; +const MOONSHOT_BROKE = + "[429]: Your account org-x is suspended due to insufficient balance, please recharge your account or check your plan and billing details"; + +test("checkFallbackError: compatible Moonshot insufficient balance is creditsExhausted", () => { + const result = checkFallbackError(429, MOONSHOT_BROKE, 0, null, MOONSHOT_COMPAT); + assert.equal(result.creditsExhausted, true); + assert.equal(result.reason, RateLimitReason.QUOTA_EXHAUSTED); +}); + +test("checkFallbackError: compatible node empty wallet without billing-suspend phrasing is creditsExhausted", () => { + const result = checkFallbackError( + 429, + "You have insufficient balance, please recharge your account", + 0, + null, + MOONSHOT_COMPAT, + ); + assert.equal(result.creditsExhausted, true); + assert.equal(result.reason, RateLimitReason.QUOTA_EXHAUSTED); +}); + +test("isDailyQuotaExhausted detects organization TPD rate limit", () => { + const { isDailyQuotaExhausted } = accountFallback; + assert.equal(isDailyQuotaExhausted(MOONSHOT_TPD), true); + assert.equal(isDailyQuotaExhausted("The engine is currently overloaded"), false); +}); + +test("checkFallbackError: TPD with node clock uses that instant, not host midnight", () => { + const now = Date.parse("2026-09-02T07:30:00Z"); + const result = checkFallbackError(429, MOONSHOT_TPD, 0, null, MOONSHOT_COMPAT, null, null, null, null, { + timezone: "Asia/Shanghai", + hour: 0, + nowMs: now, + }); + assert.equal(result.dailyQuotaExhausted, true); + assert.equal(result.cooldownMs, Date.parse("2026-09-02T16:00:00Z") - now); + assert.equal(result.reason, RateLimitReason.QUOTA_EXHAUSTED); +}); + +test("checkFallbackError: TPD without clock or header is NOT host-midnight lock", () => { + const result = checkFallbackError(429, MOONSHOT_TPD, 0, null, MOONSHOT_COMPAT); + assert.notEqual(result.dailyQuotaExhausted, true); + assert.ok(result.cooldownMs < 2 * 60 * 60 * 1000); +}); diff --git a/tests/unit/api-key-self-service.test.ts b/tests/unit/api-key-self-service.test.ts index 6b768f7950..242c3b22b3 100644 --- a/tests/unit/api-key-self-service.test.ts +++ b/tests/unit/api-key-self-service.test.ts @@ -435,3 +435,39 @@ test("self-service status normalizes Codex account quota for one explicit connec }, }); }); + +test("self-service fetches Moonshot custom-node quota via providerSpecificData host", async () => { + const metadata = { + id: "key-mnative", + name: "moonshot native", + scopes: [SELF_USAGE_SCOPE, SELF_ACCOUNT_QUOTA_SCOPE], + allowedConnections: ["conn-mnative"], + }; + const fetches: string[] = []; + const { deps } = makeDeps({ + getProviderConnectionById: async (connectionId: string) => ({ + id: connectionId, + provider: "openai-compatible-chat-e2971611-bc02-4c37-8fc5-39b8e3906fdf", + isActive: true, + providerSpecificData: { baseUrl: "https://api.moonshot.cn/v1" }, + }), + fetchAndPersistProviderLimits: async (connectionId: string) => { + fetches.push(connectionId); + return { + connection: { id: connectionId, provider: "openai-compatible-chat-e2971611-bc02-4c37-8fc5-39b8e3906fdf" }, + usage: { + plan: "Kimi 开放平台(国内)", + quotas: { + available: { remaining: 15, remainingPercentage: 100, unlimited: true, currency: "CNY" }, + }, + }, + cache: { quotas: null, plan: null, message: null, fetchedAt: "" }, + }; + }, + }); + + const status = await buildApiKeySelfServiceStatus(metadata, deps); + assert.deepEqual(fetches, ["conn-mnative"]); + assert.equal(status.accountQuotas[0].unavailable, undefined); + assert.equal(status.accountQuotas[0].plan, "Kimi 开放平台(国内)"); +}); diff --git a/tests/unit/classify429.test.ts b/tests/unit/classify429.test.ts index d75d7f648f..6d42875afe 100644 --- a/tests/unit/classify429.test.ts +++ b/tests/unit/classify429.test.ts @@ -443,3 +443,18 @@ test("classify429: retryDelay outside a RetryInfo detail is ignored", () => { }; assert.equal(classify429({ status: 429, body }), "quota_exhausted"); }); + +const MOONSHOT_TPD = + "Your account org-73b383ab6d45484eb2ef72161074495c / proj-30863601b47548bdbbabc42ff4c72eee request reached organization TPD rate limit, current: 1537190, limit: 1500000"; + +test("classify429: Moonshot organization TPD rate limit is quota_exhausted", () => { + assert.equal(classify429({ status: 429, body: MOONSHOT_TPD }), "quota_exhausted"); + assert.equal(looksLikeQuotaExhausted(MOONSHOT_TPD), true); +}); + +test("classify429: Moonshot engine overloaded stays rate_limit", () => { + assert.equal( + classify429({ status: 429, body: "The engine is currently overloaded, please try again later" }), + "rate_limit", + ); +}); diff --git a/tests/unit/daily-quota-reset.test.ts b/tests/unit/daily-quota-reset.test.ts new file mode 100644 index 0000000000..0d2c78cfaf --- /dev/null +++ b/tests/unit/daily-quota-reset.test.ts @@ -0,0 +1,54 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { + isValidIanaTimeZone, + isValidResetHour, + nodeDailyResetConfigured, + nextDailyResetAtMs, + parseTpdLimitFromText, +} from "../../open-sse/services/dailyQuotaReset.ts"; + +test("IANA: Asia/Shanghai ok, garbage rejected", () => { + assert.equal(isValidIanaTimeZone("Asia/Shanghai"), true); + assert.equal(isValidIanaTimeZone("America/New_York"), true); + assert.equal(isValidIanaTimeZone("Not/AZone"), false); + assert.equal(isValidIanaTimeZone(""), false); +}); + +test("isValidResetHour accepts 0-23 integers only", () => { + assert.equal(isValidResetHour(0), true); + assert.equal(isValidResetHour(23), true); + assert.equal(isValidResetHour(24), false); + assert.equal(isValidResetHour(-1), false); + assert.equal(isValidResetHour(1.5), false); + assert.equal(isValidResetHour(null), false); +}); + +test("nodeDailyResetConfigured requires both fields", () => { + assert.equal(nodeDailyResetConfigured("Asia/Shanghai", 0), true); + assert.equal(nodeDailyResetConfigured("Asia/Shanghai", null), false); + assert.equal(nodeDailyResetConfigured(null, 0), false); + assert.equal(nodeDailyResetConfigured("Not/AZone", 0), false); + assert.equal(nodeDailyResetConfigured("Asia/Shanghai", 24), false); +}); + +test("nextDailyResetAtMs locks to next local hour:00", () => { + // 2026-09-02 15:30 in Asia/Shanghai = 2026-09-02 07:30 UTC + const now = Date.parse("2026-09-02T07:30:00Z"); + const next = nextDailyResetAtMs("Asia/Shanghai", 0, now); + // next calendar day 00:00 Shanghai = 2026-09-02 16:00 UTC + assert.equal(next, Date.parse("2026-09-02T16:00:00Z")); +}); + +test("nextDailyResetAtMs at exact reset instant returns the following cycle", () => { + const exactly = Date.parse("2026-09-02T16:00:00Z"); // 00:00 Shanghai + const next = nextDailyResetAtMs("Asia/Shanghai", 0, exactly); + assert.equal(next, Date.parse("2026-09-03T16:00:00Z")); +}); + +test("parseTpdLimitFromText reads limit: from live body", () => { + const body = + "request reached organization TPD rate limit, current: 1537190, limit: 1500000"; + assert.equal(parseTpdLimitFromText(body), 1_500_000); + assert.equal(parseTpdLimitFromText("no numbers"), null); +}); diff --git a/tests/unit/moonshot-open-platform.test.ts b/tests/unit/moonshot-open-platform.test.ts new file mode 100644 index 0000000000..c44e1d4859 --- /dev/null +++ b/tests/unit/moonshot-open-platform.test.ts @@ -0,0 +1,72 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { + parseMoonshotOrigin, + moonshotBalanceUrl, + resolveMoonshotOrigin, + isMoonshotOpenPlatformConnection, +} from "../../open-sse/services/usage/moonshotOpenPlatform.ts"; + +const CN = "https://api.moonshot.cn/v1"; +const AI = "https://api.moonshot.ai/v1"; +const COMPAT = "openai-compatible-chat-e2971611-bc02-4c37-8fc5-39b8e3906fdf"; + +test("parseMoonshotOrigin accepts cn and ai hosts only", () => { + assert.equal(parseMoonshotOrigin(CN), "https://api.moonshot.cn"); + assert.equal( + parseMoonshotOrigin("https://api.moonshot.cn/v1/chat/completions"), + "https://api.moonshot.cn", + ); + assert.equal(parseMoonshotOrigin(AI), "https://api.moonshot.ai"); + assert.equal(parseMoonshotOrigin("https://api.openai.com/v1"), null); + assert.equal(parseMoonshotOrigin("https://api.kimi.com/coding/v1"), null); + assert.equal(parseMoonshotOrigin(""), null); + assert.equal(parseMoonshotOrigin(null), null); +}); + +test("moonshotBalanceUrl stays on the connection origin", () => { + assert.equal( + moonshotBalanceUrl("https://api.moonshot.cn"), + "https://api.moonshot.cn/v1/users/me/balance", + ); + assert.equal( + moonshotBalanceUrl("https://api.moonshot.ai"), + "https://api.moonshot.ai/v1/users/me/balance", + ); +}); + +test("resolveMoonshotOrigin prefers psd.baseUrl over node", () => { + const origin = resolveMoonshotOrigin( + { + provider: COMPAT, + providerSpecificData: { baseUrl: CN }, + }, + "https://api.moonshot.ai/v1", + ); + assert.equal(origin, "https://api.moonshot.cn"); +}); + +test("resolveMoonshotOrigin uses node baseUrl when psd has none", () => { + const origin = resolveMoonshotOrigin({ provider: COMPAT, providerSpecificData: {} }, CN); + assert.equal(origin, "https://api.moonshot.cn"); +}); + +test("resolveMoonshotOrigin uses built-in moonshot/kimi registry host", () => { + assert.equal(resolveMoonshotOrigin({ provider: "moonshot" }), "https://api.moonshot.ai"); + assert.equal(resolveMoonshotOrigin({ provider: "kimi" }), "https://api.moonshot.ai"); +}); + +test("isMoonshotOpenPlatformConnection is true for mnative-shaped rows", () => { + assert.equal( + isMoonshotOpenPlatformConnection({ + provider: COMPAT, + providerSpecificData: { baseUrl: CN, prefix: "mnative" }, + }), + true, + ); + assert.equal( + isMoonshotOpenPlatformConnection({ provider: "deepseek", providerSpecificData: {} }), + false, + ); + assert.equal(isMoonshotOpenPlatformConnection({ provider: "moonshot" }), true); +}); diff --git a/tests/unit/moonshot-quota-fetcher.test.ts b/tests/unit/moonshot-quota-fetcher.test.ts new file mode 100644 index 0000000000..1c315632b9 --- /dev/null +++ b/tests/unit/moonshot-quota-fetcher.test.ts @@ -0,0 +1,158 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { + fetchMoonshotQuota, + invalidateMoonshotQuotaCache, + registerMoonshotQuotaFetcher, + getMoonshotOpenPlatformUsage, +} from "../../open-sse/services/moonshotQuotaFetcher.ts"; +import { getUsageForProvider } from "../../open-sse/services/usage.ts"; +import { getQuotaFetcher } from "../../open-sse/services/quotaPreflight.ts"; + +const originalFetch = globalThis.fetch; +const CN = "https://api.moonshot.cn/v1"; +const COMPAT = "openai-compatible-chat-e2971611-bc02-4c37-8fc5-39b8e3906fdf"; + +test.afterEach(() => { + globalThis.fetch = originalFetch; +}); + +function jsonResponse(body: unknown, status = 200): Response { + return new Response(JSON.stringify(body), { + status, + headers: { "Content-Type": "application/json" }, + }); +} + +test("fetchMoonshotQuota parses available_balance > 0 as not exhausted", async () => { + const connectionId = `ms-pos-${Date.now()}`; + globalThis.fetch = async (url) => { + assert.equal(String(url), "https://api.moonshot.cn/v1/users/me/balance"); + return jsonResponse({ + code: 0, + data: { available_balance: 2.5, voucher_balance: 0, cash_balance: 2.5 }, + status: true, + }); + }; + const q = await fetchMoonshotQuota(connectionId, { + apiKey: "sk-test", + providerSpecificData: { baseUrl: CN }, + }); + assert.equal(q?.limitReached, false); + assert.equal(q?.percentUsed, 0); + invalidateMoonshotQuotaCache(connectionId); +}); + +test("fetchMoonshotQuota treats available_balance 0 as exhausted", async () => { + const connectionId = `ms-zero-${Date.now()}`; + globalThis.fetch = async () => + jsonResponse({ + code: 0, + data: { available_balance: 0, voucher_balance: 0, cash_balance: 0 }, + status: true, + }); + const q = await fetchMoonshotQuota(connectionId, { + apiKey: "sk-test", + providerSpecificData: { baseUrl: CN }, + }); + assert.equal(q?.limitReached, true); + invalidateMoonshotQuotaCache(connectionId); +}); + +test("fetchMoonshotQuota returns null on 401", async () => { + const connectionId = `ms-401-${Date.now()}`; + globalThis.fetch = async () => new Response(null, { status: 401 }); + const q = await fetchMoonshotQuota(connectionId, { + apiKey: "sk-test", + providerSpecificData: { baseUrl: CN }, + }); + assert.equal(q, null); +}); + +test("getUsageForProvider on custom uuid hits Moonshot path", async () => { + const connectionId = `ms-uuid-${Date.now()}`; + globalThis.fetch = async () => + jsonResponse({ + code: 0, + data: { available_balance: 15, voucher_balance: 15, cash_balance: 0 }, + status: true, + }); + const usage = await getUsageForProvider({ + id: connectionId, + provider: COMPAT, + apiKey: "sk-test", + providerSpecificData: { baseUrl: CN }, + }); + assert.equal(typeof usage === "object" && usage && "message" in usage, false); + assert.equal((usage as { plan?: string }).plan, "Kimi 开放平台(国内)"); + invalidateMoonshotQuotaCache(connectionId); +}); + +test("getMoonshotOpenPlatformUsage uses Open Platform plan label for .ai host", async () => { + const connectionId = `ms-ai-${Date.now()}`; + globalThis.fetch = async (url) => { + assert.equal(String(url), "https://api.moonshot.ai/v1/users/me/balance"); + return jsonResponse({ + code: 0, + data: { available_balance: 1, voucher_balance: 0, cash_balance: 1 }, + status: true, + }); + }; + const usage = await getMoonshotOpenPlatformUsage({ + id: connectionId, + provider: "moonshot", + apiKey: "sk-test", + providerSpecificData: { baseUrl: "https://api.moonshot.ai/v1" }, + }); + assert.equal(usage.plan, "Kimi Open Platform"); + invalidateMoonshotQuotaCache(connectionId); +}); + +test("domestic Moonshot balance is labeled CNY, international USD", async () => { + const cnId = `ms-cny-${Date.now()}`; + globalThis.fetch = async () => + jsonResponse({ + code: 0, + data: { available_balance: 15, voucher_balance: 15, cash_balance: 0 }, + status: true, + }); + const cn = await getMoonshotOpenPlatformUsage({ + id: cnId, + provider: COMPAT, + apiKey: "sk-test", + providerSpecificData: { baseUrl: CN }, + }); + assert.equal(cn.quotas?.available?.currency, "CNY"); + invalidateMoonshotQuotaCache(cnId); + + const aiId = `ms-usd-${Date.now()}`; + const ai = await getMoonshotOpenPlatformUsage({ + id: aiId, + provider: "moonshot", + apiKey: "sk-test", + providerSpecificData: { baseUrl: "https://api.moonshot.ai/v1" }, + }); + assert.equal(ai.quotas?.available?.currency, "USD"); + invalidateMoonshotQuotaCache(aiId); +}); + +test("registerMoonshotQuotaFetcher wires moonshot and kimi ids", () => { + registerMoonshotQuotaFetcher(); + assert.equal(typeof getQuotaFetcher("moonshot"), "function"); + assert.equal(typeof getQuotaFetcher("kimi"), "function"); +}); + +test("registerMoonshotFetchersForNodes registers custom node id and prefix", async () => { + const { registerMoonshotFetchersForNodes } = await import( + "../../open-sse/services/moonshotQuotaFetcher.ts" + ); + const uuid = "openai-compatible-chat-e2971611-bc02-4c37-8fc5-39b8e3906fdf"; + registerMoonshotFetchersForNodes([ + { id: uuid, prefix: "mnative", baseUrl: CN }, + { id: "other", prefix: "oc-prod", baseUrl: "https://api.openai.com/v1" }, + ]); + assert.equal(typeof getQuotaFetcher(uuid), "function"); + assert.equal(typeof getQuotaFetcher("mnative"), "function"); + assert.equal(getQuotaFetcher("oc-prod"), undefined); +}); diff --git a/tests/unit/moonshot-quota-writeback.test.ts b/tests/unit/moonshot-quota-writeback.test.ts new file mode 100644 index 0000000000..14c8e3168b --- /dev/null +++ b/tests/unit/moonshot-quota-writeback.test.ts @@ -0,0 +1,50 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { classify429 } from "../../src/shared/utils/classify429.ts"; +import { checkFallbackError } from "../../open-sse/services/accountFallback.ts"; +import { registerMoonshotFetchersForNodes } from "../../open-sse/services/moonshotQuotaFetcher.ts"; +import { getQuotaFetcher } from "../../open-sse/services/quotaPreflight.ts"; + +const MOONSHOT_TPD = + "Your account org-73b383ab6d45484eb2ef72161074495c / proj-30863601b47548bdbbabc42ff4c72eee request reached organization TPD rate limit, current: 1537190, limit: 1500000"; +const MOONSHOT_BROKE = "insufficient balance"; +const COMPAT = "openai-compatible-chat-e2971611-bc02-4c37-8fc5-39b8e3906fdf"; + +test("classify429 maps Moonshot TPD to quota_exhausted so combo persist stays on", () => { + assert.equal(classify429({ status: 429, body: MOONSHOT_TPD }), "quota_exhausted"); +}); + +test("checkFallbackError TPD with node clock returns future cooldown, not host midnight", () => { + const now = Date.parse("2026-09-02T07:30:00Z"); + const result = checkFallbackError( + 429, + MOONSHOT_TPD, + 0, + "kimi-k2.5", + COMPAT, + null, + null, + null, + null, + { timezone: "Asia/Shanghai", hour: 0, nowMs: now }, + ); + assert.equal(result.shouldFallback, true); + assert.equal(result.dailyQuotaExhausted, true); + assert.ok(result.cooldownMs > 8 * 3600_000); + const until = now + result.cooldownMs; + assert.ok(Math.abs(until - Date.parse("2026-09-02T16:00:00Z")) < 60_000); +}); + +test("checkFallbackError insufficient balance on compatible node is creditsExhausted", () => { + const result = checkFallbackError(429, MOONSHOT_BROKE, 0, "kimi-k2.5", COMPAT); + assert.equal(result.creditsExhausted, true); + assert.equal(result.shouldFallback, true); +}); + +test("registerMoonshotFetchersForNodes wires uuid and prefix after startup scan", () => { + registerMoonshotFetchersForNodes([ + { id: COMPAT, prefix: "mnative", baseUrl: "https://api.moonshot.cn/v1" }, + ]); + assert.equal(typeof getQuotaFetcher(COMPAT), "function"); + assert.equal(typeof getQuotaFetcher("mnative"), "function"); +}); diff --git a/tests/unit/provider-node-daily-reset-schema.test.ts b/tests/unit/provider-node-daily-reset-schema.test.ts new file mode 100644 index 0000000000..f20a918b68 --- /dev/null +++ b/tests/unit/provider-node-daily-reset-schema.test.ts @@ -0,0 +1,54 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +const { createProviderNodeSchema, updateProviderNodeSchema } = + await import("../../src/shared/validation/schemas/provider.ts"); + +const BASE = { + name: "Moonshot native", + prefix: "mnative", + baseUrl: "https://api.moonshot.cn/v1", +}; + +test("updateProviderNodeSchema accepts IANA timezone + hour 0", () => { + const result = updateProviderNodeSchema.safeParse({ + ...BASE, + dailyQuotaResetTimezone: "Asia/Shanghai", + dailyQuotaResetHour: 0, + }); + assert.equal(result.success, true); +}); + +test("updateProviderNodeSchema rejects unknown timezone", () => { + const result = updateProviderNodeSchema.safeParse({ + ...BASE, + dailyQuotaResetTimezone: "Shanghai", + dailyQuotaResetHour: 0, + }); + assert.equal(result.success, false); +}); + +test("updateProviderNodeSchema rejects hour 24", () => { + const result = updateProviderNodeSchema.safeParse({ + ...BASE, + dailyQuotaResetTimezone: "Asia/Shanghai", + dailyQuotaResetHour: 24, + }); + assert.equal(result.success, false); +}); + +test("updateProviderNodeSchema accepts omitted reset clock", () => { + const result = updateProviderNodeSchema.safeParse(BASE); + assert.equal(result.success, true); +}); + +test("createProviderNodeSchema accepts IANA timezone + hour", () => { + const result = createProviderNodeSchema.safeParse({ + ...BASE, + apiType: "chat", + type: "openai-compatible", + dailyQuotaResetTimezone: "UTC", + dailyQuotaResetHour: 0, + }); + assert.equal(result.success, true); +}); diff --git a/tests/unit/provider-quota-visibility.test.ts b/tests/unit/provider-quota-visibility.test.ts index 665687feed..a8551e3ccb 100644 --- a/tests/unit/provider-quota-visibility.test.ts +++ b/tests/unit/provider-quota-visibility.test.ts @@ -16,3 +16,15 @@ test("quota visibility controls are limited to providers with quota support", () assert.equal(supportsProviderQuota("codex"), true); assert.equal(supportsProviderQuota("openai"), false); }); + +test("supportsProviderQuota is true for moonshot-native shaped connection", () => { + assert.equal( + supportsProviderQuota("openai-compatible-chat-e2971611-bc02-4c37-8fc5-39b8e3906fdf", { + providerSpecificData: { baseUrl: "https://api.moonshot.cn/v1" }, + }), + true, + ); + assert.equal(supportsProviderQuota("moonshot"), true); + assert.equal(supportsProviderQuota("kimi"), true); + assert.equal(supportsProviderQuota("openai"), false); +}); diff --git a/tests/unit/qoder-usage-quota.test.ts b/tests/unit/qoder-usage-quota.test.ts index 14666abc11..8e69a8991e 100644 --- a/tests/unit/qoder-usage-quota.test.ts +++ b/tests/unit/qoder-usage-quota.test.ts @@ -148,6 +148,15 @@ test("a qoder PAT (apikey) connection is picked up by the provider-limits sync", isSupportedUsageConnection({ id: "c2", provider: "some-random-provider", authType: "apikey" }), false ); + assert.equal( + isSupportedUsageConnection({ + id: "c3", + provider: "openai-compatible-chat-e2971611-bc02-4c37-8fc5-39b8e3906fdf", + authType: "apikey", + providerSpecificData: { baseUrl: "https://api.moonshot.cn/v1" }, + }), + true, + ); }); // Guards the shared exchange contract the usage path relies on. From d9526cefeaea1f4836a947f4a56944fdaf4e8370 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 12:49:53 -0300 Subject: [PATCH 041/143] chore(quality): rebaseline file-size caps the HouMinXi batch grew past (#12619) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rebaseline medido no tip com os 9 PRs da leva mergeados. Desfaz o vermelho de file-size que os PRs empilhados deixaram; não toca codex.ts nem stream.ts, que já violavam antes da leva. --- changelog.d/maintenance/houminxi-batch-filesize.md | 1 + config/quality/file-size-baseline.json | 9 +++++---- 2 files changed, 6 insertions(+), 4 deletions(-) create mode 100644 changelog.d/maintenance/houminxi-batch-filesize.md diff --git a/changelog.d/maintenance/houminxi-batch-filesize.md b/changelog.d/maintenance/houminxi-batch-filesize.md new file mode 100644 index 0000000000..0d3426ed51 --- /dev/null +++ b/changelog.d/maintenance/houminxi-batch-filesize.md @@ -0,0 +1 @@ +- **chore(quality):** rebaseline the file-size caps the HouMinXi batch grew past when its PRs stacked (`providers/page.tsx`, `chatCore.ts`, `accountFallback.ts`) — each PR measured correctly in isolation, none saw the stacking diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index cfa43566b4..86ea2f3a86 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -415,12 +415,12 @@ "open-sse/executors/codex.ts": 1499, "open-sse/executors/cursor.ts": 1759, "open-sse/executors/muse-spark-web.ts": 1405, - "open-sse/handlers/chatCore.ts": 5981, + "open-sse/handlers/chatCore.ts": 5984, "open-sse/handlers/imageGeneration.ts": 3259, "open-sse/handlers/search.ts": 1789, "open-sse/mcp-server/schemas/tools.ts": 1621, "open-sse/mcp-server/server.ts": 1572, - "open-sse/services/accountFallback.ts": 2461, + "open-sse/services/accountFallback.ts": 2467, "open-sse/services/adobeFireflyBrowserLogin.ts": 1401, "open-sse/services/combo.ts": 4023, "open-sse/translator/response/openai-responses.ts": 1466, @@ -435,7 +435,7 @@ "src/app/(dashboard)/dashboard/costs/CostOverviewTab.tsx": 1319, "src/app/(dashboard)/dashboard/endpoint/EndpointPageClient.tsx": 2491, "src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx": 1631, - "src/app/(dashboard)/dashboard/providers/page.tsx": 2007, + "src/app/(dashboard)/dashboard/providers/page.tsx": 2025, "src/app/(dashboard)/dashboard/runtime/RuntimePageClient.tsx": 1201, "src/app/(dashboard)/dashboard/settings/components/ProxyRegistryManager.tsx": 1475, "src/app/(dashboard)/dashboard/settings/components/ResilienceTab.tsx": 1271, @@ -633,5 +633,6 @@ "open-sse/executors/chatgpt-web.ts": "3241", "_rebaseline_2026_08_30_11771_vercel_gateway_passthrough": "PR #11771 adds passthroughModels: true (1 line) to the Vercel AI Gateway registry entry — no split available, single-line provider-config addition.", "_relax_velocity_2026_08_30": "127 frozen line caps and cap/testCap raised by 20% (velocity phase; see quality-baseline.json _policy).", - "_rebaseline_2026_09_02_12325_merge_v3851": "Merge of release/v3.8.51 into #12325. Both sides grew chatCore.ts at the same chokepoint: #12239 took it 5946->5976 upstream, and this PR adds its +9 non-Codex 429 branch on top. check-file-size.mjs counts split(\"\\\\n\").length (trailing-newline empty element), so the merged file is 5981. The cap is the merged LOC, not either side alone; no other entry moves." + "_rebaseline_2026_09_02_12325_merge_v3851": "Merge of release/v3.8.51 into #12325. Both sides grew chatCore.ts at the same chokepoint: #12239 took it 5946->5976 upstream, and this PR adds its +9 non-Codex 429 branch on top. check-file-size.mjs counts split(\"\\\\n\").length (trailing-newline empty element), so the merged file is 5981. The cap is the merged LOC, not either side alone; no other entry moves.", + "_rebaseline_2026_09_03_houminxi_batch_stacked": "Crescimento medido DEPOIS que os 9 PRs da leva HouMinXi entraram, quando cada um empilhou sobre o rebaseline do anterior: providers/page.tsx 2007->2025 (+18 = feedback de erro por linha do import CSV do #12504 somado a busca por nome/baseUrl do #12495, ambos no mesmo painel de conexoes); chatCore.ts 5981->5984 (+3 = o #12325 invalida o cache generico de quota no 429 upstream, ao lado do ramo Codex ja existente); accountFallback.ts 2461->2467 (+6 = o #12566 empilha a carve-out de familia Antigravity sobre o rebaseline 2422->2461 que o #12590 registrou para o carve-out credits_exhausted da Moonshot; os dois tocam checkFallbackError). Cada PR mediu certo isoladamente, mas nenhum enxergava o empilhamento. Fiacao em chokepoints existentes. NAO cobre codex.ts nem stream.ts, que ja violavam no tip antes desta leva (drift da base)." } From d353870342590c270641bae07c937ec3c5e7614a Mon Sep 17 00:00:00 2001 From: Markus Hartung Date: Thu, 3 Sep 2026 18:00:08 +0200 Subject: [PATCH 042/143] fix(resourcePressure): log numeric detail on every rejection, recover faster (#12293) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51` (já com a leva anterior dentro): os nove boardaram **sem um único conflito**, `typecheck:core` limpo e **80/80** nos 6 arquivos de teste que os PRs trazem. O crescimento de arquivo próprio da leva foi rebaselinado num registro datado (`_rebaseline_2026_09_03_hartmark_batch`): `combos/page.tsx` 5012→5018 (#12355, tratar o estado degradado quando o bundling de tiktoken de um provider sem relação falha) e `open-sse/services/combo.ts` 4023→4036 (#12338, os fixes do universal-handoff). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. Obrigado, @hartmark. --- open-sse/utils/resourcePressure.ts | 61 +++++++++++++++++++++--- open-sse/utils/resourcePressurePolicy.ts | 26 ++++++++-- 2 files changed, 77 insertions(+), 10 deletions(-) diff --git a/open-sse/utils/resourcePressure.ts b/open-sse/utils/resourcePressure.ts index acef067a1e..f3ba7f77d9 100644 --- a/open-sse/utils/resourcePressure.ts +++ b/open-sse/utils/resourcePressure.ts @@ -67,9 +67,47 @@ function requireDuration(name: string, value: number): number { return value; } -function buildCriticalGuard(reason: PressureReason): ResourcePressureGuardResult { +/** + * Human-readable key=value detail appended to the rejection log line. Every + * rejection (immediate heap trip AND cached-critical-state reuse) goes + * through here, so this is the one place that needs the actual numbers — + * the bare reason code alone ("psi_some") gives an operator nothing to act + * on when deciding whether the guard is mistuned vs. genuinely saturated. + */ +function formatPressureDetail(detail: Record): string { + return Object.entries(detail) + .filter(([, value]) => value !== undefined) + .map(([key, value]) => `${key}=${value ?? "null"}`) + .join(" "); +} + +/** Builds buildCriticalGuard's detail object for the cached-critical-state + * reuse path in check() -- pulled out of check() itself so that function's + * own cyclomatic complexity stays under the ratchet, not because this needs + * to be reused anywhere else. */ +function describeCachedPressure(params: { + signals: ResourceSignals | null; + recoveryStreak: number; + cacheAgeMs: number; +}): Record { + const cgroup = params.signals?.cgroup; + return { + psiSomeAvg10: params.signals?.psi?.someAvg10 ?? null, + psiFullAvg10: params.signals?.psi?.fullAvg10 ?? null, + cgroupCurrentMb: cgroup?.currentBytes ? Math.round(cgroup.currentBytes / MB) : null, + cgroupMaxMb: cgroup?.maxBytes ? Math.round(cgroup.maxBytes / MB) : null, + recoveryStreak: params.recoveryStreak, + sampleAgeMs: params.cacheAgeMs, + }; +} + +function buildCriticalGuard( + reason: PressureReason, + detail: Record = {} +): ResourcePressureGuardResult { + const detailText = formatPressureDetail(detail); console.warn( - `[resourcePressure] critical pressure guard tripped (reason=${reason}); returning 503` + `[resourcePressure] critical pressure guard tripped (reason=${reason}${detailText ? " " + detailText : ""}); returning 503` ); return { success: false, @@ -97,7 +135,10 @@ function immediateHeapGuard( if (thresholdMb == null) return null; const guard = checkHeapPressureGuard(heapUsedMb, thresholdMb); if (!guard) return null; - return buildCriticalGuard("v8_heap_absolute"); + return buildCriticalGuard("v8_heap_absolute", { + heapUsedMb: Math.round(heapUsedMb), + thresholdMb: Math.round(thresholdMb), + }); } export function createResourcePressureRuntime( @@ -192,9 +233,17 @@ export function createResourcePressureRuntime( return immediate; } const cacheAge = lastSignals ? Math.max(0, now - lastRefreshAtMs) : Number.POSITIVE_INFINITY; - return cacheAge <= maxStaleMs && state.severity === "critical" - ? buildCriticalGuard(state.reason) - : null; + if (cacheAge > maxStaleMs || state.severity !== "critical") { + return null; + } + return buildCriticalGuard( + state.reason, + describeCachedPressure({ + signals: lastSignals, + recoveryStreak: state.recoveryStreak, + cacheAgeMs: cacheAge, + }) + ); }, getObservation: () => ({ signals: lastSignals, state }), whenRefreshSettled: async () => { diff --git a/open-sse/utils/resourcePressurePolicy.ts b/open-sse/utils/resourcePressurePolicy.ts index 447a084b61..3887bf9bbc 100644 --- a/open-sse/utils/resourcePressurePolicy.ts +++ b/open-sse/utils/resourcePressurePolicy.ts @@ -74,12 +74,30 @@ export const DEFAULT_RESOURCE_PRESSURE_THRESHOLDS: ResourcePressureThresholds = highRatio: 0.85, criticalRatio: 0.92, recoveryRatio: 0.75, - highPsiAvg10: 20, - criticalPsiAvg10: 40, - recoveryPsiAvg10: 10, + // Bumped 50% (20/40/10 -> 30/60/15): /proc/pressure/memory reflects + // HOST-wide PSI, not this process's own cgroup pressure (confirmed by + // comparing /proc/pressure/memory against /sys/fs/cgroup/memory.pressure + // from inside a running container -- the two differ). On a shared host + // running many unrelated workloads, host-wide memory contention from + // OTHER processes was tripping this guard even while OmniRoute's own + // usage stayed trivial. The ratio-based thresholds above stay untouched + // -- they're this process's own real OOM safety margin and unaffected by + // noisy neighbors. + highPsiAvg10: 30, + criticalPsiAvg10: 60, + recoveryPsiAvg10: 15, sustainedSamplesHigh: 2, sustainedSamplesCritical: 2, - sustainedSamplesRecovery: 3, + // PSI's own avg10 is a kernel-computed 10s rolling average, so it already + // lags real recovery by design -- requiring 3 consecutive samples *on top* + // of that (at the ~1s default sample cadence) stacked another ~2-3s of + // guard-still-shedding time after the process was actually fine again. + // isRecovered() already requires every tracked ratio/PSI value to clear + // the separate, more conservative recoveryRatio/recoveryPsiAvg10 + // thresholds (not just dip under the critical ones), so a single clean + // sample is real signal, not noise -- the streak requirement was adding + // redundant delay on top of an already-conservative bar. + sustainedSamplesRecovery: 1, heapAbsoluteThresholdMb: null, }; From c091534ffcde1279767060dd5f9bb6735a99d65a Mon Sep 17 00:00:00 2001 From: Markus Hartung Date: Thu, 3 Sep 2026 18:00:27 +0200 Subject: [PATCH 043/143] fix(providers): stop an unrelated-provider tiktoken bundling failure from crashing /api/providers (#12355) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51` (já com a leva anterior dentro): os nove boardaram **sem um único conflito**, `typecheck:core` limpo e **80/80** nos 6 arquivos de teste que os PRs trazem. O crescimento de arquivo próprio da leva foi rebaselinado num registro datado (`_rebaseline_2026_09_03_hartmark_batch`): `combos/page.tsx` 5012→5018 (#12355, tratar o estado degradado quando o bundling de tiktoken de um provider sem relação falha) e `open-sse/services/combo.ts` 4023→4036 (#12338, os fixes do universal-handoff). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. Obrigado, @hartmark. --- src/app/(dashboard)/dashboard/combos/page.tsx | 16 +++++++++----- src/app/api/providers/route.ts | 21 ++++++++++++++++--- .../providers/validation/chatgptWebCodex.ts | 13 +++++++++++- 3 files changed, 41 insertions(+), 9 deletions(-) diff --git a/src/app/(dashboard)/dashboard/combos/page.tsx b/src/app/(dashboard)/dashboard/combos/page.tsx index 978a5f1321..0105dd8723 100644 --- a/src/app/(dashboard)/dashboard/combos/page.tsx +++ b/src/app/(dashboard)/dashboard/combos/page.tsx @@ -761,15 +761,21 @@ function CombosPageContent() { const [proxyConfig, setProxyConfig] = useState(null); const { comboProxyAssignedIds, fetchComboProxyAssignments } = useComboProxyAssignments(); const [providerNodes, setProviderNodes] = useState([]); - const [showUsageGuide, setShowUsageGuide] = useState(() => { - // Lazy initializer instead of a mount effect (react-hooks/set-state-in-effect). + // SSR has no localStorage, so a lazy initializer reading it here returns a + // different value server-side (always "not dismissed") than the client's + // real stored value -- exactly the kind of source React's hydration + // mismatch check is built to catch, and in dev mode a mismatch forces a + // full client-only re-render of this tree, discarding whatever the fetch + // effects below had already populated. Start with the SSR-safe default on + // both passes and correct it client-only, after hydration, in an effect. + const [showUsageGuide, setShowUsageGuide] = useState(true); + useEffect(() => { try { - return globalThis.localStorage?.getItem(COMBO_USAGE_GUIDE_STORAGE_KEY) !== "1"; + setShowUsageGuide(globalThis.localStorage?.getItem(COMBO_USAGE_GUIDE_STORAGE_KEY) !== "1"); } catch { // Ignore storage access errors (privacy mode / restricted environments) - return true; } - }); + }, []); const [recentlyCreatedCombo, setRecentlyCreatedCombo] = useState(""); const [creatingKimiPreset, setCreatingKimiPreset] = useState(false); const [comboDragIndex, setComboDragIndex] = useState(null); diff --git a/src/app/api/providers/route.ts b/src/app/api/providers/route.ts index bb5d1ded72..409f7445ab 100644 --- a/src/app/api/providers/route.ts +++ b/src/app/api/providers/route.ts @@ -47,7 +47,15 @@ import { fetchModelSyncInternal, getModelSyncInternalBaseUrl, } from "@/shared/services/modelSyncScheduler"; -import { finalizeValidatedChatGptWebCodexSecrets } from "@omniroute/open-sse/services/chatgptWebCodexAdmin.ts"; +// Dynamically imported below, inside the one `provider === "chatgpt-web-codex"` +// branch that needs it: this module's transitive chain pulls in tiktoken's +// WASM tokenizer, which Turbopack dev mode fails to resolve for this graph +// even with `tiktoken` listed in serverExternalPackages (the standalone +// Node require works fine; only Turbopack's bundling of this import path +// doesn't). A static top-level import evaluates that whole chain on EVERY +// /api/providers request regardless of provider, turning an unrelated- +// provider bug into a route-wide 500. Loading it lazily, only when actually +// needed, avoids paying that cost (and that risk) on the common path. import { isAutoFetchModelsEnabled } from "@/lib/providerModels/modelDiscovery"; import { testSingleConnection } from "./[id]/test/route"; import { rejectRetiredCommonChatGptWebProvider } from "@/lib/providers/chatgptWebRetirementResponse"; @@ -204,6 +212,8 @@ export async function POST(request: Request) { ? providerSpecificData.validationId : ""; try { + const { finalizeValidatedChatGptWebCodexSecrets } = + await import("@omniroute/open-sse/services/chatgptWebCodexAdmin.ts"); const finalized = finalizeValidatedChatGptWebCodexSecrets(apiKey || "", validationId); persistedApiKey = finalized.encodedCredential; providerSpecificData = { ...(providerSpecificData || {}) }; @@ -324,11 +334,16 @@ export async function POST(request: Request) { }) .then((syncRes) => { if (!syncRes.ok) { - console.log(`[providers] Auto-sync failed for ${newConnection.id}: ${syncRes.status}`); + console.log( + `[providers] Auto-sync failed for ${newConnection.id}: ${syncRes.status}` + ); } }) .catch((err) => { - console.log(`[providers] Auto-sync error for ${newConnection.id}:`, err?.message || err); + console.log( + `[providers] Auto-sync error for ${newConnection.id}:`, + err?.message || err + ); }); } catch (syncSetupError) { // Defensive: if URL parsing or header construction itself throws, do diff --git a/src/lib/providers/validation/chatgptWebCodex.ts b/src/lib/providers/validation/chatgptWebCodex.ts index 3dd6eda748..d95c1475f3 100644 --- a/src/lib/providers/validation/chatgptWebCodex.ts +++ b/src/lib/providers/validation/chatgptWebCodex.ts @@ -4,7 +4,6 @@ import { rmSync } from "node:fs"; import { CHATGPT_WEB_CODEX_CONNECTOR_NAME } from "@/shared/constants/chatgptWebCodex"; import { inspectBrowserLoginCapabilities } from "@omniroute/open-sse/vendor/codex-chatgpt-web/browser-login.ts"; import { decodeChatGptWebCodexSecrets } from "@omniroute/open-sse/executors/chatgpt-web-codex/credentials.ts"; -import { detectChromeExecutable } from "@omniroute/open-sse/executors/chatgpt-web-codex.ts"; import { connectionRuntimePaths, ensureConnectionStorageState, @@ -12,6 +11,16 @@ import { } from "@omniroute/open-sse/executors/chatgpt-web-codex/storageState.ts"; import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/error.ts"; +// detectChromeExecutable (executors/chatgpt-web-codex.ts) is imported +// dynamically below, not statically here: this module is re-exported through +// the shared `@/lib/providers/validation` barrel that every provider +// validator's callers pull in, and executors/chatgpt-web-codex.ts's own +// import chain (its vendor browser adapter -> token-estimate.ts -> tiktoken's +// WASM tokenizer) fails to bundle under Turbopack dev mode even with +// `tiktoken` server-externalized -- turning validation of an unrelated +// provider into a route-wide crash for anyone who merely imports the barrel. +// A static import here evaluates that whole chain unconditionally. + export async function validateChatGptWebCodexProvider({ apiKey, providerSpecificData = {}, @@ -54,6 +63,8 @@ export async function validateChatGptWebCodexProvider({ }; } const cdpEndpoint = process.env.CHATGPT_WEB_CODEX_CDP_URL?.trim(); + const { detectChromeExecutable } = + await import("@omniroute/open-sse/executors/chatgpt-web-codex.ts"); const chromeExecutablePath = detectChromeExecutable( typeof providerSpecificData.chromeExecutablePath === "string" ? providerSpecificData.chromeExecutablePath From 7881e7eb72d4e35b6485a4389b1778c40b608f76 Mon Sep 17 00:00:00 2001 From: Markus Hartung Date: Thu, 3 Sep 2026 18:00:44 +0200 Subject: [PATCH 044/143] fix(conversations): resolve turn content OmniRoute never sends back to the client (#12447) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51` (já com a leva anterior dentro): os nove boardaram **sem um único conflito**, `typecheck:core` limpo e **80/80** nos 6 arquivos de teste que os PRs trazem. O crescimento de arquivo próprio da leva foi rebaselinado num registro datado (`_rebaseline_2026_09_03_hartmark_batch`): `combos/page.tsx` 5012→5018 (#12355, tratar o estado degradado quando o bundling de tiktoken de um provider sem relação falha) e `open-sse/services/combo.ts` 4023→4036 (#12338, os fixes do universal-handoff). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. Obrigado, @hartmark. --- open-sse/services/conversationTurnContent.ts | 101 ++++++++++++++++--- 1 file changed, 88 insertions(+), 13 deletions(-) diff --git a/open-sse/services/conversationTurnContent.ts b/open-sse/services/conversationTurnContent.ts index a95a39c939..bc5f3bdcef 100644 --- a/open-sse/services/conversationTurnContent.ts +++ b/open-sse/services/conversationTurnContent.ts @@ -23,6 +23,90 @@ export type TurnDisplayContent = { toolName: string | null; }; +type CanonicalTurnLike = { + role: "system" | "user" | "assistant" | "tool"; + text: string; + blockKind: "text" | "tool_use" | "tool_result"; + toolName: string | null; +}; + +type JsonRecord = Record; + +function asRecord(value: unknown): JsonRecord | null { + return value && typeof value === "object" ? (value as JsonRecord) : null; +} + +function turnsFromBody(body: unknown): CanonicalTurnLike[] { + const rec = asRecord(body); + return rec ? extractCanonicalTurns(rec) : []; +} + +/** + * extractCanonicalTurns's Chat Completions branch only reads a message's + * `content` -- a tool-calling assistant message carries its call in + * `tool_calls` instead with `content: null`, so it silently produces no turn + * at all and the matching conversation_turn_nodes row can never resolve. + * Deliberately scoped to this read-only display path instead of extending + * extractCanonicalTurns itself: that function also drives + * conversationTracker.ts's write-path identity/hashing, and this codebase's + * only caller of it there (chat.ts's resolveConversationId) always feeds the + * client-facing Responses-API body -- never Chat Completions + * `messages`/`tool_calls` -- so extending it there would be unreachable for + * real traffic here but still carries real write-path identity-hash risk for + * any other caller/format that function might ever serve. Mirrors + * extractCanonicalTurns's own Responses-shape function_call handling: one + * turn per call, role "tool" (matches how a Responses API function_call item, + * which also carries no `role`, canonicalizes -- not "assistant"), toolName + * from the call, text the raw arguments string untouched (already a JSON + * string in both APIs, so passing it through unmodified is what a + * byte-identical hash against the original Responses-shaped item needs). + */ +function extractChatCompletionsToolUseTurns(messages: unknown): CanonicalTurnLike[] { + if (!Array.isArray(messages)) return []; + const turns: CanonicalTurnLike[] = []; + for (const item of messages) { + const rec = asRecord(item) ?? {}; + if (rec.role !== "assistant" || !Array.isArray(rec.tool_calls)) continue; + for (const call of rec.tool_calls) { + const fn = asRecord(asRecord(call)?.function); + const args = fn?.arguments; + if (typeof args !== "string" || !args) continue; + turns.push({ + role: "tool", + text: args, + blockKind: "tool_use", + toolName: typeof fn?.name === "string" ? fn.name : null, + }); + } + } + return turns; +} + +function turnsFromClientResponse(clientResponse: unknown): CanonicalTurnLike[] { + const rec = asRecord(clientResponse); + if (!rec) return []; + const summary = asRecord(rec.summary); + const output = Array.isArray(rec.output) ? rec.output : summary?.output; + return Array.isArray(output) ? extractCanonicalTurns({ input: output }) : []; +} + +function turnsFromProviderRequest(body: unknown): CanonicalTurnLike[] { + const rec = asRecord(body); + return [...turnsFromBody(rec), ...extractChatCompletionsToolUseTurns(rec?.messages)]; +} + +function indexTurns(result: Map, turns: CanonicalTurnLike[]): void { + for (const turn of turns) { + const hash = hashTurnContent(turn); + if (result.has(hash)) continue; + result.set(hash, { + textPreview: turn.text, + blockKind: turn.blockKind, + toolName: turn.toolName, + }); + } +} + /** * Resolve display content for a batch of turn nodes, keyed by content_hash. * Content_hash is sha256(role+text) only — real traffic has plenty of @@ -64,19 +148,10 @@ export function resolveTurnDisplayContent( for (const relPath of artifactPathByCorrelationId.values()) { const { artifact, state } = readCallArtifact(relPath); if (state !== "ready") continue; - const clientRawRequest = artifact?.pipeline?.clientRawRequest as { body?: unknown } | undefined; - const body = clientRawRequest?.body; - if (!body || typeof body !== "object") continue; - - for (const turn of extractCanonicalTurns(body as Record)) { - const hash = hashTurnContent(turn); - if (result.has(hash)) continue; - result.set(hash, { - textPreview: turn.text, - blockKind: turn.blockKind, - toolName: turn.toolName, - }); - } + const pipeline = asRecord(artifact?.pipeline); + indexTurns(result, turnsFromBody(asRecord(pipeline?.clientRawRequest)?.body)); + indexTurns(result, turnsFromClientResponse(pipeline?.clientResponse)); + indexTurns(result, turnsFromProviderRequest(asRecord(pipeline?.providerRequest)?.body)); } return result; } From 4ec4ce410e7b6ef0776054222539f89fc0c9b2eb Mon Sep 17 00:00:00 2001 From: Markus Hartung Date: Thu, 3 Sep 2026 18:01:03 +0200 Subject: [PATCH 045/143] fix(sse): remap non-contiguous upstream tool_calls index to a gap-free output_index (#12445) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51` (já com a leva anterior dentro): os nove boardaram **sem um único conflito**, `typecheck:core` limpo e **80/80** nos 6 arquivos de teste que os PRs trazem. O crescimento de arquivo próprio da leva foi rebaselinado num registro datado (`_rebaseline_2026_09_03_hartmark_batch`): `combos/page.tsx` 5012→5018 (#12355, tratar o estado degradado quando o bundling de tiktoken de um provider sem relação falha) e `open-sse/services/combo.ts` 4023→4036 (#12338, os fixes do universal-handoff). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. Obrigado, @hartmark. --- .../translator/response/openai-responses.ts | 6 +- .../openai-responses/toolCallLocalIndex.ts | 35 +++++ .../translator-resp-openai-responses.test.ts | 140 ++++++++++++++++++ 3 files changed, 178 insertions(+), 3 deletions(-) create mode 100644 open-sse/translator/response/openai-responses/toolCallLocalIndex.ts diff --git a/open-sse/translator/response/openai-responses.ts b/open-sse/translator/response/openai-responses.ts index 4d255d40be..a2244f01f5 100644 --- a/open-sse/translator/response/openai-responses.ts +++ b/open-sse/translator/response/openai-responses.ts @@ -23,12 +23,12 @@ import { import { createEventEmitter } from "./openai-responses/eventEmitter.ts"; import { buildResponsesToolCallItem } from "./responsesToolItem.ts"; import { resolveRequestToolIdentity } from "./openai-responses/requestToolIdentity.ts"; +import { resolveLocalToolCallIndex } from "./openai-responses/toolCallLocalIndex.ts"; import { synthesizeCompletedToolCalls, computeFinishReason, withAssistantRoleOnFirstDelta, } from "./openai-responses/synthesizeCompletedToolCalls.ts"; - // normalizeUpstreamFailure is re-exported for external importers (tests). export { normalizeUpstreamFailure } from "./openai-responses/pureHelpers.ts"; @@ -506,7 +506,7 @@ function toolCallOutputIndexBase(state) { function emitToolCall(state, emit, tc) { const tcIdx = tc.index ?? 0; - const outputIndex = toolCallOutputIndexBase(state) + normalizeOutputIndex(tcIdx); + const outputIndex = toolCallOutputIndexBase(state) + resolveLocalToolCallIndex(state, tcIdx); const newCallId = tc.id; const funcName = tc.function?.name; @@ -609,7 +609,7 @@ function emitToolCall(state, emit, tc) { function closeToolCall(state, emit, idx, recordAsCompleted = true) { const callId = state.funcCallIds[idx]; if (callId && !state.funcItemDone[idx]) { - const normalizedIndex = toolCallOutputIndexBase(state) + normalizeOutputIndex(idx); + const normalizedIndex = toolCallOutputIndexBase(state) + resolveLocalToolCallIndex(state, idx); const args = state.funcArgsBuf[idx] || "{}"; const toolName = state.funcNames[idx] || ""; // See emitToolCall()'s isCustomTool comment — must stay in sync (both compute the diff --git a/open-sse/translator/response/openai-responses/toolCallLocalIndex.ts b/open-sse/translator/response/openai-responses/toolCallLocalIndex.ts new file mode 100644 index 0000000000..645278e1fd --- /dev/null +++ b/open-sse/translator/response/openai-responses/toolCallLocalIndex.ts @@ -0,0 +1,35 @@ +/** + * Remap a turn's raw upstream tool_calls delta `index` onto a local, + * contiguous, 0-based sequence in first-seen order. + * + * Live incident (2026-09-02, minimax-m3:free via OpenRouter/GMICloud): the + * upstream's own `index` doesn't reliably start at 0 or stay contiguous per + * turn — this turn's two calls arrived with raw index 1 and 2 (never 0). + * Adding that raw index straight onto toolCallOutputIndexBase() left a GAP + * in the emitted output_index sequence (0 for the message, then 2 and 3 for + * the calls — index 1 never used). A client that reads response.completed's + * final `output[]` array by ARRAY POSITION and expects position to equal + * output_index (the Responses API's own contract) reads output[1] (this + * turn's first call, real output_index 2) while looking it up under + * output_index 1, misses it, then reads output[2] (the second call, real + * output_index 3) under output_index 2 — landing on the FIRST call's tracked + * slot with a different call_id, which a spec-following client correctly + * treats as "stream changed output item identity" and aborts. + */ + +export type ToolCallLocalIndexState = { + toolCallLocalIndex?: Record; + toolCallLocalIndexNext?: number; +}; + +export function resolveLocalToolCallIndex( + state: ToolCallLocalIndexState, + tcIdx: string | number +): number { + if (!state.toolCallLocalIndex) state.toolCallLocalIndex = {}; + if (state.toolCallLocalIndex[tcIdx] === undefined) { + state.toolCallLocalIndex[tcIdx] = state.toolCallLocalIndexNext ?? 0; + state.toolCallLocalIndexNext = state.toolCallLocalIndex[tcIdx] + 1; + } + return state.toolCallLocalIndex[tcIdx]; +} diff --git a/tests/unit/translator-resp-openai-responses.test.ts b/tests/unit/translator-resp-openai-responses.test.ts index 2253c655ac..3a42eca624 100644 --- a/tests/unit/translator-resp-openai-responses.test.ts +++ b/tests/unit/translator-resp-openai-responses.test.ts @@ -994,3 +994,143 @@ test("OpenAI -> Responses: a text message and a following tool call in the same "completed output must include the tool call" ); }); + +// Live incident (2026-09-02): a free-tier streaming model, after a short text +// preamble, opened two tool calls whose upstream `tool_calls[].index` was 1 +// and 2 -- never 0. toolCallOutputIndexBase()+index therefore emitted +// output_index 0 (message), 2, 3 -- skipping 1 entirely. A spec-following +// Responses-API client reads response.completed's final `output[]` array by +// ARRAY POSITION and expects position === output_index (the API's own +// contract): output[1] (this turn's first call, real output_index 2) gets +// looked up under output_index 1 and missed, then output[2] (the second +// call, real output_index 3) gets looked up under output_index 2 and +// collides with the FIRST call's tracked slot -- two different call_ids on +// what the client thinks is one identity, which it correctly refuses to +// treat as anything but a broken stream. Reproduced verbatim (anonymized +// content, same index/id shape) against OpenClaw's own +// createResponsesOutputTracker before this fix; content and tool/model names +// below are placeholders, not the real incident's. +test("OpenAI -> Responses: tool-call output_index stays gap-free when the upstream's own index doesn't start at 0", () => { + const events = collectEvents([ + { + id: "chatcmpl-gap1", + model: "stub-model", + choices: [ + { index: 0, delta: { content: "Status:", role: "assistant" }, finish_reason: null }, + ], + }, + { + id: "chatcmpl-gap1", + model: "stub-model", + choices: [ + { index: 0, delta: { content: " all clear.", role: "assistant" }, finish_reason: null }, + ], + }, + { + id: "chatcmpl-gap1", + model: "stub-model", + choices: [ + { + index: 0, + delta: { + content: null, + role: "assistant", + tool_calls: [ + { + index: 1, + id: "call_stub_1", + type: "function", + function: { name: "notify", arguments: "" }, + }, + ], + }, + finish_reason: null, + }, + ], + }, + { + id: "chatcmpl-gap1", + model: "stub-model", + choices: [ + { + index: 0, + delta: { + content: null, + role: "assistant", + tool_calls: [{ index: 1, function: { arguments: '{"a":1}' } }], + }, + finish_reason: null, + }, + ], + }, + { + id: "chatcmpl-gap1", + model: "stub-model", + choices: [ + { + index: 0, + delta: { + content: null, + role: "assistant", + tool_calls: [ + { + index: 2, + id: "call_stub_2", + type: "function", + function: { name: "notify", arguments: "" }, + }, + ], + }, + finish_reason: null, + }, + ], + }, + { + id: "chatcmpl-gap1", + model: "stub-model", + choices: [ + { + index: 0, + delta: { + content: null, + role: "assistant", + tool_calls: [{ index: 2, function: { arguments: '{"a":2}' } }], + }, + finish_reason: null, + }, + ], + }, + { + id: "chatcmpl-gap1", + model: "stub-model", + choices: [ + { index: 0, delta: { content: "", role: "assistant" }, finish_reason: "tool_calls" }, + ], + usage: { prompt_tokens: 10, completion_tokens: 5, total_tokens: 15 }, + }, + null, + ]); + + const addedEvents = events.filter((e) => e.event === "response.output_item.added"); + const indexes = addedEvents.map((e) => e.data.output_index).sort((a, b) => a - b); + const sequential = indexes.map((_, i) => i); + assert.deepEqual( + indexes, + sequential, + `output_index values must be a gap-free 0..n-1 sequence (position === output_index is the Responses API's own contract); got ${JSON.stringify(indexes)}` + ); + + // The exact client-observable symptom: response.completed's output[] + // array, read by array position, must match each item's own tracked + // output_index -- otherwise a client keying by array position resolves + // the wrong item. + const completedGap = events.find((e) => e.event === "response.completed"); + completedGap.data.response.output.forEach((item, position) => { + const addedEvent = addedEvents.find((e) => e.data.item?.id === item.id); + assert.equal( + addedEvent?.data.output_index, + position, + `item ${item.id} (type ${item.type}) streamed at output_index ${addedEvent?.data.output_index} but sits at array position ${position} in the completed output` + ); + }); +}); From 2e4a79ca5022e4a564b94d3361ef5dd3e57652f0 Mon Sep 17 00:00:00 2001 From: Markus Hartung Date: Thu, 3 Sep 2026 18:01:25 +0200 Subject: [PATCH 046/143] fix(quality): detect duplicate tool_calls entries in one response (#12446) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51` (já com a leva anterior dentro): os nove boardaram **sem um único conflito**, `typecheck:core` limpo e **80/80** nos 6 arquivos de teste que os PRs trazem. O crescimento de arquivo próprio da leva foi rebaselinado num registro datado (`_rebaseline_2026_09_03_hartmark_batch`): `combos/page.tsx` 5012→5018 (#12355, tratar o estado degradado quando o bundling de tiktoken de um provider sem relação falha) e `open-sse/services/combo.ts` 4023→4036 (#12338, os fixes do universal-handoff). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. Obrigado, @hartmark. --- open-sse/handlers/chatCore/attemptLogging.ts | 10 ++++ .../chatCore/toolCallSpecViolationAudit.ts | 46 +++++++++++++++ open-sse/services/combo/validateQuality.ts | 50 +++++++++++++++- tests/unit/chatcore-attempt-logging.test.ts | 59 +++++++++++++++++++ 4 files changed, 162 insertions(+), 3 deletions(-) create mode 100644 open-sse/handlers/chatCore/toolCallSpecViolationAudit.ts diff --git a/open-sse/handlers/chatCore/attemptLogging.ts b/open-sse/handlers/chatCore/attemptLogging.ts index 65a492a541..07de9f19b2 100644 --- a/open-sse/handlers/chatCore/attemptLogging.ts +++ b/open-sse/handlers/chatCore/attemptLogging.ts @@ -13,6 +13,7 @@ import { extractProviderWarnings } from "@/lib/compliance/providerAudit"; import { logAuditEvent } from "@/lib/compliance"; import { emit } from "@/lib/events/eventBus"; +import { maybeLogToolCallSpecViolation } from "./toolCallSpecViolationAudit.ts"; import type { RequestCompletedPayload, RequestFailedPayload } from "@/lib/events/types"; import { saveCallLog } from "@/lib/usageDb"; import type { VideoBridgeLogRedactionEntry } from "@/lib/guardrails/videoBridge"; @@ -392,6 +393,15 @@ export function persistAttemptLogs(args: PersistAttemptLogsArgs, ctx: PersistAtt }); } + maybeLogToolCallSpecViolation({ + responseBody, + provider, + model, + connectionId: finalConnectionId, + httpStatus: status, + requestId: skillRequestId, + }); + const capturedPipeline = reqLogger?.getPipelinePayloads?.() ?? null; const pipelinePayloads = detailedLoggingEnabled ? (capturedPipeline ?? {}) diff --git a/open-sse/handlers/chatCore/toolCallSpecViolationAudit.ts b/open-sse/handlers/chatCore/toolCallSpecViolationAudit.ts new file mode 100644 index 0000000000..573daae59e --- /dev/null +++ b/open-sse/handlers/chatCore/toolCallSpecViolationAudit.ts @@ -0,0 +1,46 @@ +/** + * Post-request on-spec audit for duplicated tool_calls. + * + * Extracted from persistAttemptLogs so attemptLogging.ts stays at the frozen + * complexity count. validateResponseQuality's streaming peek only sees the + * START of a stream, so a duplicate that arrives after real content has + * already been relayed cannot fail the attempt over — this is the first + * point the fully assembled body is available. Too late to retry; a durable + * audit row still beats a clean HTTP 200 with no trace. + * + * Observed: minimax-m3:free via OpenRouter/GMICloud, 2026-09-02, duplicated + * a heartbeat_respond call byte-for-byte. + */ + +import { logAuditEvent } from "@/lib/compliance"; +import { findToolCallSpecViolation } from "../../services/combo/validateQuality.ts"; + +export function maybeLogToolCallSpecViolation(input: { + responseBody: unknown; + provider: string | null | undefined; + model: string | null | undefined; + connectionId: string | null; + httpStatus: number; + requestId: string; +}): void { + const violation = findToolCallSpecViolation(input.responseBody); + if (!violation) return; + logAuditEvent({ + action: "provider.spec_violation", + actor: "system", + target: + [input.provider, input.connectionId].filter(Boolean).join(":") || + input.provider || + input.model, + resourceType: "provider_spec_violation", + status: "warning", + requestId: input.requestId, + details: { + provider: input.provider, + model: input.model, + connectionId: input.connectionId, + httpStatus: input.httpStatus, + violation, + }, + }); +} diff --git a/open-sse/services/combo/validateQuality.ts b/open-sse/services/combo/validateQuality.ts index be70038994..723a10b614 100644 --- a/open-sse/services/combo/validateQuality.ts +++ b/open-sse/services/combo/validateQuality.ts @@ -16,6 +16,45 @@ import { evaluateResponseValidation, type ResponseValidationConfig } from "./res import { getReasoningTokens } from "../../../src/lib/usage/tokenAccounting.ts"; import type { ComboRetryAfter } from "./types.ts"; +/** + * Detects tool_calls entries within one assistant message that repeat the + * exact same function name + arguments verbatim -- always a bug (no + * legitimate use calls one tool twice with identical arguments in the same + * turn), and a real observed failure mode of at least one free-tier + * streaming model (minimax-m3:free via OpenRouter/GMICloud, 2026-09-02: + * duplicated a heartbeat_respond call byte-for-byte, confirmed at the raw + * SSE wire level -- an upstream bug, not an OmniRoute reconstruction + * artifact). Used two ways: to fail a non-streaming response over to a + * sibling combo target (see validateResponseQuality below), and, post- + * stream, to flag an already-relayed streaming response as an on-spec + * violation despite its clean HTTP 200 (see attemptLogging.ts's + * persistAttemptLogs) -- a streaming response can't be retried once real + * content has started reaching the client (the quality-gate peek below only + * ever validates the START of a stream, by design, to avoid buffering the + * whole response and defeating streaming's latency purpose), so flagging it + * after the fact is what's actually achievable for that path. + */ +export function findToolCallSpecViolation(responseBody: unknown): string | null { + const json = isRecord(responseBody) ? responseBody : null; + const choices = json?.choices; + const firstChoice = Array.isArray(choices) ? choices[0] : null; + const message = isRecord(firstChoice) ? firstChoice.message : null; + const toolCalls = isRecord(message) ? message.tool_calls : null; + if (!Array.isArray(toolCalls) || toolCalls.length < 2) return null; + + const seen = new Set(); + for (const call of toolCalls) { + const fn = isRecord(call) ? call.function : null; + if (!isRecord(fn) || typeof fn.name !== "string" || typeof fn.arguments !== "string") { + continue; + } + const signature = `${fn.name}\u0000${fn.arguments}`; + if (seen.has(signature)) return `duplicate tool_calls entry for "${fn.name}"`; + seen.add(signature); + } + return null; +} + export function toRetryAfterDisplayValue(value: ComboRetryAfter): string | Date { if (typeof value !== "number") return value; if (value > 0 && value < 1_000_000_000) { @@ -327,9 +366,9 @@ export async function validateResponseQuality( function isTerminalUsageOnlyChunk(parsed: Record, eventType: string): boolean { return Boolean( parsed.usage && - typeof parsed.usage === "object" && - !Array.isArray(parsed.choices) && - !eventType.startsWith("response.") + typeof parsed.usage === "object" && + !Array.isArray(parsed.choices) && + !eventType.startsWith("response.") ); } @@ -734,6 +773,11 @@ export async function validateResponseQuality( } const hasToolCalls = Array.isArray(toolCalls) && toolCalls.length > 0; + const specViolation = findToolCallSpecViolation(json); + if (specViolation) { + return { valid: false, reason: specViolation }; + } + if (!hasContent && !hasToolCalls) { return { valid: false, reason: "empty content and no tool_calls in response" }; } diff --git a/tests/unit/chatcore-attempt-logging.test.ts b/tests/unit/chatcore-attempt-logging.test.ts index b9fe57fbd3..2af6bbd41c 100644 --- a/tests/unit/chatcore-attempt-logging.test.ts +++ b/tests/unit/chatcore-attempt-logging.test.ts @@ -16,6 +16,7 @@ process.env.DATA_DIR = testDataDir; const coreDb = await import("../../src/lib/db/core.ts"); const { getCallLogById } = await import("../../src/lib/usage/callLogs.ts"); const { persistAttemptLogs } = await import("../../open-sse/handlers/chatCore/attemptLogging.ts"); +const { getAuditLog } = await import("../../src/lib/compliance/index.ts"); type CodexRotationEnvelope = { _omniroute?: { @@ -136,3 +137,61 @@ test("connectionId falls back to credentials.connectionId when null, and error i assert.equal(row.status, 502); assert.match(String(row.error ?? ""), /upstream boom/); }); + +function duplicateHeartbeatBody() { + return { + choices: [ + { + message: { + tool_calls: [ + { function: { name: "heartbeat_respond", arguments: "{}" } }, + { function: { name: "heartbeat_respond", arguments: "{}" } }, + ], + }, + }, + ], + }; +} + +test("duplicate tool_calls in the assembled body writes provider.spec_violation audit", () => { + persistAttemptLogs( + { status: 200, responseBody: duplicateHeartbeatBody() }, + baseCtx({ pendingRequestId: "attempt-spec-violation-1", skillRequestId: "skill-spec-1" }) + ); + // logAuditEvent is synchronous; do not wait on the fire-and-forget saveCallLog. + const rows = getAuditLog({ action: "provider.spec_violation", requestId: "skill-spec-1" }); + assert.equal(rows.length, 1); + assert.equal(rows[0]?.resourceType, "provider_spec_violation"); + const details = rows[0]?.details; + assert.ok(details && typeof details === "object"); + assert.equal( + (details as { violation?: string }).violation, + 'duplicate tool_calls entry for "heartbeat_respond"' + ); +}); + +test("unique tool_calls do not write provider.spec_violation audit", () => { + persistAttemptLogs( + { + status: 200, + responseBody: { + choices: [ + { + message: { + tool_calls: [ + { function: { name: "heartbeat_respond", arguments: "{}" } }, + { function: { name: "other_tool", arguments: "{}" } }, + ], + }, + }, + ], + }, + }, + baseCtx({ pendingRequestId: "attempt-spec-clean-1", skillRequestId: "skill-spec-clean-1" }) + ); + const rows = getAuditLog({ + action: "provider.spec_violation", + requestId: "skill-spec-clean-1", + }); + assert.equal(rows.length, 0); +}); From 0019a47f24099df09a03dca2af0e085209f4224c Mon Sep 17 00:00:00 2001 From: Markus Hartung Date: Thu, 3 Sep 2026 18:01:43 +0200 Subject: [PATCH 047/143] fix(responses-continuation): fail closed on a collector-truncated, empty output array (#12460) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51` (já com a leva anterior dentro): os nove boardaram **sem um único conflito**, `typecheck:core` limpo e **80/80** nos 6 arquivos de teste que os PRs trazem. O crescimento de arquivo próprio da leva foi rebaselinado num registro datado (`_rebaseline_2026_09_03_hartmark_batch`): `combos/page.tsx` 5012→5018 (#12355, tratar o estado degradado quando o bundling de tiktoken de um provider sem relação falha) e `open-sse/services/combo.ts` 4023→4036 (#12338, os fixes do universal-handoff). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. Obrigado, @hartmark. --- src/lib/db/responsesContinuationStore.ts | 20 +++++++ .../unit/responses-continuation-store.test.ts | 58 +++++++++++++++++++ 2 files changed, 78 insertions(+) diff --git a/src/lib/db/responsesContinuationStore.ts b/src/lib/db/responsesContinuationStore.ts index ef7a6e3de3..3e0f79b7ca 100644 --- a/src/lib/db/responsesContinuationStore.ts +++ b/src/lib/db/responsesContinuationStore.ts @@ -107,5 +107,25 @@ export function resolvePreviousResponseState( if (!Array.isArray(input) || !Array.isArray(output)) return null; if (containsTruncatedArrayMarker(input) || containsTruncatedArrayMarker(output)) return null; + // Live incident (2026-09-02): a huge/reasoning-heavy response can blow past + // createStructuredSSECollector's own event-count cap mid-stream -- the + // stored clientResponse then carries `_truncated: true` and + // `summary.status: "in_progress"` (never reached "completed") with a + // genuinely EMPTY `summary.output`, not a bounded array with a + // containsTruncatedArrayMarker sentinel (that marker only covers an + // array capped mid-array, not a collector that stopped before ever + // populating output at all). An empty output array passed the checks + // above and got merged into the next turn's request as this response's + // entire contribution -- reconstructing to zero real messages, which the + // upstream provider then rejected outright ("Input required: specify + // prompt or messages"), breaking the conversation with no client-visible + // continuation path. A response the client received as real (successful, + // non-empty) always has at least one output item; failing closed here + // makes the caller ask the client to resend full history instead of + // silently reconstructing an empty one, exactly like a real + // previous_response_not_found from OpenAI itself. + if ((clientResponse as { _truncated?: unknown } | undefined)?._truncated === true) return null; + if (output.length === 0) return null; + return { input, output }; } diff --git a/tests/unit/responses-continuation-store.test.ts b/tests/unit/responses-continuation-store.test.ts index 6c75d53c24..4f414c4e16 100644 --- a/tests/unit/responses-continuation-store.test.ts +++ b/tests/unit/responses-continuation-store.test.ts @@ -261,6 +261,64 @@ test("resolvePreviousResponseState fails closed when the stored input array was assert.equal(store.resolvePreviousResponseState("resp_gen-truncated-history", "key-1"), null); }); +test("resolvePreviousResponseState fails closed when the streaming collector truncated the response", () => { + // Live incident (2026-09-02): a huge/reasoning-heavy response blew past + // createStructuredSSECollector's own event-count cap mid-stream. The + // stored clientResponse then carries `_truncated: true` and + // `summary.status: "in_progress"` (never reached "completed") with a + // genuinely empty `summary.output` -- not a bounded array with an + // `_omniroute_truncated_array` sentinel (that only covers an array capped + // mid-array, not a collector that stopped before populating output at + // all). The empty array previously passed every check here and got + // merged into the next turn's request as this response's entire + // contribution -- reconstructing to zero real messages, which the + // upstream provider then rejected outright ("Input required: specify + // prompt or messages"), breaking the conversation. Measured live: ~22% + // of a sample of recent successful Ping responses carried this flag. + insertCallLog({ + id: "log-8", + responseId: "resp_gen-collector-truncated", + apiKeyId: "key-1", + detailState: "ready", + artifactRelPath: "2026-01-01/log-8.json", + }); + writeArtifact("2026-01-01/log-8.json", { + clientRawRequest: { body: { input: [{ type: "message", role: "user", content: "hi" }] } }, + providerRequest: { body: { input: [{ type: "message", role: "user", content: "hi" }] } }, + clientResponse: { + _streamed: true, + _truncated: true, + _droppedEvents: 24, + summary: { id: "resp_gen-collector-truncated", status: "in_progress", output: [] }, + }, + }); + + assert.equal(store.resolvePreviousResponseState("resp_gen-collector-truncated", "key-1"), null); +}); + +test("resolvePreviousResponseState fails closed on an empty output array even without the _truncated flag", () => { + // Belt-and-suspenders for the same failure class when the collector + // truncated without ever setting `_truncated` (or for a non-streaming + // response that somehow logged zero output items): a response the + // client actually received as real/successful always has at least one + // output item, so an empty array here is never a legitimate prior turn + // to reconstruct from. + insertCallLog({ + id: "log-9", + responseId: "resp_gen-empty-output", + apiKeyId: "key-1", + detailState: "ready", + artifactRelPath: "2026-01-01/log-9.json", + }); + writeArtifact("2026-01-01/log-9.json", { + clientRawRequest: { body: { input: [{ type: "message", role: "user", content: "hi" }] } }, + providerRequest: { body: { input: [{ type: "message", role: "user", content: "hi" }] } }, + clientResponse: { id: "resp_gen-empty-output", output: [] }, + }); + + assert.equal(store.resolvePreviousResponseState("resp_gen-empty-output", "key-1"), null); +}); + test("resolvePreviousResponseState returns null when detail logging was never captured for this row", () => { insertCallLog({ id: "log-5", From 87299096228a717d095e28430e1ac65aede6b847 Mon Sep 17 00:00:00 2001 From: Markus Hartung Date: Thu, 3 Sep 2026 18:02:00 +0200 Subject: [PATCH 048/143] fix(logging): raise the SSE payload collector's default cap (#12461) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51` (já com a leva anterior dentro): os nove boardaram **sem um único conflito**, `typecheck:core` limpo e **80/80** nos 6 arquivos de teste que os PRs trazem. O crescimento de arquivo próprio da leva foi rebaselinado num registro datado (`_rebaseline_2026_09_03_hartmark_batch`): `combos/page.tsx` 5012→5018 (#12355, tratar o estado degradado quando o bundling de tiktoken de um provider sem relação falha) e `open-sse/services/combo.ts` 4023→4036 (#12338, os fixes do universal-handoff). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. Obrigado, @hartmark. --- open-sse/utils/streamPayloadCollector.ts | 20 ++++++- .../unit/stream-payload-collector-cap.test.ts | 57 +++++++++++++++++++ 2 files changed, 76 insertions(+), 1 deletion(-) create mode 100644 tests/unit/stream-payload-collector-cap.test.ts diff --git a/open-sse/utils/streamPayloadCollector.ts b/open-sse/utils/streamPayloadCollector.ts index c7aab45b7a..f0353ce0f5 100644 --- a/open-sse/utils/streamPayloadCollector.ts +++ b/open-sse/utils/streamPayloadCollector.ts @@ -885,8 +885,26 @@ export function compactStructuredStreamPayload(payload: unknown): unknown { }; } +// Live incident (2026-09-02): a reasoning-heavy response streams reasoning +// token-by-token as hundreds to thousands of tiny SSE deltas BEFORE the real +// output/tool_calls ever arrive. At the old defaults (200 events / 48KB) the +// cap was routinely exhausted during the reasoning phase alone, dropping the +// completion event entirely -- measured live: ~22% of a sample of recent +// successful responses hit this. For a caller with no `format` (no live +// reducer -- see the CollectorOptions.format doc comment), the logged +// summary is reconstructed from getEvents() (open-sse/utils/stream.ts), so a +// dropped completion event produced a served-successfully response logged +// with status "in_progress" and empty output -- which +// src/lib/db/responsesContinuationStore.ts then had nothing real to +// reconstruct a later continuation turn from (see its own fail-closed fix, +// 2026-09-02). Raising the cap doesn't eliminate the class of bug for an +// arbitrarily long stream, but it removes it as a routine, everyday failure; +// the format-driven live reducer (used by providerPayloadCollector, an +// analogous prior fix) is the cap-independent fix and remains the deeper +// follow-up for a caller that still wants build()'s summary correct beyond +// any fixed cap. export function createStructuredSSECollector(options: CollectorOptions = {}) { - const { maxEvents = 200, maxBytes = 49152, stage, format, fallbackModel } = options; + const { maxEvents = 2000, maxBytes = 524288, stage, format, fallbackModel } = options; const events: StructuredSSEEvent[] = []; let usedBytes = 0; let droppedEvents = 0; diff --git a/tests/unit/stream-payload-collector-cap.test.ts b/tests/unit/stream-payload-collector-cap.test.ts new file mode 100644 index 0000000000..a737c7a88d --- /dev/null +++ b/tests/unit/stream-payload-collector-cap.test.ts @@ -0,0 +1,57 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +const { createStructuredSSECollector } = + await import("../../open-sse/utils/streamPayloadCollector.ts"); + +// Live incident (2026-09-02): a reasoning-heavy response streams reasoning +// token-by-token as hundreds to thousands of tiny SSE deltas before the real +// output/tool_calls ever arrive. At the old defaults (200 events / 48KB), +// the cap was routinely exhausted during the reasoning phase alone, +// dropping the completion event entirely -- for a caller with no `format` +// (no live reducer), the logged summary is reconstructed from getEvents() +// (see open-sse/utils/stream.ts), so the dropped completion silently +// produced a served-successfully response logged as _truncated with an +// empty output array. src/lib/db/responsesContinuationStore.ts then had +// nothing real to reconstruct a later continuation turn from. Measured +// live: ~22% of a sample of recent successful responses hit this. +test("createStructuredSSECollector retains a realistic reasoning-heavy event burst without dropping (regression for the 2026-09-02 truncated-continuation incident)", () => { + const collector = createStructuredSSECollector({ stage: "client_response" }); + + // Anonymized, real-incident shape: ~1600 small reasoning deltas (the + // observed volume for a genuinely reasoning-heavy turn) followed by the + // actual completion event -- exactly the ordering that exhausted the old + // 200-event/48KB cap before the completion event ever arrived. + for (let i = 0; i < 1600; i++) { + collector.push({ + type: "response.reasoning_summary_text.delta", + delta: "token ", + sequence_number: i, + }); + } + collector.push({ + type: "response.completed", + response: { id: "resp_test", status: "completed", output: [{ type: "message" }] }, + }); + + const built = collector.build(undefined, { includeEvents: true }); + assert.equal(built._truncated, undefined, "a realistic reasoning burst must not hit the cap"); + assert.equal(built._droppedEvents, undefined); + + const events = collector.getEvents(); + const completedEvent = events.find((e) => e.data?.type === "response.completed"); + assert.ok(completedEvent, "the completion event must survive to build()'s retained events"); +}); + +test("createStructuredSSECollector still reports _truncated once a stream genuinely exceeds the (raised) cap", () => { + // The cap protects against a truly pathological/runaway stream -- raising + // it must not remove that protection, only its false-positive rate on + // realistic reasoning-heavy traffic. + const collector = createStructuredSSECollector({ stage: "client_response", maxEvents: 5 }); + for (let i = 0; i < 10; i++) { + collector.push({ type: "response.output_text.delta", delta: "x", sequence_number: i }); + } + const built = collector.build(undefined, { includeEvents: false }); + assert.equal(built._truncated, true); + assert.equal(built._droppedEvents, 5); +}); From 4866f927ad1c80596e40272593df58bf1f2019aa Mon Sep 17 00:00:00 2001 From: Markus Hartung Date: Thu, 3 Sep 2026 18:02:24 +0200 Subject: [PATCH 049/143] =?UTF-8?q?fix(combo):=20universal-handoff=20fixes?= =?UTF-8?q?=20=E2=80=94=20bare-fallback=20note,=20same-request=20scoping,?= =?UTF-8?q?=20silent-failure=20logging=20(#12338)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51` (já com a leva anterior dentro): os nove boardaram **sem um único conflito**, `typecheck:core` limpo e **80/80** nos 6 arquivos de teste que os PRs trazem. O crescimento de arquivo próprio da leva foi rebaselinado num registro datado (`_rebaseline_2026_09_03_hartmark_batch`): `combos/page.tsx` 5012→5018 (#12355, tratar o estado degradado quando o bundling de tiktoken de um provider sem relação falha) e `open-sse/services/combo.ts` 4023→4036 (#12338, os fixes do universal-handoff). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. Obrigado, @hartmark. --- open-sse/services/combo.ts | 19 ++++++- open-sse/services/contextHandoff.ts | 68 ++++++++++++++++++++------ tests/unit/combo-context-relay.test.ts | 15 ++++-- tests/unit/universal-handoff.test.ts | 8 ++- 4 files changed, 88 insertions(+), 22 deletions(-) diff --git a/open-sse/services/combo.ts b/open-sse/services/combo.ts index c18cd37333..be662c600f 100644 --- a/open-sse/services/combo.ts +++ b/open-sse/services/combo.ts @@ -1661,8 +1661,16 @@ async function handleComboChatInner({ } } - // Universal handoff: inject existing handoff if model changed + // Universal handoff: inject existing handoff if model changed. i === 0 + // only: a fallback target (i > 0) serves the SAME client request the + // failed primary target would have served, with the original messages + // already intact -- there's nothing to hand off, since the client never + // saw the earlier target fail. Injecting a handoff note there replaces + // real context with a context-free note, which weaker fallback models + // have been observed treating as license to fabricate content instead + // of just answering the actual request (#12227 follow-up). if ( + i === 0 && universalHandoffConfig.enabled && relayOptions?.sessionId && !(body as Record)?.[SKIP_UNIVERSAL_HANDOFF_FLAG] @@ -1928,7 +1936,14 @@ async function handleComboChatInner({ provider, target.connectionId ?? undefined ); - if (prevModel && prevModel !== modelStr) { + // i === 0 only: a same-request fallback target (i > 0) never + // needs a summary generated for it -- see the injection-site + // comment above. recordSessionModelUsage above stays + // unconditional regardless of i: it must reflect whichever + // model actually served THIS response, since the next + // request's i === 0 comparison depends on that being + // accurate even when this response came from a fallback. + if (i === 0 && prevModel && prevModel !== modelStr) { const handoffSourceMessages = Array.isArray(body?.messages) && body.messages.length > 0 ? body.messages diff --git a/open-sse/services/contextHandoff.ts b/open-sse/services/contextHandoff.ts index bd169be606..c43b7f55c1 100644 --- a/open-sse/services/contextHandoff.ts +++ b/open-sse/services/contextHandoff.ts @@ -407,7 +407,10 @@ async function generateHandoffAsync(options: { relayConfig.relayMode ); const historyText = formatMessagesForPrompt(selectedMessages); - if (!historyText) return; + if (!historyText) { + logUniversalHandoffOutcome("unavailable", options.comboName, "empty selected-message history"); + return; + } const summaryPrompt = HANDOFF_PROMPT_TEMPLATE.replace("{HISTORY}", historyText); const summaryBody = { @@ -421,7 +424,14 @@ async function generateHandoffAsync(options: { }; const response = await options.handleSingleModel(summaryBody, summaryModel); - if (!response.ok) return; + if (!response.ok) { + logUniversalHandoffOutcome( + "unavailable", + options.comboName, + `summary model call failed: status=${response.status} model=${summaryModel}` + ); + return; + } let content = ""; try { @@ -436,7 +446,14 @@ async function generateHandoffAsync(options: { } const parsed = parseHandoffJSON(content); - if (!parsed) return; + if (!parsed) { + logUniversalHandoffOutcome( + "unparseable", + options.comboName, + `model=${summaryModel} contentPreview=${JSON.stringify(content.slice(0, 200))}` + ); + return; + } upsertHandoff({ sessionId: options.sessionId, @@ -572,7 +589,7 @@ export function buildUniversalHandoffSystemMessage( ${escapedReason} ${escapedPrev} ${escapedCurr} -A continuación se resume toda la conversacion para continuar sin perder el hilo. +No prior-session summary is available for this handoff. The input below (e.g. a tool result) is the entire context you have -- do not assume or invent details about a broader conversation you cannot see. `; } @@ -687,6 +704,23 @@ export function resetUniversalHandoffCooldowns(): void { universalHandoffCooldowns.clear(); } +// Every non-"generated" outcome across both handoff generators (this one and +// the older generateHandoffAsync above) used to be silent -- context_handoffs +// staying empty gave no signal on WHY (upstream call failing vs. malformed +// output vs. no history to summarize). Every live handoff then falls back to +// the bare no-summary note (buildUniversalHandoffSystemMessage's `!payload` +// branch / the context-relay equivalent), which is what actually reaches the +// model/user; without this log that always reads as a mystery instead of a +// traceable cause. +function logUniversalHandoffOutcome( + outcome: "unavailable" | "unparseable", + comboName: string, + detail: string +): void { + if (process.env.NODE_ENV === "test") return; + console.warn(`[universal-handoff] ${outcome} (combo=${comboName}): ${detail}`); +} + /** * Generate a universal handoff summary for any model/provider switch. */ @@ -709,7 +743,10 @@ async function generateUniversalHandoffAsync(options: { options.relayMode ); const historyText = formatMessagesForPrompt(selectedMessages); - if (!historyText) return "unavailable"; + if (!historyText) { + logUniversalHandoffOutcome("unavailable", options.comboName, "empty selected-message history"); + return "unavailable"; + } const summaryPrompt = HANDOFF_PROMPT_TEMPLATE.replace("{HISTORY}", historyText); const summaryModel = options.handoffModel || options.currModel; @@ -735,22 +772,25 @@ async function generateUniversalHandoffAsync(options: { }; const response = await options.handleSingleModel(summaryBody, summaryModel); - if (!response.ok) return "unavailable"; + if (!response.ok) { + const detail = `summary model call failed: status=${response.status} model=${summaryModel}`; + logUniversalHandoffOutcome("unavailable", options.comboName, detail); + return "unavailable"; + } let content = ""; try { - const json = (await response.clone().json()) as Record; - content = getResponseText(json); + content = getResponseText((await response.clone().json()) as Record); } catch { - try { - content = await response.clone().text(); - } catch { - content = ""; - } + content = await response.clone().text().catch(() => ""); } const parsed = parseHandoffJSON(content); - if (!parsed) return "unparseable"; + if (!parsed) { + const preview = JSON.stringify(content.slice(0, 200)); + logUniversalHandoffOutcome("unparseable", options.comboName, `model=${summaryModel} contentPreview=${preview}`); + return "unparseable"; + } upsertHandoff({ sessionId: options.sessionId, diff --git a/tests/unit/combo-context-relay.test.ts b/tests/unit/combo-context-relay.test.ts index e984d7a696..82e88e6fd5 100644 --- a/tests/unit/combo-context-relay.test.ts +++ b/tests/unit/combo-context-relay.test.ts @@ -404,7 +404,14 @@ test("getLastSessionModel uses latest id as deterministic tie-breaker", async () assert.equal(handoffDb.getLastSessionModel(sessionId, comboName), "anthropic/new"); }); -test("handleComboChat universal handoff does not accumulate injected handoffs across fallback targets", async () => { +test("handleComboChat universal handoff skips same-request fallback targets entirely", async () => { + // #12227 follow-up: a same-request fallback target (i > 0) serves the SAME + // client request the failed primary target would have served -- the client + // never saw the earlier target fail, so there's no genuine "handoff" to + // explain. Injecting one there replaces real conversation content with a + // context-free note; weaker fallback models have been observed fabricating + // content instead of just answering the actual request when handed that + // note. The fallback target must receive the original request untouched. const sessionId = "sess-universal-no-mutate"; const comboName = "universal-no-mutate"; @@ -463,10 +470,8 @@ test("handleComboChat universal handoff does not accumulate injected handoffs ac typeof message?.content === "string" && message.content.includes("") ); - assert.equal(handoffMessages.length, 1); - assert.match(handoffMessages[0].content, /openai\/previous/); - assert.match(handoffMessages[0].content, /anthropic\/fallback/); - assert.doesNotMatch(handoffMessages[0].content, /openai\/failed/); + assert.equal(handoffMessages.length, 0); + assert.deepEqual(fallbackBody.messages, [{ role: "user", content: "Continue" }]); }); test("handleComboChat universal handoff detects model switch before recording current model", async () => { diff --git a/tests/unit/universal-handoff.test.ts b/tests/unit/universal-handoff.test.ts index 45c06f963f..addf4b633e 100644 --- a/tests/unit/universal-handoff.test.ts +++ b/tests/unit/universal-handoff.test.ts @@ -237,7 +237,13 @@ test("buildUniversalHandoffSystemMessage basic when payload null", () => { test("buildUniversalHandoffSystemMessage basic when payload summary empty", () => { const msg = buildUniversalHandoffSystemMessage(PREV, CURR, REASON, makePayload({ summary: "" })); - assert.ok(msg.includes("continuar sin perder el hilo")); + // The bare-fallback note must not claim continuity it can't provide: a + // model landing here with only trimmed input (e.g. a bare tool result) + // and no real history has been observed fabricating plausible-sounding + // but entirely invented content when told "the conversation continues + // without losing context" -- the note now tells it the opposite. + assert.ok(msg.includes("No prior-session summary is available")); + assert.ok(msg.includes("do not assume or invent")); }); test("buildUniversalHandoffSystemMessage full XML with valid payload", () => { From ffdc7360604fc4f831e1f992c94f1a5cbe610af0 Mon Sep 17 00:00:00 2001 From: Markus Hartung Date: Thu, 3 Sep 2026 18:02:43 +0200 Subject: [PATCH 050/143] feat(dashboard): parent-link, genuine-continuation badge, and modal perf fixes (#12448) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 9 PRs desta leva sobre o tip de `release/v3.8.51` (já com a leva anterior dentro): os nove boardaram **sem um único conflito**, `typecheck:core` limpo e **80/80** nos 6 arquivos de teste que os PRs trazem. O crescimento de arquivo próprio da leva foi rebaselinado num registro datado (`_rebaseline_2026_09_03_hartmark_batch`): `combos/page.tsx` 5012→5018 (#12355, tratar o estado degradado quando o bundling de tiktoken de um provider sem relação falha) e `open-sse/services/combo.ts` 4023→4036 (#12338, os fixes do universal-handoff). As violações restantes (`codex.ts`, `stream.ts`) foram medidas também no tip puro e são drift da base, não desta leva. Obrigado, @hartmark. --- .../dashboard/conversations/page.tsx | 128 +++++++++++++++--- src/app/api/conversations/[id]/route.ts | 36 +++++ src/app/api/conversations/route.ts | 66 ++++++--- src/app/api/logs/[id]/route.ts | 19 ++- src/lib/db/agenticConversations.ts | 74 +++++++--- src/lib/db/responsesContinuationStore.ts | 104 ++++++++++++++ .../RequestLoggerDetail.sections.tsx | 14 ++ 7 files changed, 381 insertions(+), 60 deletions(-) create mode 100644 src/app/api/conversations/[id]/route.ts diff --git a/src/app/(dashboard)/dashboard/conversations/page.tsx b/src/app/(dashboard)/dashboard/conversations/page.tsx index f9fb3b1848..14b8bdf4d7 100644 --- a/src/app/(dashboard)/dashboard/conversations/page.tsx +++ b/src/app/(dashboard)/dashboard/conversations/page.tsx @@ -26,6 +26,11 @@ interface ConversationRow { // streaming (call_logs only gets its row on completion). Used to poll // /api/logs/[id] for this conversation's live partial assistant text. activeCallLogId: string | null; + // Whether the latest turn actually used previous_response_id and it + // resolved server-side — distinct from this row existing at all, which + // only means the client-side content-hash tracker saw >= 2 turns + // regardless of transport (see isGenuineContinuationTurn). + isGenuineContinuation: boolean; } // Same spinner used for an in-flight request on /dashboard/logs @@ -101,6 +106,23 @@ function StatusBadge({ status }: { status: number | null }) { ); } +// Distinguishes a conversation whose latest turn actually used +// previous_response_id (server-verified — see isGenuineContinuationTurn) +// from one the content-hash tracker merely counts as multi-turn while still +// resending full history each request. +function ContinuationBadge({ isGenuine }: { isGenuine: boolean }) { + if (!isGenuine) return null; + return ( + + bolt + continuation + + ); +} + /** * Builds the exact NormalizedBlock (src/mitm/inspector/types.ts) the * request-detail panel already builds from buildRequestTurns/ @@ -268,13 +290,15 @@ function ConversationsPageContent() { // itself in the poll effect's dependency array (which would tear down and // restart the interval on every single appended turn). const newestSeqRef = useRef(null); + // Tracks the PREVIOUS render's activeCallLogId truthiness, so the + // reply-just-finished effect below can detect the true->false transition + // specifically (not "is currently falsy", which would also fire on mount + // / switching conversations). + const wasReplyActiveRef = useRef(false); - // Extracted so openConversation can force an immediate refresh instead of - // waiting for the next scheduled tick — see its call site for why: a - // conversation opened right after a new reply starts streaming otherwise - // shows no live text until this poll's own interval happens to land, - // because activeCallLogId only updates via the resync effect below, which - // depends on this list actually having been refetched. + // The background list poll below only runs this while no conversation + // modal is open — see loadActiveConversationSummary and the poll effect + // for the lighter single-row path used while one is open. const loadConversations = useCallback(() => { if (document.visibilityState !== "visible") return; return fetch("/api/conversations?limit=100", { cache: "no-store" }) @@ -290,22 +314,51 @@ function ConversationsPageContent() { }); }, []); + // While the modal is open, only the one open conversation's summary needs + // to stay live (see the resync effect below) — refetching and + // re-annotating the whole up-to-100-row list every poll tick just to pluck + // that one row back out is pure waste, and at a 1s poll interval it's + // waste on every tick. Patches the row in place so the existing resync + // effect (keyed on `conversations`) picks it up unchanged. + const loadActiveConversationSummary = useCallback((id: string) => { + if (document.visibilityState !== "visible") return; + return fetch(`/api/conversations/${id}`, { cache: "no-store" }) + .then((res) => (res.ok ? res.json() : null)) + .then((data) => { + const fresh = data?.conversation; + if (!fresh) return; + setConversations((prev) => { + const idx = prev.findIndex((c) => c.id === fresh.id); + if (idx === -1) return prev; + const next = prev.slice(); + next[idx] = fresh; + return next; + }); + }) + .catch(() => {}); + }, []); + useEffect(() => { - loadConversations(); - const interval = setInterval(loadConversations, pollSeconds * 1000); + const poll = () => + activeConversationId + ? loadActiveConversationSummary(activeConversationId) + : loadConversations(); + poll(); + const interval = setInterval(poll, pollSeconds * 1000); return () => { clearInterval(interval); }; - }, [pollSeconds, loadConversations]); + }, [pollSeconds, loadConversations, loadActiveConversationSummary, activeConversationId]); // activeConversation is a snapshot taken once at openConversation() time — // it's never touched again while the modal stays open (the turns-poll // effect below only appends conversationNodes). Without this, "Goto latest // request" and any other displayed summary field (lastModel/lastStatus/ // turnCount) go stale the moment a new request lands in this conversation - // while you're still reading it, even though the list poll above (which - // runs regardless of whether the modal is open) already has the fresh - // row. Re-sync from it whenever the list refreshes. + // while you're still reading it. Re-synced from `conversations` whenever + // that refreshes — the effect above keeps it fresh whether the modal is + // closed (full list poll) or open (single-conversation poll patches this + // same row in place). useEffect(() => { if (!activeConversationId) return; const fresh = conversations.find((c) => c.id === activeConversationId); @@ -461,13 +514,13 @@ function ConversationsPageContent() { // ignore navigation errors } // `row` is a snapshot from whenever the list last polled — if a reply - // started streaming after that tick, row.activeCallLogId is still - // null and the live-text poll effect never starts until the next - // scheduled list refresh happens to land (the exact "opened it and - // saw nothing, closed and reopened and saw it live" report). Force - // one now so activeConversation resyncs with the current isActive/ - // activeCallLogId immediately instead of waiting on pollSeconds. - loadConversations(); + // started streaming after that tick, row.activeCallLogId is still null + // and the live-text poll effect never starts until a fresh summary + // lands (the exact "opened it and saw nothing, closed and reopened and + // saw it live" report). setActiveConversation above already changes + // activeConversationId, which is a dependency of the poll effect below + // — it tears down and re-fires immediately on that change, forcing the + // single-row resync here for free without a second explicit call. fetchConversationPage(row.id, `limit=${CONVERSATION_PAGE_SIZE}`) .then((page) => { setConversationNodes(page?.nodes ?? []); @@ -480,7 +533,7 @@ function ConversationsPageContent() { scrollToBottom(); }); }, - [router, fetchConversationPage, scrollToBottom, loadConversations] + [router, fetchConversationPage, scrollToBottom] ); const closeConversation = useCallback(() => { @@ -583,6 +636,35 @@ function ConversationsPageContent() { return () => clearInterval(interval); }, [activeConversationId, pollSeconds, fetchConversationPage]); + // Live incident (2026-09-02): resolveConversationId reassigns a node's + // last_correlation_id to the CURRENT request at request-START (before its + // reply streams), but that request's call-log artifact -- what + // resolveTurnDisplayContent needs to show real text -- is only written at + // completion. A node touched by a still-in-flight request therefore + // legitimately resolves empty if fetched during that window; the afterSeq + // poll above only ever APPENDS strictly newer nodes, so one already + // rendered empty stays empty in local state forever, even once its + // artifact exists moments later -- the exact "empty until you close and + // reopen the conversation" symptom. Once a reply that was streaming + // finishes (activeCallLogId's true -> false transition -- see the + // wasReplyActiveRef doc comment), re-fetch the recent page and merge it in + // by id (never drop older "Load more" history) so any node that resolved + // empty during the race gets its real content without a manual reopen. + useEffect(() => { + const wasActive = wasReplyActiveRef.current; + wasReplyActiveRef.current = Boolean(activeCallLogId); + if (!wasActive || activeCallLogId || !activeConversationId) return; + + fetchConversationPage(activeConversationId, `limit=${CONVERSATION_PAGE_SIZE}`).then((page) => { + if (!page || page.nodes.length === 0) return; + setConversationNodes((prev) => { + const byId = new Map(prev.map((n) => [n.id, n] as const)); + for (const n of page.nodes) byId.set(n.id, n); + return [...byId.values()].sort((a, b) => a.seq - b.seq); + }); + }); + }, [activeCallLogId, activeConversationId, fetchConversationPage]); + // Live preview of the CURRENTLY streaming reply, if any: conversation_turn_nodes // only gains a node for an assistant turn once the client resends it as // history on its NEXT request (resolveConversationId reads only the request @@ -655,6 +737,7 @@ function ConversationsPageContent() { lastStatus: null, isActive: false, activeCallLogId: null, + isGenuineContinuation: false, } ); }, [initialConversationParam, loading, conversations, openConversation]); @@ -759,6 +842,7 @@ function ConversationsPageContent() { > {row.id.slice(0, 16)}… + {row.turnCount} turns @@ -785,6 +869,7 @@ function ConversationsPageContent() { Conversation Turns + Continuation Last Model Provider Status @@ -816,6 +901,9 @@ function ConversationsPageContent() { {row.turnCount} + + + {row.lastModel ?? "—"} diff --git a/src/app/api/conversations/[id]/route.ts b/src/app/api/conversations/[id]/route.ts new file mode 100644 index 0000000000..999e2147df --- /dev/null +++ b/src/app/api/conversations/[id]/route.ts @@ -0,0 +1,36 @@ +import { NextResponse } from "next/server"; +import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; +import { getMultiTurnConversationById } from "@/lib/db/agenticConversations"; +import { annotateConversationRow, buildActiveCallLogIdByConversation } from "../route"; + +export const dynamic = "force-dynamic"; + +/** + * Single-conversation summary — used by the dashboard's conversation modal + * to keep lastModel/lastStatus/isActive/activeCallLogId fresh on the auto- + * refresh interval while it's open, instead of the list route re-fetching + * and re-annotating up to 200 rows just to pluck one back out. The turns + * themselves live-update through the separate .../tree poll; this only + * covers the summary fields the modal header and "Goto latest request" + * read off the row. + */ +export async function GET(req: Request, { params }: { params: Promise<{ id: string }> }) { + const authError = await requireManagementAuth(req); + if (authError) return authError; + + try { + const { id } = await params; + if (!id) return NextResponse.json({ error: "Missing id" }, { status: 400 }); + + const row = getMultiTurnConversationById(id); + if (!row) return NextResponse.json({ error: "Not found" }, { status: 404 }); + + const activeCallLogIdByConversation = buildActiveCallLogIdByConversation(); + const conversation = annotateConversationRow(row, activeCallLogIdByConversation); + + return NextResponse.json({ conversation }); + } catch (err) { + console.error("[API ERROR] /api/conversations/[id] failed:", err); + return NextResponse.json({ error: "Failed to fetch conversation" }, { status: 500 }); + } +} diff --git a/src/app/api/conversations/route.ts b/src/app/api/conversations/route.ts index 9b5ebc3746..e603460dfb 100644 --- a/src/app/api/conversations/route.ts +++ b/src/app/api/conversations/route.ts @@ -1,10 +1,53 @@ import { NextResponse } from "next/server"; import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; +import type { MultiTurnConversationRow } from "@/lib/db/agenticConversations"; import { listMultiTurnConversations } from "@/lib/db/agenticConversations"; import { getPendingById } from "@/lib/usage/usageHistory"; +import { isGenuineContinuationTurn } from "@/lib/db/responsesContinuationStore"; export const dynamic = "force-dynamic"; +/** + * Shared row -> API-shape annotation for both the list route and the + * single-conversation route below: isActive/activeCallLogId (pending-request + * cross reference) and isGenuineContinuation (artifact-backed, cached — see + * isGenuineContinuationTurn) need the exact same computation regardless of + * whether the caller asked for one row or many. Strips the internal-only + * lastArtifactRelPath/lastApiKeyId fields before they reach the client. + */ +export function annotateConversationRow( + row: MultiTurnConversationRow, + activeCallLogIdByConversation: ReadonlyMap +) { + const { lastArtifactRelPath, lastApiKeyId, ...rest } = row; + return { + ...rest, + isActive: activeCallLogIdByConversation.has(row.id), + activeCallLogId: activeCallLogIdByConversation.get(row.id) ?? null, + isGenuineContinuation: isGenuineContinuationTurn(lastArtifactRelPath, lastApiKeyId), + }; +} + +/** + * A pending (still-streaming) request's sessionTag is the conversation's own + * id (agentic_conversations.id === call_logs.session_tag) — cross reference + * so a conversation row can show "in progress" without a separate poll. + * `call_logs` only gets its row on completion (src/lib/usage/callLogs.ts's + * INSERT needs duration/status/tokens, none of which exist yet), so + * lastCallLogId always lags one request behind while a reply is still + * streaming — it can't be used to fetch the in-flight response. Surfacing + * the pending request's own id separately lets the conversation panel poll + * /api/logs/[id] for it directly (same live-partial-text path + * RequestLoggerDetail already uses). + */ +export function buildActiveCallLogIdByConversation(): Map { + const map = new Map(); + for (const pending of getPendingById().values()) { + if (pending.sessionTag) map.set(pending.sessionTag, pending.id); + } + return map; +} + export async function GET(req: Request) { const authError = await requireManagementAuth(req); if (authError) return authError; @@ -19,25 +62,10 @@ export async function GET(req: Request) { offset: Number.isFinite(offset) ? offset : undefined, }); - // A pending (still-streaming) request's sessionTag is the conversation's - // own id (agentic_conversations.id === call_logs.session_tag) — cross - // reference so the list can show "in progress" without a separate poll. - // `call_logs` only gets its row on completion (src/lib/usage/callLogs.ts's - // INSERT needs duration/status/tokens, none of which exist yet), so - // `lastCallLogId` from listMultiTurnConversations always lags one request - // behind while a reply is still streaming — it can't be used to fetch the - // in-flight response. Surface the pending request's own id separately so - // the conversation panel can poll /api/logs/[id] for it directly (same - // live-partial-text path RequestLoggerDetail already uses). - const activeCallLogIdByConversation = new Map(); - for (const pending of getPendingById().values()) { - if (pending.sessionTag) activeCallLogIdByConversation.set(pending.sessionTag, pending.id); - } - const conversations = rows.map((row) => ({ - ...row, - isActive: activeCallLogIdByConversation.has(row.id), - activeCallLogId: activeCallLogIdByConversation.get(row.id) ?? null, - })); + const activeCallLogIdByConversation = buildActiveCallLogIdByConversation(); + const conversations = rows.map((row) => + annotateConversationRow(row, activeCallLogIdByConversation) + ); return NextResponse.json({ conversations, total }); } catch (err) { diff --git a/src/app/api/logs/[id]/route.ts b/src/app/api/logs/[id]/route.ts index 7c20cfb819..afdb7d2432 100644 --- a/src/app/api/logs/[id]/route.ts +++ b/src/app/api/logs/[id]/route.ts @@ -2,6 +2,10 @@ import { NextResponse } from "next/server"; import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { getCallLogById } from "@/lib/usageDb"; import { getCompletedDetails, getPendingById } from "@/lib/usage/usageHistory"; +import { + extractPreviousResponseId, + resolveCallLogIdByResponseId, +} from "@/lib/db/responsesContinuationStore"; // Each logged chunk-array element is one raw network read, timestamp-prefixed // for the debug display — NOT one complete SSE `data:` line. A single JSON @@ -159,7 +163,20 @@ export async function GET( if (!persistedRequest) return NextResponse.json({ error: "Not found" }, { status: 404 }); - return NextResponse.json(persistedRequest); + // "Continues from" link for the dashboard's conversation panel: resolve + // this entry's own previous_response_id back to the call-log row that + // produced it. Persisted-only (apiKeyId isn't plumbed onto the + // pending/in-memory branches above) -- required for the same tenant + // scoping resolveCallLogIdByResponseId enforces, so an active/in-memory + // entry simply renders no parent link rather than resolving unscoped. + const previousResponseId = extractPreviousResponseId( + persistedRequest.pipelinePayloads as Record | null | undefined + ); + const parentLogId = previousResponseId + ? resolveCallLogIdByResponseId(previousResponseId, persistedRequest.apiKeyId ?? null) + : null; + + return NextResponse.json({ ...persistedRequest, previousResponseId, parentLogId }); } catch (err) { console.error("[API ERROR] /api/logs/[id] failed:", err); return NextResponse.json({ error: "Failed to fetch log" }, { status: 500 }); diff --git a/src/lib/db/agenticConversations.ts b/src/lib/db/agenticConversations.ts index 646fce57a7..0bdb9ff4e2 100644 --- a/src/lib/db/agenticConversations.ts +++ b/src/lib/db/agenticConversations.ts @@ -357,6 +357,12 @@ export interface MultiTurnConversationRow extends AgenticConversationRow { lastModel: string | null; lastProvider: string | null; lastStatus: number | null; + // Exposed so the API layer can check whether the latest turn genuinely + // used HTTP continuation (see isGenuineContinuationTurn in + // responsesContinuationStore.ts) without a second query — this row's own + // artifact/tenant already identify it, no separate lookup needed. + lastArtifactRelPath: string | null; + lastApiKeyId: string | null; } /** @@ -400,16 +406,7 @@ export function listMultiTurnConversations( const rows = db .prepare( - `SELECT ac.*, latest.id as last_call_log_id, latest.model as last_model, - latest.provider as last_provider, latest.status as last_status - FROM agentic_conversations ac - LEFT JOIN ( - SELECT cl1.id, cl1.session_tag, cl1.model, cl1.provider, cl1.status - FROM call_logs cl1 - WHERE cl1.timestamp = ( - SELECT MAX(cl2.timestamp) FROM call_logs cl2 WHERE cl2.session_tag = cl1.session_tag - ) - ) latest ON latest.session_tag = ac.id + `${MULTI_TURN_CONVERSATION_SELECT} WHERE (SELECT COUNT(*) FROM conversation_turn_nodes n WHERE n.conversation_id = ac.id) >= 2 ORDER BY ac.last_seen_at DESC LIMIT ? OFFSET ?` @@ -418,15 +415,52 @@ export function listMultiTurnConversations( return { total: Number(total ?? 0), - rows: rows.map((r) => { - const rec = asRecord(r); - return { - ...toRow(rec), - lastCallLogId: typeof rec.last_call_log_id === "string" ? rec.last_call_log_id : null, - lastModel: typeof rec.last_model === "string" ? rec.last_model : null, - lastProvider: typeof rec.last_provider === "string" ? rec.last_provider : null, - lastStatus: typeof rec.last_status === "number" ? rec.last_status : null, - }; - }), + rows: rows.map(toMultiTurnConversationRow), }; } + +function toMultiTurnConversationRow(value: unknown): MultiTurnConversationRow { + const rec = asRecord(value); + return { + ...toRow(rec), + lastCallLogId: typeof rec.last_call_log_id === "string" ? rec.last_call_log_id : null, + lastModel: typeof rec.last_model === "string" ? rec.last_model : null, + lastProvider: typeof rec.last_provider === "string" ? rec.last_provider : null, + lastStatus: typeof rec.last_status === "number" ? rec.last_status : null, + lastArtifactRelPath: + typeof rec.last_artifact_relpath === "string" ? rec.last_artifact_relpath : null, + lastApiKeyId: typeof rec.last_api_key_id === "string" ? rec.last_api_key_id : null, + }; +} + +const MULTI_TURN_CONVERSATION_SELECT = ` + SELECT ac.*, latest.id as last_call_log_id, latest.model as last_model, + latest.provider as last_provider, latest.status as last_status, + latest.artifact_relpath as last_artifact_relpath, + latest.api_key_id as last_api_key_id + FROM agentic_conversations ac + LEFT JOIN ( + SELECT cl1.id, cl1.session_tag, cl1.model, cl1.provider, cl1.status, + cl1.artifact_relpath, cl1.api_key_id + FROM call_logs cl1 + WHERE cl1.timestamp = ( + SELECT MAX(cl2.timestamp) FROM call_logs cl2 WHERE cl2.session_tag = cl1.session_tag + ) + ) latest ON latest.session_tag = ac.id +`; + +/** + * Single-conversation equivalent of listMultiTurnConversations, for the + * dashboard's conversation modal: while it's open, polling this one row on + * the refresh interval (instead of the whole up-to-200-row list just to + * pluck one row back out of it) is what actually needs to stay live — + * lastCallLogId/lastStatus for "Goto latest request" and isActive detection. + * Unlike the list, this intentionally has no turn-count floor: a + * specifically-requested conversation should resolve even if it hasn't (yet) + * reached 2 turn nodes. + */ +export function getMultiTurnConversationById(id: string): MultiTurnConversationRow | null { + const db = getDbInstance(); + const row = db.prepare(`${MULTI_TURN_CONVERSATION_SELECT} WHERE ac.id = ?`).get(id); + return row ? toMultiTurnConversationRow(row) : null; +} diff --git a/src/lib/db/responsesContinuationStore.ts b/src/lib/db/responsesContinuationStore.ts index 3e0f79b7ca..9c55d8edff 100644 --- a/src/lib/db/responsesContinuationStore.ts +++ b/src/lib/db/responsesContinuationStore.ts @@ -129,3 +129,107 @@ export function resolvePreviousResponseState( return { input, output }; } + +/** + * Resolve the call-log id that produced `responseId`, for the dashboard's + * "continues from" link. Reuses the same `call_logs.response_id` index and + * `api_key_id` tenant scoping as `resolvePreviousResponseState` above -- a + * parent link must never point across API keys, even just to surface its id. + * Returns null on any lookup miss so the caller renders no link rather than + * a broken one. + */ +export function resolveCallLogIdByResponseId( + responseId: string, + apiKeyId: string | null | undefined +): string | null { + if (!responseId || !apiKeyId) return null; + + const db = getDbInstance(); + const row = db + .prepare( + `SELECT id FROM call_logs + WHERE response_id = ? AND api_key_id = ? + ORDER BY timestamp DESC LIMIT 1` + ) + .get(responseId, apiKeyId) as { id: string } | undefined; + + return row?.id ?? null; +} + +/** + * Extract `previous_response_id` from a call-log's own pipeline payload. + * Persisted artifacts key the client's own request `clientRawRequest`; the + * pending/in-flight in-memory shape keys the same thing `clientRequest` + * instead (RequestLoggerDetail.tsx's payloadSections list carries both keys + * for the same reason) -- check both so callers get the same answer + * regardless of which shape the payload came back as. + */ +export function extractPreviousResponseId( + pipelinePayloads: Record | null | undefined +): string | null { + if (!pipelinePayloads) return null; + for (const key of ["clientRawRequest", "clientRequest"]) { + const envelope = pipelinePayloads[key]; + const body = isPlainRecord(envelope) && "body" in envelope ? envelope.body : envelope; + if (isPlainRecord(body) && typeof body.previous_response_id === "string") { + return body.previous_response_id; + } + } + return null; +} + +// isGenuineContinuationTurn is a pure function of one call-log's own +// artifact, which is immutable once written (see callLogs.ts -- detailState +// only flips to "ready" after the artifact is fully persisted) -- the same +// artifactRelPath always answers the same way, forever. Without this cache, +// the dashboard's own default auto-refresh polls the whole conversation list +// on an interval the operator controls (down to 1s), so every tick re-reads +// and re-parses one artifact per visible row for an answer that can never +// change once computed. Keyed on artifactRelPath alone (1:1 with the owning +// call-log row, so apiKeyId never varies for a given key) with simple FIFO +// eviction -- correctness never depends on which entries survive, only on +// staying bounded. +const GENUINE_CONTINUATION_CACHE_MAX = 5000; +const genuineContinuationCache = new Map(); + +function cacheGenuineContinuation(key: string, value: boolean): boolean { + genuineContinuationCache.set(key, value); + if (genuineContinuationCache.size > GENUINE_CONTINUATION_CACHE_MAX) { + const oldest = genuineContinuationCache.keys().next().value; + if (oldest !== undefined) genuineContinuationCache.delete(oldest); + } + return value; +} + +/** + * Whether a call-log's own request genuinely continued a prior response + * server-side: it carried `previous_response_id` AND that id resolved to a + * real, same-tenant prior call-log row. Backs the /dashboard/conversations + * "genuine continuation" badge -- a conversation the client-side turn + * tracker counts as multi-turn (conversationTracker.ts's content-hash chain, + * independent of transport) is not necessarily one actually running on the + * `previous_response_id` wire optimization; this checks the transport fact, + * not the content-hash one. + */ +export function isGenuineContinuationTurn( + artifactRelPath: string | null | undefined, + apiKeyId: string | null | undefined +): boolean { + if (!artifactRelPath) return false; + const cached = genuineContinuationCache.get(artifactRelPath); + if (cached !== undefined) return cached; + + const { artifact, state } = readCallArtifact(artifactRelPath); + if (state !== "ready" || !artifact?.pipeline) { + return cacheGenuineContinuation(artifactRelPath, false); + } + const previousResponseId = extractPreviousResponseId( + artifact.pipeline as Record + ); + if (!previousResponseId) return cacheGenuineContinuation(artifactRelPath, false); + + return cacheGenuineContinuation( + artifactRelPath, + resolveCallLogIdByResponseId(previousResponseId, apiKeyId) !== null + ); +} diff --git a/src/shared/components/RequestLoggerDetail.sections.tsx b/src/shared/components/RequestLoggerDetail.sections.tsx index e7a145638d..78527ab495 100644 --- a/src/shared/components/RequestLoggerDetail.sections.tsx +++ b/src/shared/components/RequestLoggerDetail.sections.tsx @@ -259,6 +259,20 @@ export function ConversationContextSection({ log, detail }) { {open ? "expand_less" : "expand_more"} + {liveDetail?.parentLogId && ( + // Full navigation, not client-side routing: the logs page only reads + // ?id from a fresh mount (useState(() => searchParams.get("id")) in + // dashboard/logs/page.tsx), so an in-page route change wouldn't load + // the parent entry if the user is already on this page. + + reply + continues from parent + + )}
{open && (
From 1baa8c36304bfda49d20185b9e36d65d0bbda667 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 13:02:52 -0300 Subject: [PATCH 051/143] fix(quality): re-point the zcodeProtocol public-creds allowlist to line 313 (#12615) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `check:public-creds` has been failing on every open PR against release/v3.8.51, twice over for the same literal: ✗ 1 entrada(s) obsoleta(s) na allowlist — zcodeProtocol.ts:302 ✗ 1 credencial(is) pública(s) como string literal — zcodeProtocol.ts L313 Both are the same `clientId: \`omniroute-${process.pid}\`` in the local ZCode handshake. Nothing regressed: the allowlist key is `file:line:value`, so an edit that shifted the statement from 302 to 313 invalidated the frozen key and the gate reported the entry as stale AND the literal as new. Re-pointed the key and its comment. The literal itself is unchanged and still frozen — the entry is not removed and the detector is not weakened (the gate's own test still asserts that renaming the value to `upstream-client-` is flagged). `tests/unit/check-public-creds.test.ts` synthesizes the source with a newline count to land the statement on the allowlisted line; that count moves with it, 302 -> 313, so the test keeps pinning the real contract instead of a stale one. Also documented the sharp edge inline: keying by line number means any edit near this statement breaks the gate in two places at once, and the fix is to re-point the line, never to drop the entry. Tightening the key to `file:value` would remove the trap but widens what the entry freezes, so it is left as a note rather than folded into a base-red drain. check:public-creds OK (3 frozen literals), check-public-creds tests 20/20, check:tracked-artifacts OK, prettier clean. --- scripts/check/check-public-creds.mjs | 8 ++++++-- tests/unit/check-public-creds.test.ts | 5 ++++- 2 files changed, 10 insertions(+), 3 deletions(-) diff --git a/scripts/check/check-public-creds.mjs b/scripts/check/check-public-creds.mjs index 7e065705d0..62549001bb 100644 --- a/scripts/check/check-public-creds.mjs +++ b/scripts/check/check-public-creds.mjs @@ -90,14 +90,18 @@ const ENV_KEY_RE = /(clientId|clientSecret|apiKey)Env\s*:/; // The MiniMax family was extracted from services/usage.ts into services/usage/minimax.ts // (god-file decomposition), so the FP moved with the getMiniMaxUsage signature. // -// open-sse/executors/zcodeProtocol.ts L302: `clientId: \`omniroute-${process.pid}\`` +// open-sse/executors/zcodeProtocol.ts L313: `clientId: \`omniroute-${process.pid}\`` // is the per-process identifier in the local ZCode app-server handshake. It is // generated from the process PID, is not an upstream OAuth/client credential, and // must remain visible in the wire contract. Frozen by file:line:value key. +// NOTE: the key includes the LINE, so any edit that shifts this statement breaks +// the gate twice over — a stale-entry error plus a "new violation" for the same +// literal. That is what happened here (L302 -> L313). Re-point the line; do not +// remove the entry. export const KNOWN_LITERAL_CREDS = new Set([ "open-sse/services/usage/minimax.ts:213:minimax", // TODO(6A.8): pre-existing FP — TS fn-param type, not a credential (getMiniMaxUsage signature) "open-sse/services/usage/minimax.ts:213:minimax-cn", // TODO(6A.8): pre-existing FP — TS fn-param type, not a credential (getMiniMaxUsage signature) - "open-sse/executors/zcodeProtocol.ts:302:omniroute-${process.pid}", // local per-process ZCode handshake ID, not an upstream credential + "open-sse/executors/zcodeProtocol.ts:313:omniroute-${process.pid}", // local per-process ZCode handshake ID, not an upstream credential ]); /** diff --git a/tests/unit/check-public-creds.test.ts b/tests/unit/check-public-creds.test.ts index 5531c5d336..b726218495 100644 --- a/tests/unit/check-public-creds.test.ts +++ b/tests/unit/check-public-creds.test.ts @@ -68,7 +68,10 @@ test("allowlist freezes a literal by file:line:value key", () => { }); test("allowlist preserves the local ZCode handshake client ID without weakening credential detection", () => { - const src = `${"\n".repeat(301)}clientId: \`omniroute-\${process.pid}\`,`; + // 312 newlines puts the statement on line 313, which is where it lives in + // zcodeProtocol.ts today. The allowlist key carries the line number, so this + // literal has to be kept in step with the source (it moved 302 -> 313). + const src = `${"\n".repeat(312)}clientId: \`omniroute-\${process.pid}\`,`; assert.deepEqual( findLiteralCreds(src, KNOWN_LITERAL_CREDS, "open-sse/executors/zcodeProtocol.ts"), [] From ad500de9e3b4ece9cab24ccec5678c1b0851e015 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 13:04:34 -0300 Subject: [PATCH 052/143] chore(quality): rebaseline file-size caps the hartmark batch grew past (#12623) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rebaseline medido no tip com os 9 PRs da leva hartmark mergeados. Não toca codex.ts nem stream.ts, drift anterior à leva. --- changelog.d/maintenance/hartmark-batch-filesize.md | 1 + config/quality/file-size-baseline.json | 7 ++++--- 2 files changed, 5 insertions(+), 3 deletions(-) create mode 100644 changelog.d/maintenance/hartmark-batch-filesize.md diff --git a/changelog.d/maintenance/hartmark-batch-filesize.md b/changelog.d/maintenance/hartmark-batch-filesize.md new file mode 100644 index 0000000000..d55f5b4029 --- /dev/null +++ b/changelog.d/maintenance/hartmark-batch-filesize.md @@ -0,0 +1 @@ +- **chore(quality):** rebaseline the file-size caps the hartmark batch grew past (`combos/page.tsx` via [#12355](https://github.com/diegosouzapw/OmniRoute/pull/12355), `open-sse/services/combo.ts` via [#12338](https://github.com/diegosouzapw/OmniRoute/pull/12338)) diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index 86ea2f3a86..b8415c5e8d 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -422,7 +422,7 @@ "open-sse/mcp-server/server.ts": 1572, "open-sse/services/accountFallback.ts": 2467, "open-sse/services/adobeFireflyBrowserLogin.ts": 1401, - "open-sse/services/combo.ts": 4023, + "open-sse/services/combo.ts": 4036, "open-sse/translator/response/openai-responses.ts": 1466, "open-sse/utils/cursorAgentProtobuf.ts": 1547, "open-sse/utils/proxyFetch.ts": 1271, @@ -431,7 +431,7 @@ "open-sse/vendor/codex-chatgpt-web/bridge.ts": 1322, "src/app/(dashboard)/dashboard/HomePageClient.tsx": 1344, "src/app/(dashboard)/dashboard/api-manager/ApiManagerPageClient.tsx": 3186, - "src/app/(dashboard)/dashboard/combos/page.tsx": 5012, + "src/app/(dashboard)/dashboard/combos/page.tsx": 5018, "src/app/(dashboard)/dashboard/costs/CostOverviewTab.tsx": 1319, "src/app/(dashboard)/dashboard/endpoint/EndpointPageClient.tsx": 2491, "src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx": 1631, @@ -634,5 +634,6 @@ "_rebaseline_2026_08_30_11771_vercel_gateway_passthrough": "PR #11771 adds passthroughModels: true (1 line) to the Vercel AI Gateway registry entry — no split available, single-line provider-config addition.", "_relax_velocity_2026_08_30": "127 frozen line caps and cap/testCap raised by 20% (velocity phase; see quality-baseline.json _policy).", "_rebaseline_2026_09_02_12325_merge_v3851": "Merge of release/v3.8.51 into #12325. Both sides grew chatCore.ts at the same chokepoint: #12239 took it 5946->5976 upstream, and this PR adds its +9 non-Codex 429 branch on top. check-file-size.mjs counts split(\"\\\\n\").length (trailing-newline empty element), so the merged file is 5981. The cap is the merged LOC, not either side alone; no other entry moves.", - "_rebaseline_2026_09_03_houminxi_batch_stacked": "Crescimento medido DEPOIS que os 9 PRs da leva HouMinXi entraram, quando cada um empilhou sobre o rebaseline do anterior: providers/page.tsx 2007->2025 (+18 = feedback de erro por linha do import CSV do #12504 somado a busca por nome/baseUrl do #12495, ambos no mesmo painel de conexoes); chatCore.ts 5981->5984 (+3 = o #12325 invalida o cache generico de quota no 429 upstream, ao lado do ramo Codex ja existente); accountFallback.ts 2461->2467 (+6 = o #12566 empilha a carve-out de familia Antigravity sobre o rebaseline 2422->2461 que o #12590 registrou para o carve-out credits_exhausted da Moonshot; os dois tocam checkFallbackError). Cada PR mediu certo isoladamente, mas nenhum enxergava o empilhamento. Fiacao em chokepoints existentes. NAO cobre codex.ts nem stream.ts, que ja violavam no tip antes desta leva (drift da base)." + "_rebaseline_2026_09_03_houminxi_batch_stacked": "Crescimento medido DEPOIS que os 9 PRs da leva HouMinXi entraram, quando cada um empilhou sobre o rebaseline do anterior: providers/page.tsx 2007->2025 (+18 = feedback de erro por linha do import CSV do #12504 somado a busca por nome/baseUrl do #12495, ambos no mesmo painel de conexoes); chatCore.ts 5981->5984 (+3 = o #12325 invalida o cache generico de quota no 429 upstream, ao lado do ramo Codex ja existente); accountFallback.ts 2461->2467 (+6 = o #12566 empilha a carve-out de familia Antigravity sobre o rebaseline 2422->2461 que o #12590 registrou para o carve-out credits_exhausted da Moonshot; os dois tocam checkFallbackError). Cada PR mediu certo isoladamente, mas nenhum enxergava o empilhamento. Fiacao em chokepoints existentes. NAO cobre codex.ts nem stream.ts, que ja violavam no tip antes desta leva (drift da base).", + "_rebaseline_2026_09_03_hartmark_batch": "Leva hartmark (#12293 #12355 #12447 #12445 #12446 #12460 #12461 #12338 #12448) crescimento proprio, medido no tip com os nove mergeados: src/app/(dashboard)/dashboard/combos/page.tsx 5012->5018 (+6, #12355 impede que a falha de bundling do tiktoken de um provider sem relacao derrube /api/providers, e o painel passa a lidar com o estado degradado); open-sse/services/combo.ts 4023->4036 (+13, #12338 nos fixes do universal-handoff: nota de bare-fallback, escopo por mesma requisicao e log da falha silenciosa). Fiacao em chokepoints existentes do roteamento de combo. NAO cobre codex.ts nem stream.ts, ja violando no tip antes desta leva (drift da base)." } From f51cd295c800cb3c3182affe916b236a5aa1ce08 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 13:49:46 -0300 Subject: [PATCH 053/143] chore(providers): bump Claude Code wire identity + Devin bridge pin to 2.1.258 (#12604) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 17 de 18 checks verdes, **zero falhas** — incluindo os quatro shards de unit, Vitest, CodeQL, semgrep, Docs Gates, Merge integrity e o **Fast Quality Gates**, que era exatamente o que o rebaseline do `RoutingTab.tsx` (1606→1607, a linha do seletor que acompanha a nova versão de identidade) veio consertar. O único check restante, `No new ESLint warnings`, ficou enfileirado sem iniciar (duração 0) atrás da saturação de runners de hoje. Cobri os dois comandos que ele executa, localmente e sobre este HEAD: - `npm run check:codeql-ratchet` → **0 alertas abertos** contra baseline 6 (sem regressão). - `npx eslint --max-warnings 0` nos 9 arquivos de código que esta branch altera → **exit 0**, nenhum warning. Conteúdo: os dois commits do #12402 que não estavam subsumidos, com autoria do @ggiak preservada pelo cherry-pick `-x`, mais o rebaseline e o fragmento de changelog que faltava. `typecheck:core` limpo e **138/138** nos testes focados. As três versões (`claudeCodeClient.ts`, `Dockerfile`, `compose.yml`) conferem em 2.1.258, e `npm view @anthropic-ai/claude-code@2.1.258` resolve — o Dockerfile instala esse pin exato. --- .env.example | 2 +- .../maintenance/12402-claude-code-2-1-258.md | 1 + config/quality/file-size-baseline.json | 3 ++- docker/devin-bridge/Dockerfile | 2 +- docker/devin-bridge/compose.yml | 2 +- docs/DEVIN_CLAUDE_BRIDGE.md | 18 +++++++++----- docs/providers/AGENTROUTER.md | 2 +- docs/reference/ENVIRONMENT.md | 2 +- docs/security/STEALTH_GUIDE.md | 6 ++--- .../settings/components/RoutingTab.tsx | 3 ++- src/shared/constants/claudeCodeClient.ts | 6 ++--- tests/snapshots/provider/translate-path.json | 24 +++++++++---------- .../unit/anthropic-cache-fingerprint.test.ts | 2 +- tests/unit/cc-bridge-transforms.test.ts | 2 +- ...claude-codex-identity-version-sync.test.ts | 10 ++++---- tests/unit/client-identity-profiles.test.ts | 6 ++--- tests/unit/executor-default-base.test.ts | 4 ++-- tests/unit/glm-executor.test.ts | 8 +++---- tests/unit/system-transforms.test.ts | 2 +- 19 files changed, 57 insertions(+), 48 deletions(-) create mode 100644 changelog.d/maintenance/12402-claude-code-2-1-258.md diff --git a/.env.example b/.env.example index 81b03d5b11..9527957b3d 100644 --- a/.env.example +++ b/.env.example @@ -1290,7 +1290,7 @@ GITHUB_OAUTH_CLIENT_ID=Iv1.b507a08c87ecfe98 # Used by: open-sse/executors/base.ts — buildHeaders() dynamic lookup. # Update these when providers release new CLI versions to avoid blocks. -CLAUDE_USER_AGENT="claude-cli/2.1.219 (external, cli)" +CLAUDE_USER_AGENT="claude-cli/2.1.258 (external, cli)" # Disable the deterministic tool-name cloak applied on both Anthropic-bound paths # (executors/base.ts native OAuth + executors/cliproxyapi.ts CLIProxyAPI) — diff --git a/changelog.d/maintenance/12402-claude-code-2-1-258.md b/changelog.d/maintenance/12402-claude-code-2-1-258.md new file mode 100644 index 0000000000..6b5ba0b108 --- /dev/null +++ b/changelog.d/maintenance/12402-claude-code-2-1-258.md @@ -0,0 +1 @@ +- **chore(providers):** bump the Claude Code wire identity and the Devin bridge image pin from `2.1.220` to `2.1.258` ([#12402](https://github.com/diegosouzapw/OmniRoute/pull/12402)) — thanks @ggiak diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index b8415c5e8d..c39d3d5d41 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -439,7 +439,7 @@ "src/app/(dashboard)/dashboard/runtime/RuntimePageClient.tsx": 1201, "src/app/(dashboard)/dashboard/settings/components/ProxyRegistryManager.tsx": 1475, "src/app/(dashboard)/dashboard/settings/components/ResilienceTab.tsx": 1271, - "src/app/(dashboard)/dashboard/settings/components/RoutingTab.tsx": 1606, + "src/app/(dashboard)/dashboard/settings/components/RoutingTab.tsx": 1607, "src/app/(dashboard)/dashboard/settings/components/SystemStorageTab.tsx": 1597, "src/app/(dashboard)/dashboard/usage/components/EvalsTab.tsx": 2152, "src/app/api/providers/[id]/models/route.ts": 2432, @@ -635,5 +635,6 @@ "_relax_velocity_2026_08_30": "127 frozen line caps and cap/testCap raised by 20% (velocity phase; see quality-baseline.json _policy).", "_rebaseline_2026_09_02_12325_merge_v3851": "Merge of release/v3.8.51 into #12325. Both sides grew chatCore.ts at the same chokepoint: #12239 took it 5946->5976 upstream, and this PR adds its +9 non-Codex 429 branch on top. check-file-size.mjs counts split(\"\\\\n\").length (trailing-newline empty element), so the merged file is 5981. The cap is the merged LOC, not either side alone; no other entry moves.", "_rebaseline_2026_09_03_houminxi_batch_stacked": "Crescimento medido DEPOIS que os 9 PRs da leva HouMinXi entraram, quando cada um empilhou sobre o rebaseline do anterior: providers/page.tsx 2007->2025 (+18 = feedback de erro por linha do import CSV do #12504 somado a busca por nome/baseUrl do #12495, ambos no mesmo painel de conexoes); chatCore.ts 5981->5984 (+3 = o #12325 invalida o cache generico de quota no 429 upstream, ao lado do ramo Codex ja existente); accountFallback.ts 2461->2467 (+6 = o #12566 empilha a carve-out de familia Antigravity sobre o rebaseline 2422->2461 que o #12590 registrou para o carve-out credits_exhausted da Moonshot; os dois tocam checkFallbackError). Cada PR mediu certo isoladamente, mas nenhum enxergava o empilhamento. Fiacao em chokepoints existentes. NAO cobre codex.ts nem stream.ts, que ja violavam no tip antes desta leva (drift da base).", + "_rebaseline_2026_09_03_12604_claude_code_2_1_258": "PR #12604 (bump da wire identity do Claude Code 2.1.220->2.1.258, commits do @ggiak vindos do #12402) crescimento proprio: src/app/(dashboard)/dashboard/settings/components/RoutingTab.tsx 1606->1607 (+1, a linha do seletor que acompanha a nova versao de identidade). Uma linha num painel de settings ja existente; nao ha o que extrair. Coberto por client-identity-profiles e claude-codex-identity-version-sync (138/138 focados).", "_rebaseline_2026_09_03_hartmark_batch": "Leva hartmark (#12293 #12355 #12447 #12445 #12446 #12460 #12461 #12338 #12448) crescimento proprio, medido no tip com os nove mergeados: src/app/(dashboard)/dashboard/combos/page.tsx 5012->5018 (+6, #12355 impede que a falha de bundling do tiktoken de um provider sem relacao derrube /api/providers, e o painel passa a lidar com o estado degradado); open-sse/services/combo.ts 4023->4036 (+13, #12338 nos fixes do universal-handoff: nota de bare-fallback, escopo por mesma requisicao e log da falha silenciosa). Fiacao em chokepoints existentes do roteamento de combo. NAO cobre codex.ts nem stream.ts, ja violando no tip antes desta leva (drift da base)." } diff --git a/docker/devin-bridge/Dockerfile b/docker/devin-bridge/Dockerfile index 3fd2a30dba..895455cf9f 100644 --- a/docker/devin-bridge/Dockerfile +++ b/docker/devin-bridge/Dockerfile @@ -1,6 +1,6 @@ FROM node:26.0.0-bookworm-slim -ARG CLAUDE_CODE_VERSION=2.1.220 +ARG CLAUDE_CODE_VERSION=2.1.258 ARG DEVIN_CLI_VERSION=3000.2.17 ARG TARGETARCH diff --git a/docker/devin-bridge/compose.yml b/docker/devin-bridge/compose.yml index c850414dbc..1a4ab8884b 100644 --- a/docker/devin-bridge/compose.yml +++ b/docker/devin-bridge/compose.yml @@ -28,7 +28,7 @@ x-runtime: &runtime context: ../.. dockerfile: docker/devin-bridge/Dockerfile args: - CLAUDE_CODE_VERSION: 2.1.220 + CLAUDE_CODE_VERSION: 2.1.258 DEVIN_CLI_VERSION: 3000.2.17 user: "10001:10001" read_only: true diff --git a/docs/DEVIN_CLAUDE_BRIDGE.md b/docs/DEVIN_CLAUDE_BRIDGE.md index 4a6d56d370..30b3bf3d4c 100644 --- a/docs/DEVIN_CLAUDE_BRIDGE.md +++ b/docs/DEVIN_CLAUDE_BRIDGE.md @@ -4,16 +4,22 @@ Messages endpoint while the official Devin CLI supplies model responses over ACP stdio. It does not modify the existing Anthropic, Claude OAuth, Claude Web, or `devin-cli` providers. -> **Current status: offline and live validated.** The pinned Claude Code `2.1.220` completed -> three isolated scenarios through Devin CLI `3000.2.17` and model -> `swe-1-7-lightning`. The final live run proved client-owned `Read`, `Edit`, and `Bash` -> turns, successful `npm test` results, project command and skill discovery, Devin-only -> routing, and zero Claude egress. +> **Current status: pinned Claude Code `2.1.258`; offline and live validation last recorded +> on `2.1.220`.** The `2.1.220` pin completed three isolated scenarios through Devin CLI +> `3000.2.17` and model `swe-1-7-lightning`; that final live run proved client-owned `Read`, +> `Edit`, and `Bash` turns, successful `npm test` results, project command and skill +> discovery, Devin-only routing, and zero Claude egress. The pin was then raised to `2.1.258` +> (the CLI generation OmniRoute's Claude identity impersonates, and the first line that ships +> the Fable 5.1 tier natively). On the new pin the install layer and `claude --version` were +> verified on the pinned base image, and the bridge unit suite, `compose config` and the +> static isolation proof pass — but the offline mock scenario and the live three-scenario +> suite have not been re-run yet. Re-run them (see "Updating pinned tools") before relying on +> the bridge with this pin. ## Architecture ```text -Claude Code 2.1.220 (isolated non-root Linux container) +Claude Code 2.1.258 (isolated non-root Linux container) -> http://omniroute:20128/v1/messages -> devin-cli-agentic (Claude-format, no-auth provider) -> devin acp --agent-type summarizer (official ACP stdio, no Devin tools) diff --git a/docs/providers/AGENTROUTER.md b/docs/providers/AGENTROUTER.md index aa28834c75..ea375f43c9 100644 --- a/docs/providers/AGENTROUTER.md +++ b/docs/providers/AGENTROUTER.md @@ -130,7 +130,7 @@ request (see `open-sse/services/claudeCodeCompatible.ts`): | Header | Value | | ------------------------------------------- | ------------------------------------------------------------------------------------------------------- | | `Authorization` | `Bearer ` | -| `User-Agent` | `claude-cli/2.1.219 (external, sdk-cli)` | +| `User-Agent` | `claude-cli/2.1.258 (external, sdk-cli)` | | `anthropic-version` | `2023-06-01` | | `anthropic-beta` | `claude-code-20250219,interleaved-thinking-2025-05-14,effort-2025-11-24` | | Per-connection redact-thinking beta toggle | Adds `redact-thinking-2026-02-12` for upstreams that specifically require redacted thinking streams | diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index 9dd2fff9f0..056a37beb9 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -622,7 +622,7 @@ process.env[`${PROVIDER_ID}_USER_AGENT`] | Variable | Default Value | When to Update | | -------------------------------- | --------------------------------------------- | ------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| `CLAUDE_USER_AGENT` | `claude-cli/2.1.219 (external, cli)` | When Anthropic releases a new CLI version | +| `CLAUDE_USER_AGENT` | `claude-cli/2.1.258 (external, cli)` | When Anthropic releases a new CLI version | | `CLAUDE_DISABLE_TOOL_NAME_CLOAK` | `false` | `executors/base.ts` + `executors/cliproxyapi.ts` | Set to `1`/`true` to forward third-party harness tool names verbatim to Anthropic on both Anthropic-bound paths (native OAuth and CLIProxyAPI). By default the executor deterministically aliases non-Claude-Code tool names (Claude Code canonical mapping where one exists, otherwise PascalCase) and reverses them on the response via `_toolNameMap`, so harnesses with snake_case tools are not refused as fingerprinted third-party clients. Debugging only. | | `CODEX_USER_AGENT` | `codex-cli/0.142.0 (Windows 10.0.26200; x64)` | When OpenAI updates the Codex CLI | | `CODEX_CLIENT_VERSION` | `0.131.0` | Override Codex client version independently of full UA string | diff --git a/docs/security/STEALTH_GUIDE.md b/docs/security/STEALTH_GUIDE.md index 456a558b0c..767eb90a0b 100644 --- a/docs/security/STEALTH_GUIDE.md +++ b/docs/security/STEALTH_GUIDE.md @@ -117,8 +117,8 @@ Applied to: `system` blocks, all `messages[].content`, and `tools[].description` For third-party Anthropic relays that only accept "real Claude Code" traffic: -- `CLAUDE_CODE_COMPATIBLE_USER_AGENT = "claude-cli/2.1.220 (external, sdk-cli)"` -- `CLAUDE_CODE_COMPATIBLE_STAINLESS_PACKAGE_VERSION = "0.94.0"` +- `CLAUDE_CODE_COMPATIBLE_USER_AGENT = "claude-cli/2.1.258 (external, sdk-cli)"` +- `CLAUDE_CODE_COMPATIBLE_STAINLESS_PACKAGE_VERSION = "0.112.1"` - `CLAUDE_CODE_COMPATIBLE_STAINLESS_RUNTIME_VERSION = "v26.3.0"` - `anthropic-beta = "claude-code-20250219,interleaved-thinking-2025-05-14,effort-2025-11-24"` by default - The per-connection "Enable redact-thinking beta" toggle adds `redact-thinking-2026-02-12` when a CC Compatible upstream specifically requires redacted thinking streams @@ -241,7 +241,7 @@ All MITM endpoints require management auth (`requireCliToolsAuth`). The sudo pas | Variable | Default | | ------------------------ | --------------------------------------------------------------- | -| `CLAUDE_USER_AGENT` | `claude-cli/2.1.220 (external, cli)` | +| `CLAUDE_USER_AGENT` | `claude-cli/2.1.258 (external, cli)` | | `CODEX_USER_AGENT` | `codex-cli/0.149.0 (Windows 10.0.26200; x64)` | | `GITHUB_USER_AGENT` | `GitHubCopilotChat/0.54.0` | | `ANTIGRAVITY_USER_AGENT` | `antigravity/2.0.1 linux/arm64 google-api-nodejs-client/10.3.0` | diff --git a/src/app/(dashboard)/dashboard/settings/components/RoutingTab.tsx b/src/app/(dashboard)/dashboard/settings/components/RoutingTab.tsx index 5c844f80cb..f23b36ff5d 100644 --- a/src/app/(dashboard)/dashboard/settings/components/RoutingTab.tsx +++ b/src/app/(dashboard)/dashboard/settings/components/RoutingTab.tsx @@ -11,6 +11,7 @@ import { normalizeCliCompatProviderId, } from "@/shared/constants/cliCompatProviders"; import { AI_PROVIDERS } from "@/shared/constants/providers"; +import { CLAUDE_CODE_CLIENT_BUILD_REVISION } from "@/shared/constants/claudeCodeClient"; import { compareTr } from "@/shared/utils/turkishText"; import { HERMES } from "./systemTransformsHermesDefaults"; @@ -169,7 +170,7 @@ const DEFAULT_SYSTEM_TRANSFORMS_CLIENT = { entrypoint: "sdk-cli", versionFormat: "ex-machina", cchAlgo: "sha256-first-user", - buildRevision: "1f2", + buildRevision: CLAUDE_CODE_CLIENT_BUILD_REVISION, }, ], }, diff --git a/src/shared/constants/claudeCodeClient.ts b/src/shared/constants/claudeCodeClient.ts index 415068bc2d..dd72f6246e 100644 --- a/src/shared/constants/claudeCodeClient.ts +++ b/src/shared/constants/claudeCodeClient.ts @@ -4,10 +4,10 @@ * Keep this leaf dependency-free so server executors, compatibility bridges, * and client-facing identity presets can share one source of truth. */ -export const CLAUDE_CODE_CLIENT_VERSION = "2.1.220"; -export const CLAUDE_CODE_CLIENT_BUILD_REVISION = "1f2"; +export const CLAUDE_CODE_CLIENT_VERSION = "2.1.258"; +export const CLAUDE_CODE_CLIENT_BUILD_REVISION = "1e2"; export const CLAUDE_CODE_CLIENT_BILLING_VERSION = `${CLAUDE_CODE_CLIENT_VERSION}.${CLAUDE_CODE_CLIENT_BUILD_REVISION}`; -export const CLAUDE_CODE_SDK_PACKAGE_VERSION = "0.94.0"; +export const CLAUDE_CODE_SDK_PACKAGE_VERSION = "0.112.1"; export const CLAUDE_CODE_RUNTIME_VERSION = "v26.3.0"; export type ClaudeCodeEntrypoint = "cli" | "sdk-cli"; diff --git a/tests/snapshots/provider/translate-path.json b/tests/snapshots/provider/translate-path.json index 63db4728b1..fbd4d1d22c 100644 --- a/tests/snapshots/provider/translate-path.json +++ b/tests/snapshots/provider/translate-path.json @@ -28,11 +28,11 @@ "apiKey": { "Accept": "text/event-stream", "Content-Type": "application/json", - "User-Agent": "claude-cli/2.1.220 (external, sdk-cli)", + "User-Agent": "claude-cli/2.1.258 (external, sdk-cli)", "X-Stainless-Arch": "", "X-Stainless-Lang": "js", "X-Stainless-OS": "MacOS", - "X-Stainless-Package-Version": "0.94.0", + "X-Stainless-Package-Version": "0.112.1", "X-Stainless-Retry-Count": "0", "X-Stainless-Runtime": "node", "X-Stainless-Runtime-Version": "v26.3.0", @@ -47,11 +47,11 @@ "nonStream": { "Accept": "application/json", "Content-Type": "application/json", - "User-Agent": "claude-cli/2.1.220 (external, sdk-cli)", + "User-Agent": "claude-cli/2.1.258 (external, sdk-cli)", "X-Stainless-Arch": "", "X-Stainless-Lang": "js", "X-Stainless-OS": "MacOS", - "X-Stainless-Package-Version": "0.94.0", + "X-Stainless-Package-Version": "0.112.1", "X-Stainless-Retry-Count": "0", "X-Stainless-Runtime": "node", "X-Stainless-Runtime-Version": "v26.3.0", @@ -66,11 +66,11 @@ "oauth": { "Accept": "text/event-stream", "Content-Type": "application/json", - "User-Agent": "claude-cli/2.1.220 (external, sdk-cli)", + "User-Agent": "claude-cli/2.1.258 (external, sdk-cli)", "X-Stainless-Arch": "", "X-Stainless-Lang": "js", "X-Stainless-OS": "MacOS", - "X-Stainless-Package-Version": "0.94.0", + "X-Stainless-Package-Version": "0.112.1", "X-Stainless-Retry-Count": "0", "X-Stainless-Runtime": "node", "X-Stainless-Runtime-Version": "v26.3.0", @@ -989,13 +989,13 @@ "Anthropic-Dangerous-Direct-Browser-Access": "true", "Anthropic-Version": "2023-06-01", "Content-Type": "application/json", - "User-Agent": "claude-cli/2.1.220 (external, cli)", + "User-Agent": "claude-cli/2.1.258 (external, cli)", "X-App": "cli", "X-Stainless-Arch": "", "X-Stainless-Helper-Method": "stream", "X-Stainless-Lang": "js", "X-Stainless-Os": "", - "X-Stainless-Package-Version": "0.94.0", + "X-Stainless-Package-Version": "0.112.1", "X-Stainless-Retry-Count": "0", "X-Stainless-Runtime": "node", "X-Stainless-Runtime-Version": "v26.3.0", @@ -1007,13 +1007,13 @@ "Anthropic-Dangerous-Direct-Browser-Access": "true", "Anthropic-Version": "2023-06-01", "Content-Type": "application/json", - "User-Agent": "claude-cli/2.1.220 (external, cli)", + "User-Agent": "claude-cli/2.1.258 (external, cli)", "X-App": "cli", "X-Stainless-Arch": "", "X-Stainless-Helper-Method": "stream", "X-Stainless-Lang": "js", "X-Stainless-Os": "", - "X-Stainless-Package-Version": "0.94.0", + "X-Stainless-Package-Version": "0.112.1", "X-Stainless-Retry-Count": "0", "X-Stainless-Runtime": "node", "X-Stainless-Runtime-Version": "v26.3.0", @@ -1026,13 +1026,13 @@ "Anthropic-Dangerous-Direct-Browser-Access": "true", "Anthropic-Version": "2023-06-01", "Content-Type": "application/json", - "User-Agent": "claude-cli/2.1.220 (external, cli)", + "User-Agent": "claude-cli/2.1.258 (external, cli)", "X-App": "cli", "X-Stainless-Arch": "", "X-Stainless-Helper-Method": "stream", "X-Stainless-Lang": "js", "X-Stainless-Os": "", - "X-Stainless-Package-Version": "0.94.0", + "X-Stainless-Package-Version": "0.112.1", "X-Stainless-Retry-Count": "0", "X-Stainless-Runtime": "node", "X-Stainless-Runtime-Version": "v26.3.0", diff --git a/tests/unit/anthropic-cache-fingerprint.test.ts b/tests/unit/anthropic-cache-fingerprint.test.ts index ebdb9690af..59c736a63c 100644 --- a/tests/unit/anthropic-cache-fingerprint.test.ts +++ b/tests/unit/anthropic-cache-fingerprint.test.ts @@ -5,6 +5,6 @@ import { CLAUDE_CODE_CLIENT_BILLING_VERSION } from "../../src/shared/constants/c describe("Anthropic billing header fingerprint (#1638)", () => { it("uses the immutable build revision captured from the signed CLI", () => { - assert.equal(CLAUDE_CODE_CLIENT_BILLING_VERSION, "2.1.220.1f2"); + assert.equal(CLAUDE_CODE_CLIENT_BILLING_VERSION, "2.1.258.1e2"); }); }); diff --git a/tests/unit/cc-bridge-transforms.test.ts b/tests/unit/cc-bridge-transforms.test.ts index ba915dcaca..cbaf19c50a 100644 --- a/tests/unit/cc-bridge-transforms.test.ts +++ b/tests/unit/cc-bridge-transforms.test.ts @@ -52,7 +52,7 @@ test("DEFAULT_CC_BRIDGE_PIPELINE places billing header at [0] and identity at [1 DEFAULT_CC_BRIDGE_PIPELINE ); const blocks = result.body.system as any[]; - assert.ok(blocks[0].text.startsWith("x-anthropic-billing-header: cc_version=2.1.220.1f2;")); + assert.ok(blocks[0].text.startsWith("x-anthropic-billing-header: cc_version=2.1.258.1e2;")); assert.equal(blocks[1].text, CLAUDE_AGENT_SDK_IDENTITY); }); diff --git a/tests/unit/claude-codex-identity-version-sync.test.ts b/tests/unit/claude-codex-identity-version-sync.test.ts index d274bb9f34..cda482fba9 100644 --- a/tests/unit/claude-codex-identity-version-sync.test.ts +++ b/tests/unit/claude-codex-identity-version-sync.test.ts @@ -38,11 +38,11 @@ test("Claude CLI version constants are in lockstep across all 4 sources", () => ); }); -test("Claude CLI wire versions match the captured 2.1.220 binary", () => { - assert.equal(canonical.CLAUDE_CODE_CLIENT_VERSION, "2.1.220"); - assert.equal(canonical.CLAUDE_CODE_CLIENT_BUILD_REVISION, "1f2"); - assert.equal(canonical.CLAUDE_CODE_CLIENT_BILLING_VERSION, "2.1.220.1f2"); - assert.equal(canonical.CLAUDE_CODE_SDK_PACKAGE_VERSION, "0.94.0"); +test("Claude CLI wire versions match the captured 2.1.258 binary", () => { + assert.equal(canonical.CLAUDE_CODE_CLIENT_VERSION, "2.1.258"); + assert.equal(canonical.CLAUDE_CODE_CLIENT_BUILD_REVISION, "1e2"); + assert.equal(canonical.CLAUDE_CODE_CLIENT_BILLING_VERSION, "2.1.258.1e2"); + assert.equal(canonical.CLAUDE_CODE_SDK_PACKAGE_VERSION, "0.112.1"); assert.equal(canonical.CLAUDE_CODE_RUNTIME_VERSION, "v26.3.0"); assert.equal( compat.CLAUDE_CODE_COMPATIBLE_STAINLESS_PACKAGE_VERSION, diff --git a/tests/unit/client-identity-profiles.test.ts b/tests/unit/client-identity-profiles.test.ts index 54e00736a5..6c24c54eaa 100644 --- a/tests/unit/client-identity-profiles.test.ts +++ b/tests/unit/client-identity-profiles.test.ts @@ -39,7 +39,7 @@ test("getClientIdentityProfileHeaders: unknown profile id falls back to no heade test("getClientIdentityProfileHeaders: known CLI profiles expose their preset headers", () => { const claudeCli = getClientIdentityProfileHeaders("claude-cli"); - assert.equal(claudeCli["User-Agent"], "claude-cli/2.1.220 (external, cli)"); + assert.equal(claudeCli["User-Agent"], "claude-cli/2.1.258 (external, cli)"); assert.equal(claudeCli["X-App"], "cli"); const codexCli = getClientIdentityProfileHeaders("codex-cli"); @@ -55,7 +55,7 @@ test("getClientIdentityProfileHeaders: returns a fresh mutable copy (catalog sta headers["User-Agent"] = "tampered"; assert.equal( CLIENT_IDENTITY_PROFILES["claude-cli"].headers["User-Agent"], - "claude-cli/2.1.220 (external, cli)" + "claude-cli/2.1.258 (external, cli)" ); }); @@ -100,7 +100,7 @@ test("profile headers merged into customHeaders survive applyCustomHeaders sanit true ) as Record; - assert.equal(headers["User-Agent"], "claude-cli/2.1.220 (external, cli)"); + assert.equal(headers["User-Agent"], "claude-cli/2.1.258 (external, cli)"); assert.equal(headers["X-App"], "cli"); assert.equal(headers["Authorization"], "Bearer test-key"); }); diff --git a/tests/unit/executor-default-base.test.ts b/tests/unit/executor-default-base.test.ts index 0e21678054..469ffe992c 100644 --- a/tests/unit/executor-default-base.test.ts +++ b/tests/unit/executor-default-base.test.ts @@ -1571,12 +1571,12 @@ test("DefaultExecutor.execute does not produce duplicate anthropic-version heade assert.equal(versionKeys.length, 1, "Duplicate anthropic-version header keys found"); assert.equal(capturedHeaders[versionKeys[0]], "2023-06-01"); assert.equal(capturedHeaders["X-Stainless-Runtime-Version"], "v26.3.0"); - assert.equal(capturedHeaders["X-Stainless-Package-Version"], "0.94.0"); + assert.equal(capturedHeaders["X-Stainless-Package-Version"], "0.112.1"); const sentBody = JSON.parse(capturedBody) as { system?: Array<{ text?: string }> }; assert.match( sentBody.system?.[0]?.text ?? "", - /^x-anthropic-billing-header: cc_version=2\.1\.220\.1f2; cc_entrypoint=cli; cch=[0-9a-f]{5};$/ + /^x-anthropic-billing-header: cc_version=2\.1\.258\.1e2; cc_entrypoint=cli; cch=[0-9a-f]{5};$/ ); }); diff --git a/tests/unit/glm-executor.test.ts b/tests/unit/glm-executor.test.ts index 4cb31e90f6..26ef4f8526 100644 --- a/tests/unit/glm-executor.test.ts +++ b/tests/unit/glm-executor.test.ts @@ -130,9 +130,9 @@ test("GlmExecutor normalizes GLM coding and Anthropic URLs without duplicating e }); test("GlmExecutor separates OpenAI-compatible coding headers from Anthropic headers", async () => { - assert.equal(await getExecutor("glm") instanceof GlmExecutor, true); - assert.equal(await getExecutor("glm-cn") instanceof GlmExecutor, true); - assert.equal(await getExecutor("glmt") instanceof GlmExecutor, true); + assert.equal((await getExecutor("glm")) instanceof GlmExecutor, true); + assert.equal((await getExecutor("glm-cn")) instanceof GlmExecutor, true); + assert.equal((await getExecutor("glmt")) instanceof GlmExecutor, true); const executor = new GlmExecutor("glm"); const codingHeaders = executor.buildHeaders( @@ -183,7 +183,7 @@ test("GlmExecutor separates OpenAI-compatible coding headers from Anthropic head assert.equal(anthropicHeaders["anthropic-version"], "2023-06-01"); assert.match(anthropicHeaders["anthropic-beta"], /claude-code-20250219/); assert.equal(anthropicHeaders["anthropic-dangerous-direct-browser-access"], "true"); - assert.match(anthropicHeaders["User-Agent"], /^claude-cli\/2\.1\.220 \(external, sdk-cli\)$/); + assert.match(anthropicHeaders["User-Agent"], /^claude-cli\/2\.1\.258 \(external, sdk-cli\)$/); assert.equal(anthropicHeaders["X-Stainless-Lang"], "js"); assert.equal(anthropicHeaders["X-Stainless-Runtime"], "node"); }); diff --git a/tests/unit/system-transforms.test.ts b/tests/unit/system-transforms.test.ts index 047063d663..e5683766eb 100644 --- a/tests/unit/system-transforms.test.ts +++ b/tests/unit/system-transforms.test.ts @@ -559,7 +559,7 @@ const UI_DEFAULTS_SNAPSHOT = { entrypoint: "sdk-cli", versionFormat: "ex-machina", cchAlgo: "sha256-first-user", - buildRevision: "1f2", + buildRevision: "1e2", }, ], }, From 910f58c5cc664cccef25d79dc73d2230c474990d Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 17:26:52 -0300 Subject: [PATCH 054/143] docs(quality): document how the CodeQL ratchet refreshes and how to tighten it (#12611) Co-authored-by: Markus Hartung --- docs/architecture/QUALITY_GATES.md | 52 +++++++++++++++++++++++------- 1 file changed, 41 insertions(+), 11 deletions(-) diff --git a/docs/architecture/QUALITY_GATES.md b/docs/architecture/QUALITY_GATES.md index a3bca79c6b..4b4315cc4d 100644 --- a/docs/architecture/QUALITY_GATES.md +++ b/docs/architecture/QUALITY_GATES.md @@ -90,17 +90,17 @@ Runs on every PR to `main`. Blocks merge on failure. Runs after `test-coverage`. Blocks merge on failure. -| Script | Validates | Blocking | -| ---------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------- | -| `quality:collect` | Emits `quality-metrics.json` (ESLint warning count, coverage from merged shard report) | Yes (upstream of ratchet) | -| `quality:ratchet` | Each metric in `quality-baseline.json` has not regressed (ESLint warnings ≤ baseline; coverage ≥ baseline) | Yes | -| `check:duplication` | Code duplication (jscpd@4) does not exceed baseline in `quality-baseline.json` | Yes | -| `check:complexity` | File-level cyclomatic complexity does not exceed the cap (core ESLint `complexity` + `max-lines-per-function`) | Yes | -| `check:cognitive-complexity` | Cognitive complexity ratchet (`eslint-plugin-sonarjs`) — separate ESLint pass; CI runs both merged as the single `check:complexity-ratchets` step | Yes | -| `check:dead-code` | Unused exports / files ratchet (knip) does not regress vs baseline | Yes | -| `check:compression-budget` | Compression benchmark budget — per-engine token-savings floors must not regress | Yes | -| `check:type-coverage` | Percent-typed ratchet (`type-coverage`) does not regress; largely subsumes `typecheck:noimplicit:core` | Yes | -| `check:codeql-ratchet` | Open CodeQL alert count does not regress (reads via `gh api`; graceful-skip without token) | Yes | +| Script | Validates | Blocking | +| ---------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------- | +| `quality:collect` | Emits `quality-metrics.json` (ESLint warning count, coverage from merged shard report) | Yes (upstream of ratchet) | +| `quality:ratchet` | Each metric in `quality-baseline.json` has not regressed (ESLint warnings ≤ baseline; coverage ≥ baseline) | Yes | +| `check:duplication` | Code duplication (jscpd@4) does not exceed baseline in `quality-baseline.json` | Yes | +| `check:complexity` | File-level cyclomatic complexity does not exceed the cap (core ESLint `complexity` + `max-lines-per-function`) | Yes | +| `check:cognitive-complexity` | Cognitive complexity ratchet (`eslint-plugin-sonarjs`) — separate ESLint pass; CI runs both merged as the single `check:complexity-ratchets` step | Yes | +| `check:dead-code` | Unused exports / files ratchet (knip) does not regress vs baseline | Yes | +| `check:compression-budget` | Compression benchmark budget — per-engine token-savings floors must not regress | Yes | +| `check:type-coverage` | Percent-typed ratchet (`type-coverage`) does not regress; largely subsumes `typecheck:noimplicit:core` | Yes | +| `check:codeql-ratchet` | Open CodeQL alert count does not regress (reads via `gh api`; graceful-skip without token) — refresh cadence and manual trigger: see "CodeQL ratchet" below | Yes | ### Job: `quality-extended` @@ -324,6 +324,36 @@ Commit this file alongside the change that improved the metric. A PR that improv metric without updating the baseline will be caught by `--require-tighten` (Fase 6A.5, pending implementation). +### CodeQL ratchet: refresh cadence and manual trigger + +`check:codeql-ratchet` reads **repo state, refreshed on a schedule — not per PR.** +`gh api repos/diegosouzapw/OmniRoute/code-scanning/default-setup` reports +`state: configured`, `schedule: weekly`: GitHub's default-setup scan, not a per-push +analysis. Consequence: after a PR that FIXES alerts merges, the ratchet keeps reading +the old, higher count until the next scheduled scan runs — so it reports a regression +on every open PR, including the fixing PR's own follow-ups, until the scan catches up. + +**Manual refresh**: `gh workflow run codeql.yml --ref release/vX.Y.Z` re-runs the +analysis and republishes alerts within minutes. Read `.github/workflows/codeql.yml` +first — its header explains it is `workflow_dispatch`-only **because it conflicts with +GitHub's "default setup"** (`CodeQL analyses from advanced configurations cannot be +processed when the default setup is enabled`). Restoring `push`/`pull_request`/ +`schedule` triggers requires an **owner action first**: Settings → Code security → +CodeQL: Default → Advanced. Do not add a `schedule:` trigger without that switch — it +will only produce failing runs. + +**Tighten the baseline after the count drops** — `node scripts/check/check-codeql-ratchet.mjs +--update` writes the new measured count into `quality-baseline.json` → +`metrics.codeqlAlerts.value`, so the ratchet does not silently permit a regression back +up to the old ceiling. Worked example (2026-09-02/03): PR #12502 fixed 7 real alerts +(13 → 6 measured open); PR #12530 tightened the frozen baseline 11 → 6 to match; the +remaining 6 were then dismissed with per-alert justification down to 0 open. + +**Dismissals are the operator's call (Hard Rule #14)** — never dismiss a CodeQL alert +without recording the technical justification in the dismissal comment: `won't fix` for +an upstream-protocol requirement, `used in tests` for a test fixture, `false positive` +for a sanitizer CodeQL cannot see (precedent: `docs/security/ERROR_SANITIZATION.md`). + --- ## Test Retry Policy (WS5.4, v3.8.49) From 8a95a2bced078fae2abafec011e54dab9a17fd96 Mon Sep 17 00:00:00 2001 From: Davide Baraldo Date: Fri, 4 Sep 2026 01:48:40 +0200 Subject: [PATCH 055/143] fix(settings): cache-config alwaysPreserveClientCache was a runtime no-op (#12304) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em worktree combinada sobre o tip de release/v3.8.51: os dois boardaram sem conflito, typecheck:core limpo, check-file-size sem violação nova (as duas restantes — codex.ts e stream.ts — são drift anterior) e 51/51 nos 5 arquivos de teste que os PRs trazem. --- ...2304-cache-config-preserve-client-cache.md | 1 + src/app/api/settings/cache-config/route.ts | 30 +++++-- ...cache-config-preserve-client-cache.test.ts | 82 +++++++++++++++++++ 3 files changed, 104 insertions(+), 9 deletions(-) create mode 100644 changelog.d/fixes/12304-cache-config-preserve-client-cache.md create mode 100644 tests/unit/cache-config-preserve-client-cache.test.ts diff --git a/changelog.d/fixes/12304-cache-config-preserve-client-cache.md b/changelog.d/fixes/12304-cache-config-preserve-client-cache.md new file mode 100644 index 0000000000..1c4da7f4c7 --- /dev/null +++ b/changelog.d/fixes/12304-cache-config-preserve-client-cache.md @@ -0,0 +1 @@ +- **fix(settings):** `PUT /api/settings/cache-config` now persists `alwaysPreserveClientCache` to the flat general settings the runtime cache-control policy actually reads; previously the value landed in the databaseSettings "cache" section and was silently ignored, so the endpoint had no effect on `cache_control` passthrough ([#12304](https://github.com/diegosouzapw/OmniRoute/pull/12304)) — thanks @davidebaraldo diff --git a/src/app/api/settings/cache-config/route.ts b/src/app/api/settings/cache-config/route.ts index fc5a6634aa..cbf777b634 100644 --- a/src/app/api/settings/cache-config/route.ts +++ b/src/app/api/settings/cache-config/route.ts @@ -58,8 +58,12 @@ export async function GET(request: NextRequest) { const flatSettings = await getSettings(); const config: Record = {}; for (const key of CACHE_CONFIG_KEYS) { - if (key === "idempotencyWindowMs") { - config[key] = flatSettings.idempotencyWindowMs ?? DEFAULTS[key]; + if (key === "idempotencyWindowMs" || key === "alwaysPreserveClientCache") { + // These live in the flat general settings (src/lib/db/settings.ts): + // idempotencyLayer and getCacheControlSettings() both read from there, + // so reporting the databaseSettings "cache" copy would show a value the + // runtime never uses. + config[key] = flatSettings[key] ?? DEFAULTS[key]; } else { config[key] = (cache as Record)[key] ?? DEFAULTS[key]; } @@ -106,9 +110,6 @@ export async function PUT(request: NextRequest) { if (body.promptCacheStrategy !== undefined) { updates.promptCacheStrategy = body.promptCacheStrategy; } - if (body.alwaysPreserveClientCache !== undefined) { - updates.alwaysPreserveClientCache = body.alwaysPreserveClientCache; - } if (body.modelCatalogCacheTtlMs !== undefined) { updates.modelCatalogCacheTtlMs = body.modelCatalogCacheTtlMs; } @@ -116,12 +117,23 @@ export async function PUT(request: NextRequest) { // updateDatabaseSettings() calls invalidateDbCache("settings") internally, // which bumps the model-catalog cache version so in-flight responses pick // up the fresh TTL — no separate version bump needed here. - updateDatabaseSettings({ cache: updates }); + if (Object.keys(updates).length > 0) { + updateDatabaseSettings({ cache: updates }); + } - // idempotencyWindowMs is not part of the databaseSettings "cache" section — - // persist it through the flat general settings module instead (see GET). + // idempotencyWindowMs and alwaysPreserveClientCache are read from the flat + // general settings (see GET) — persisting them into the databaseSettings + // "cache" section would be a silent no-op for the runtime, which is what + // made this endpoint's alwaysPreserveClientCache writes ineffective before. + const flatUpdates: Record = {}; if (body.idempotencyWindowMs !== undefined) { - await updateSettings({ idempotencyWindowMs: body.idempotencyWindowMs }); + flatUpdates.idempotencyWindowMs = body.idempotencyWindowMs; + } + if (body.alwaysPreserveClientCache !== undefined) { + flatUpdates.alwaysPreserveClientCache = body.alwaysPreserveClientCache; + } + if (Object.keys(flatUpdates).length > 0) { + await updateSettings(flatUpdates); } return NextResponse.json({ ok: true }); diff --git a/tests/unit/cache-config-preserve-client-cache.test.ts b/tests/unit/cache-config-preserve-client-cache.test.ts new file mode 100644 index 0000000000..ab2b3c825e --- /dev/null +++ b/tests/unit/cache-config-preserve-client-cache.test.ts @@ -0,0 +1,82 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +// Regression guard: PUT /api/settings/cache-config persisted +// `alwaysPreserveClientCache` into the databaseSettings "cache" section, but +// the runtime (getCacheControlSettings → getSettings) reads the FLAT general +// settings key — so the endpoint accepted the value, GET echoed it back, and +// the router never changed behavior. This test proves the value written +// through the route is the one the cache-control policy actually consumes. + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-cache-config-flat-")); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.DISABLE_SQLITE_AUTO_BACKUP = "true"; + +const core = await import("../../src/lib/db/core.ts"); + +function resetStorage() { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +function makeJsonRequest(method: string, body?: unknown): Request { + return new Request("http://localhost/api/settings/cache-config", { + method, + headers: { "Content-Type": "application/json" }, + body: body === undefined ? undefined : JSON.stringify(body), + }); +} + +test.beforeEach(() => { + resetStorage(); +}); + +test.after(() => { + resetStorage(); +}); + +test("alwaysPreserveClientCache set via cache-config reaches the runtime read path", async (t) => { + const cacheConfigRoute = await import("../../src/app/api/settings/cache-config/route.ts"); + const { getSettings } = await import("../../src/lib/db/settings.ts"); + const { getCacheControlSettings, invalidateCacheControlSettingsCache } = + await import("../../src/lib/cacheControlSettings.ts"); + + await t.test("PUT persists to the flat settings the runtime reads", async () => { + const putResponse = await cacheConfigRoute.PUT( + makeJsonRequest("PUT", { alwaysPreserveClientCache: "always" }) as never + ); + assert.equal(putResponse.status, 200); + + // The runtime read path: getCacheControlSettings() → getSettings() (flat). + // RED before the fix: the route wrote databaseSettings "cache" instead, + // so both of these still reported the default "auto". + const flatSettings = await getSettings(); + assert.equal(flatSettings.alwaysPreserveClientCache, "always"); + + invalidateCacheControlSettingsCache(); + assert.equal(await getCacheControlSettings(), "always"); + }); + + await t.test("GET reports the flat value, not the ignored cache-section copy", async () => { + // Seed a stale value in the databaseSettings "cache" section — the store + // the runtime never reads. GET must not surface it. + const { updateDatabaseSettings } = await import("../../src/lib/db/databaseSettings.ts"); + updateDatabaseSettings({ + cache: { alwaysPreserveClientCache: "never" }, + } as Parameters[0]); + + const putResponse = await cacheConfigRoute.PUT( + makeJsonRequest("PUT", { alwaysPreserveClientCache: "always" }) as never + ); + assert.equal(putResponse.status, 200); + + const getResponse = await cacheConfigRoute.GET(makeJsonRequest("GET") as never); + const body = await getResponse.json(); + assert.equal(getResponse.status, 200); + assert.equal(body.alwaysPreserveClientCache, "always"); + }); +}); From 4a37c7f46eb43ddece4494de5fe6e90af4aa241b Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 20:48:58 -0300 Subject: [PATCH 056/143] =?UTF-8?q?fix(security):=20close=203=20advisories?= =?UTF-8?q?=20=E2=80=94=20search=20baseUrl=20exfil,=20sk-=20in=20the=20err?= =?UTF-8?q?or=20sanitizer,=20bifrost=20relay=20header=20leak=20(#12620)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em worktree combinada sobre o tip de release/v3.8.51: os dois boardaram sem conflito, typecheck:core limpo, check-file-size sem violação nova (as duas restantes — codex.ts e stream.ts — são drift anterior) e 51/51 nos 5 arquivos de teste que os PRs trazem. --- open-sse/config/searchRegistry.ts | 16 ++ open-sse/handlers/search.ts | 23 +-- open-sse/handlers/search/baseUrl.ts | 66 ++++++++ open-sse/utils/error.ts | 30 +++- open-sse/utils/upstreamErrorPassthrough.ts | 22 ++- open-sse/utils/upstreamResponseHeaders.ts | 33 ++++ .../relay/chat/completions/bifrost/route.ts | 35 +++- .../api/v1/relay/chat/completions/route.ts | 5 +- stryker.conf.json | 1 + .../bifrost-relay-response-leak-9m72.test.ts | 103 ++++++++++++ .../unit/error-sanitizer-sk-key-qv45.test.ts | 93 ++++++++++ ...earch-baseurl-client-override-3f8g.test.ts | 159 ++++++++++++++++++ tests/unit/search-baseurl-ssrf-guard.test.ts | 14 +- 13 files changed, 570 insertions(+), 30 deletions(-) create mode 100644 open-sse/handlers/search/baseUrl.ts create mode 100644 tests/unit/bifrost-relay-response-leak-9m72.test.ts create mode 100644 tests/unit/error-sanitizer-sk-key-qv45.test.ts create mode 100644 tests/unit/search-baseurl-client-override-3f8g.test.ts diff --git a/open-sse/config/searchRegistry.ts b/open-sse/config/searchRegistry.ts index e4d4546212..3e9105ad35 100644 --- a/open-sse/config/searchRegistry.ts +++ b/open-sse/config/searchRegistry.ts @@ -33,6 +33,17 @@ export interface SearchProviderConfig { */ fallbackOnly?: boolean; disabled?: boolean; + /** + * May a CALLER-supplied `provider_options.baseUrl` redirect this provider? + * + * Only ever true for a keyless, self-hosted provider. For anything with + * `authType: "apikey"` the builder attaches the OPERATOR's key to whatever + * host the base URL resolves to, so honoring a caller-chosen value hands that + * key to the caller's server (GHSA-3f8g-pfh9-j687). The invariant + * "never set alongside authType: apikey" is enforced by + * tests/unit/search-baseurl-client-override-3f8g.test.ts. + */ + allowClientBaseUrlOverride?: boolean; } export const SEARCH_PROVIDERS: Record = { @@ -232,6 +243,11 @@ export const SEARCH_PROVIDERS: Record = { timeoutMs: 10_000, cacheTTLMs: 3 * 60 * 1000, fallbackOnly: true, + // Keyless and self-hosted by definition: the caller names their own SearXNG + // instance and no operator credential travels with the request. This is a + // documented flow (tests/unit/search-route.test.ts). Still block-metadata + // guarded, so IMDS stays unreachable. + allowClientBaseUrlOverride: true, }, "ollama-search": { diff --git a/open-sse/handlers/search.ts b/open-sse/handlers/search.ts index b1397f3c5d..e68603e459 100644 --- a/open-sse/handlers/search.ts +++ b/open-sse/handlers/search.ts @@ -20,6 +20,9 @@ import { randomUUID } from "crypto"; * } */ +export { resolveSearchBaseUrl, SearchBaseUrlOverrideError } from "./search/baseUrl.ts"; +import { resolveSearchBaseUrl } from "./search/baseUrl.ts"; + import { getSearchProvider, isUnconfiguredLoopbackSearchProvider, @@ -36,7 +39,6 @@ import * as anysearchSearch from "./search/anysearchSearch.ts"; import { freeWebSearch } from "../services/freeWebSearch.ts"; import { saveCallLog } from "@/lib/usageDb"; import { safeOutboundFetch } from "@/shared/network/safeOutboundFetch"; -import { parseAndValidateNonMetadataUrl } from "@/shared/network/outboundUrlGuard"; import { Client } from "@modelcontextprotocol/sdk/client/index.js"; import { StreamableHTTPClientTransport } from "@modelcontextprotocol/sdk/client/streamableHttp.js"; import { z } from "zod"; @@ -319,25 +321,6 @@ function getProviderSettingString( return undefined; } -export function resolveSearchBaseUrl( - config: SearchProviderConfig, - params: SearchRequestParams -): string { - const override = getProviderSettingString(params, "baseUrl"); - if (override) { - // GHSA-j7j4-g9qc-q69c: the override is client-controlled (provider_options / - // providerSpecificData) and flows into a plain fetch() sink — validate it - // before any builder uses it as the server-side fetch target. Mode is - // block-metadata (NOT public-only): the primary searxng use case is a - // self-hosted instance on loopback/LAN, so private hosts keep working, - // while cloud-metadata endpoints (IMDS credential theft) are rejected. - // The catalog's own config.baseUrl is operator config and stays untouched. - parseAndValidateNonMetadataUrl(override); - return override.replace(/\/+$/, ""); - } - return config.baseUrl.replace(/\/+$/, ""); -} - function toSearchPageNumber(offset: number | undefined, maxResults: number): number | undefined { if (typeof offset !== "number" || offset <= 0 || maxResults <= 0) return undefined; return Math.floor(offset / maxResults) + 1; diff --git a/open-sse/handlers/search/baseUrl.ts b/open-sse/handlers/search/baseUrl.ts new file mode 100644 index 0000000000..ba1e1fe6b0 --- /dev/null +++ b/open-sse/handlers/search/baseUrl.ts @@ -0,0 +1,66 @@ +/** + * Base-URL resolution for /v1/search — the trust decision, on its own. + * + * The two override sources are NOT equally trusted, and reading them through + * one call is what produced GHSA-3f8g-pfh9-j687: + * + * - `providerSpecificData` is the stored provider connection + * (`credentials?.providerSpecificData`) — OPERATOR config. Honored under + * block-metadata, so a self-hosted SearXNG on loopback/LAN keeps working + * while cloud metadata (IMDS credential theft) stays rejected (GHSA-j7j4). + * - `providerOptions` is `body.provider_options` — CALLER input. Honored only + * by a provider that is keyless AND opted in via + * `allowClientBaseUrlOverride`. A keyed builder attaches the OPERATOR's key + * to whatever host this resolves to (`key=`/`api_key=` in the query for + * google-pse/searchapi, `X-API-Key`/`Authorization` for + * you.com/linkup/nimble/ollama), so a caller-chosen host would collect it — + * and a block-metadata check does nothing about that, because the attacker + * simply names their own public host. + * + * Full coverage: tests/unit/search-baseurl-client-override-3f8g.test.ts. + */ + +import { parseAndValidateNonMetadataUrl } from "@/shared/network/outboundUrlGuard"; +import type { SearchProviderConfig } from "../../config/searchRegistry.ts"; + +interface BaseUrlParams { + providerOptions?: Record; + providerSpecificData?: Record; +} + +/** Read one string setting from a SINGLE source, so callers can distinguish trust. */ +function readSetting(source: Record | undefined, key: string): string | undefined { + const value = source?.[key]; + return typeof value === "string" && value.trim().length > 0 ? value.trim() : undefined; +} + +/** Refusal for a caller-supplied `provider_options.baseUrl` (GHSA-3f8g-pfh9-j687). */ +export class SearchBaseUrlOverrideError extends Error { + readonly code = "SEARCH_BASE_URL_OVERRIDE_REFUSED"; + constructor(providerId: string) { + super( + `provider_options.baseUrl is not accepted for search provider "${providerId}". ` + + `Set the base URL on the provider connection instead.` + ); + this.name = "SearchBaseUrlOverrideError"; + } +} + +export function resolveSearchBaseUrl(config: SearchProviderConfig, params: BaseUrlParams): string { + const operatorOverride = readSetting(params.providerSpecificData, "baseUrl"); + if (operatorOverride) { + parseAndValidateNonMetadataUrl(operatorOverride); + return operatorOverride.replace(/\/+$/, ""); + } + + const callerOverride = readSetting(params.providerOptions, "baseUrl"); + if (callerOverride) { + if (!config.allowClientBaseUrlOverride || config.authType === "apikey") { + throw new SearchBaseUrlOverrideError(config.id); + } + parseAndValidateNonMetadataUrl(callerOverride); + return callerOverride.replace(/\/+$/, ""); + } + + return config.baseUrl.replace(/\/+$/, ""); +} diff --git a/open-sse/utils/error.ts b/open-sse/utils/error.ts index 4fafd05b1a..66563c618f 100644 --- a/open-sse/utils/error.ts +++ b/open-sse/utils/error.ts @@ -40,8 +40,36 @@ function looksLikeAbsolutePath(tok: string): boolean { return (SOURCE_EXT as readonly string[]).includes(ext); } +/** + * Raw credential shapes that carry no `key=` label to key off — the token IS the + * whole match, so the only way to redact them is to recognize the shape. + * + * GHSA-qv45-56jc-4wmj: `upstreamErrorPassthrough.ts` already recognized `sk-` + * and refused verbatim passthrough for bodies containing it, then handed those + * bodies to THIS sanitizer — which had no such pattern, so the key came back to + * the caller anyway. The passthrough file's comment claimed to "mirror the + * vocabulary of redactSensitiveErrorText"; the mirror had drifted. It now + * imports this array instead of keeping a second copy, so the two cannot drift + * again. + * + * Quantifiers are upper-bounded (AGENTS.md → PII learnings §1, ReDoS): these run + * over untrusted upstream error bodies. + */ +export const RAW_CREDENTIAL_PATTERNS: ReadonlyArray = [ + // OpenAI/Anthropic/Stripe-style secret keys: sk-…, sk-ant-…, sk_live_… + /\bsk[-_][A-Za-z0-9._-]{8,200}/g, + // Google API keys + /\bAIza[A-Za-z0-9_-]{20,200}/g, + // JWTs (three base64url segments) + /\beyJ[A-Za-z0-9_-]{8,400}\.[A-Za-z0-9_-]{8,800}\.[A-Za-z0-9_-]{8,800}/g, +]; + export function redactSensitiveErrorText(value: string): string { - return value + let out = value; + for (const pattern of RAW_CREDENTIAL_PATTERNS) { + out = out.replace(pattern, "[REDACTED_CREDENTIAL]"); + } + return out .replace(/data:[^,\s]+;base64,[A-Za-z0-9+/=_-]+/gi, "[REDACTED_DATA_URL]") .replace(/\b(Bearer|Basic)\s+[A-Za-z0-9._~+/=-]+/gi, "$1 [REDACTED]") .replace( diff --git a/open-sse/utils/upstreamErrorPassthrough.ts b/open-sse/utils/upstreamErrorPassthrough.ts index 21d0c6c964..b62fff2adf 100644 --- a/open-sse/utils/upstreamErrorPassthrough.ts +++ b/open-sse/utils/upstreamErrorPassthrough.ts @@ -1,3 +1,4 @@ +import { RAW_CREDENTIAL_PATTERNS } from "./error.ts"; /** * Selective upstream 4xx error passthrough (Claude Code auto-recover contract). * @@ -24,8 +25,23 @@ const INTERNAL_LEAK_RE = /\sat\s\/|node_modules|omniroute\//i; // caller fall back to the sanitized buildErrorBody path. Bodies without a // secret (the overwhelming majority, carrying capability/quota wording) still // relay verbatim. Mirrors the vocabulary of redactSensitiveErrorText in error.ts. -const CREDENTIAL_LEAK_RE = - /\b(?:Bearer|Basic)\s+[A-Za-z0-9._~+/=-]{8,}|\bsk-[A-Za-z0-9._-]{8,}|(?:api[_-]?key|access[_-]?token|refresh[_-]?token|authorization|cookie|secret)\\?["']?\s*[:=]\s*\\?["']?[^"'\\,\s}]{6,}/i; +const LABELLED_CREDENTIAL_RE = + /\b(?:Bearer|Basic)\s+[A-Za-z0-9._~+/=-]{8,}|(?:api[_-]?key|access[_-]?token|refresh[_-]?token|authorization|cookie|secret)\\?["']?\s*[:=]\s*\\?["']?[^"'\\,\s}]{6,}/i; + +/** + * The raw-token shapes (sk-…, AIza…, JWT) come from error.ts's + * RAW_CREDENTIAL_PATTERNS rather than a second local copy. The previous local + * copy carried `sk-` while the sanitizer this file falls back to did NOT, so a + * body recognized as leaky here was returned unredacted there + * (GHSA-qv45-56jc-4wmj). One source, no drift. + */ +function containsCredential(text: string): boolean { + if (LABELLED_CREDENTIAL_RE.test(text)) return true; + return RAW_CREDENTIAL_PATTERNS.some((pattern) => { + pattern.lastIndex = 0; // the shared patterns are /g — reset before .test() + return pattern.test(text); + }); +} export function shouldPassthroughUpstreamError(statusCode: number, upstreamBody: unknown): boolean { if (statusCode < PASSTHROUGH_MIN || statusCode > PASSTHROUGH_MAX) return false; @@ -34,7 +50,7 @@ export function shouldPassthroughUpstreamError(statusCode: number, upstreamBody: const text = JSON.stringify(upstreamBody); if (INTERNAL_LEAK_RE.test(text)) return false; // Refuse passthrough when the provider echoed a credential back to us. - if (CREDENTIAL_LEAK_RE.test(text)) return false; + if (containsCredential(text)) return false; return true; } diff --git a/open-sse/utils/upstreamResponseHeaders.ts b/open-sse/utils/upstreamResponseHeaders.ts index 834d354f18..9f6c16d1e6 100644 --- a/open-sse/utils/upstreamResponseHeaders.ts +++ b/open-sse/utils/upstreamResponseHeaders.ts @@ -45,3 +45,36 @@ export function filterUpstreamResponseHeaderEntries( } export const STRIP_UPSTREAM_HEADER_NAMES: ReadonlySet = STRIP_HEADER_NAMES; + +/** + * Response headers that must never be relayed back to a client. + * + * A relay sends its own credential upstream (the bifrost route sends + * `Authorization: Bearer ${BIFROST_API_KEY}` to the sidecar). If that upstream + * echoes the header back — or sets its own session cookie — copying the response + * headers wholesale hands it to whoever holds the relay token + * (GHSA-9m72-44hg-w32g). `set-cookie` matters as much as `authorization`: it is + * a session, and the browser would store it against OUR origin. + */ +const SENSITIVE_RESPONSE_HEADER_NAMES: ReadonlyArray = [ + "authorization", + "proxy-authorization", + "x-api-key", + "x-goog-api-key", + "api-key", + "cookie", + "set-cookie", +]; + +/** + * New Headers with the stale framing set AND any echoed credential/session + * header removed. Use this instead of `new Headers(upstream.headers)` on every + * path that relays an upstream response to a client. Does not mutate the input. + */ +export function stripSensitiveResponseHeaders(input: Headers): Headers { + return new Headers( + filterUpstreamResponseHeaderEntries(input.entries(), SENSITIVE_RESPONSE_HEADER_NAMES) + ); +} + +export { SENSITIVE_RESPONSE_HEADER_NAMES }; diff --git a/src/app/api/v1/relay/chat/completions/bifrost/route.ts b/src/app/api/v1/relay/chat/completions/bifrost/route.ts index 31b007959f..8674125c38 100644 --- a/src/app/api/v1/relay/chat/completions/bifrost/route.ts +++ b/src/app/api/v1/relay/chat/completions/bifrost/route.ts @@ -29,9 +29,14 @@ */ import { CORS_HEADERS, handleCorsOptions } from "@/shared/utils/cors"; +import { stripSensitiveResponseHeaders } from "@omniroute/open-sse/utils/upstreamResponseHeaders"; import { createInjectionGuard } from "@/middleware/promptInjectionGuard"; import { getRelayTokenByHash, checkRateLimit, recordRelayUsage } from "@/lib/db/relayProxies"; -import { buildErrorBody } from "@omniroute/open-sse/utils/error"; +import { + buildErrorBody, + parseUpstreamError, + sanitizeErrorMessage, +} from "@omniroute/open-sse/utils/error"; import { getProviderPluginManifestHeader } from "@omniroute/open-sse/config/providerPluginManifestUrl.ts"; import { z } from "zod"; import { @@ -299,7 +304,11 @@ export async function POST(request: Request) { }); }; - const newHeaders = new Headers(upstream.headers); + // Never copy the upstream headers wholesale: the sidecar receives our + // `Authorization: Bearer ${BIFROST_API_KEY}` and anything it echoes back — + // that header, its own set-cookie — would reach the relay-token holder + // (GHSA-9m72-44hg-w32g). + const newHeaders = stripSensitiveResponseHeaders(upstream.headers); newHeaders.set("X-Routed-By", "bifrost"); newHeaders.set("X-Relay-Token", token.tokenPrefix + "..."); if (!wantsStream) { @@ -321,6 +330,28 @@ export async function POST(request: Request) { } clearTimeout(tid); + + // Normalize non-2xx through parseUpstreamError + buildErrorBody instead of + // relaying the sidecar body verbatim — parity with the TS sibling route and + // Hard Rule #12 (GHSA-9m72-44hg-w32g). + if (!upstream.ok) { + const parsed = await parseUpstreamError(upstream, null); + const errorBody = buildErrorBody( + parsed.statusCode, + sanitizeErrorMessage(parsed.message), + parsed.responseBody + ); + newHeaders.set("Content-Type", "application/json"); + if (parsed.retryAfterMs && parsed.retryAfterMs > 0) { + newHeaders.set("Retry-After", String(Math.ceil(parsed.retryAfterMs / 1000))); + } + recordUsage("error", parsed.statusCode); + return new Response(JSON.stringify(errorBody), { + status: parsed.statusCode, + headers: newHeaders, + }); + } + recordUsage(upstream.status < 500 ? "success" : "error", upstream.status); return new Response(upstream.body, { diff --git a/src/app/api/v1/relay/chat/completions/route.ts b/src/app/api/v1/relay/chat/completions/route.ts index fa40d4a009..4e43bb7a13 100644 --- a/src/app/api/v1/relay/chat/completions/route.ts +++ b/src/app/api/v1/relay/chat/completions/route.ts @@ -7,6 +7,7 @@ */ import { CORS_HEADERS, handleCorsOptions } from "@/shared/utils/cors"; +import { stripSensitiveResponseHeaders } from "@omniroute/open-sse/utils/upstreamResponseHeaders"; import { handleChat } from "@/sse/handlers/chat"; import { withChatAdmission } from "@/shared/middleware/withChatAdmission"; import { createInjectionGuard } from "@/middleware/promptInjectionGuard"; @@ -109,7 +110,9 @@ async function forwardToBifrost( signal: ac.signal, }); - const headers = new Headers(upstream.headers); + // Same strip as the bifrost sibling: an echoed upstream credential or + // set-cookie must not reach the relay-token holder (GHSA-9m72-44hg-w32g). + const headers = stripSensitiveResponseHeaders(upstream.headers); headers.set("X-Routed-By", "bifrost"); headers.set("X-Routing-Backend", "bifrost"); headers.set("X-Relay-Token", token.tokenPrefix + "..."); diff --git a/stryker.conf.json b/stryker.conf.json index 96f421f624..27e2825662 100644 --- a/stryker.conf.json +++ b/stryker.conf.json @@ -244,6 +244,7 @@ "tests/unit/embeddings-auth.test.ts", "tests/unit/error-classification.test.ts", "tests/unit/error-message-sanitization.test.ts", + "tests/unit/error-sanitizer-sk-key-qv45.test.ts", "tests/unit/error-sensitive-redaction.test.ts", "tests/unit/execute-chat-resource-pressure-breaker.test.ts", "tests/unit/executor-antigravity.test.ts", diff --git a/tests/unit/bifrost-relay-response-leak-9m72.test.ts b/tests/unit/bifrost-relay-response-leak-9m72.test.ts new file mode 100644 index 0000000000..c5affaf655 --- /dev/null +++ b/tests/unit/bifrost-relay-response-leak-9m72.test.ts @@ -0,0 +1,103 @@ +/** + * GHSA-9m72-44hg-w32g — the standalone bifrost relay route copied ALL upstream + * response headers (`new Headers(upstream.headers)`) and returned non-2xx bodies + * verbatim, while its TypeScript sibling routed non-2xx through + * parseUpstreamError + buildErrorBody + stripStaleEncodingHeaders. + * + * The relay sends `Authorization: Bearer ${BIFROST_API_KEY}` to the sidecar, so + * anything the sidecar (or a further upstream) echoes back — that header, its own + * `set-cookie`, an `x-api-key` — reached the relay-token holder untouched. + * + * Both relay routes copy upstream headers, so the strip is a shared helper used + * by both rather than a fix in one and a second copy waiting to drift (the + * failure mode of GHSA-v7g9 and GHSA-qv45). + * + * Run with: + * node --import tsx/esm --test tests/unit/bifrost-relay-response-leak-9m72.test.ts + */ + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { fileURLToPath } from "node:url"; + +import { stripSensitiveResponseHeaders } from "../../open-sse/utils/upstreamResponseHeaders.ts"; + +const read = (rel: string) => + readFileSync(fileURLToPath(new URL(`../../${rel}`, import.meta.url)), "utf8"); + +const BIFROST_ROUTE = "src/app/api/v1/relay/chat/completions/bifrost/route.ts"; +const TS_ROUTE = "src/app/api/v1/relay/chat/completions/route.ts"; + +describe("stripSensitiveResponseHeaders", () => { + it("drops credentials and cookies the upstream echoed back", () => { + const upstream = new Headers([ + ["authorization", "Bearer sidecar-secret"], + ["x-api-key", "op-key"], + ["x-goog-api-key", "goog-key"], + ["api-key", "azure-key"], + ["cookie", "session=abc"], + ["set-cookie", "session=abc; HttpOnly"], + ["proxy-authorization", "Basic zzz"], + ["content-type", "application/json"], + ["x-request-id", "keep-me"], + ]); + const out = stripSensitiveResponseHeaders(upstream); + for (const gone of [ + "authorization", + "x-api-key", + "x-goog-api-key", + "api-key", + "cookie", + "set-cookie", + "proxy-authorization", + ]) { + assert.equal(out.get(gone), null, `${gone} survived`); + } + assert.equal(out.get("content-type"), "application/json"); + assert.equal(out.get("x-request-id"), "keep-me"); + }); + + it("also drops the stale framing headers", () => { + const out = stripSensitiveResponseHeaders( + new Headers([ + ["content-encoding", "gzip"], + ["content-length", "123"], + ["transfer-encoding", "chunked"], + ["x-keep", "yes"], + ]) + ); + assert.equal(out.get("content-encoding"), null); + assert.equal(out.get("content-length"), null); + assert.equal(out.get("transfer-encoding"), null); + assert.equal(out.get("x-keep"), "yes"); + }); + + it("does not mutate the input Headers", () => { + const input = new Headers([["authorization", "Bearer x"]]); + stripSensitiveResponseHeaders(input); + assert.equal(input.get("authorization"), "Bearer x"); + }); +}); + +describe("both relay routes use the shared strip (GHSA-9m72-44hg-w32g)", () => { + for (const route of [BIFROST_ROUTE, TS_ROUTE]) { + it(`${route} strips sensitive upstream response headers`, () => { + const src = read(route); + assert.ok( + src.includes("stripSensitiveResponseHeaders"), + `${route} relays upstream headers verbatim — a sidecar-echoed credential reaches the caller` + ); + assert.ok( + !/new Headers\(upstream\.headers\)/.test(src), + `${route} still copies upstream headers wholesale` + ); + }); + } + + it("the bifrost route normalizes non-2xx through the error sanitizer", () => { + const src = read(BIFROST_ROUTE); + assert.ok(src.includes("buildErrorBody"), "bifrost non-2xx must not be relayed verbatim"); + assert.ok(src.includes("sanitizeErrorMessage"), "bifrost non-2xx must be sanitized (HR#12)"); + }); +}); diff --git a/tests/unit/error-sanitizer-sk-key-qv45.test.ts b/tests/unit/error-sanitizer-sk-key-qv45.test.ts new file mode 100644 index 0000000000..b92d5ebbb1 --- /dev/null +++ b/tests/unit/error-sanitizer-sk-key-qv45.test.ts @@ -0,0 +1,93 @@ +/** + * GHSA-qv45-56jc-4wmj — two copies of one redaction rule, and only one got the + * `sk-` pattern. + * + * `upstreamErrorPassthrough.ts`'s CREDENTIAL_LEAK_RE matches `\bsk-[…]{8,}` and + * REFUSES verbatim passthrough when an upstream 4xx echoes a key — correctly + * treating it as a leak. The body then falls through to `buildErrorBody` → + * `sanitizeErrorMessage` → `redactSensitiveErrorText`, which had no `sk-` + * pattern at all. So the layer that recognized the credential handed it to a + * layer that did not, and OpenAI-style `Incorrect API key provided: sk-proj-…` + * bodies were returned to the caller verbatim. + * + * The passthrough file's own comment says it "mirrors the vocabulary of + * redactSensitiveErrorText" — the mirror had diverged. This suite pins both + * directions so it cannot diverge again. + * + * Run with: + * node --import tsx/esm --test tests/unit/error-sanitizer-sk-key-qv45.test.ts + */ + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; + +import { redactSensitiveErrorText, sanitizeErrorMessage } from "../../open-sse/utils/error.ts"; + +const LEAKY_BODIES = [ + "Incorrect API key provided: sk-proj-AbCdEfGhIjKlMnOpQrStUv. You can find your API key at …", + "401 Unauthorized: sk-ant-api03-abcdefghijklmnopqrstuvwxyz-1234567890", + "invalid key sk_live_51H8xKzAbCdEfGhIjKlMn", + "Bad credentials for AIzaSyA1B2C3D4E5F6G7H8I9J0KaLbMcNdOeP", + "token rejected: eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.eyJzdWIiOiIxMjM0NTY3ODkwIn0.dozjgNryP4J3jVmNHl0w5N_XgL0n3I9PlFUP0THsR8U", +]; + +describe("redactSensitiveErrorText — raw credential patterns (GHSA-qv45-56jc-4wmj)", () => { + for (const body of LEAKY_BODIES) { + it(`redacts the raw credential in: ${body.slice(0, 42)}…`, () => { + const out = redactSensitiveErrorText(body); + assert.ok(!/\bsk[-_][A-Za-z0-9._-]{8,}/.test(out), `sk- survived: ${out}`); + assert.ok(!/\bAIza[A-Za-z0-9_-]{20,}/.test(out), `Google key survived: ${out}`); + assert.ok(!/\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\./.test(out), `JWT survived: ${out}`); + assert.ok(out.includes("[REDACTED"), `nothing was redacted: ${out}`); + }); + } + + it("redacts through sanitizeErrorMessage, the path buildErrorBody actually uses", () => { + const out = sanitizeErrorMessage("Incorrect API key provided: sk-proj-AbCdEfGhIjKlMnOpQr."); + assert.ok(!out.includes("sk-proj-AbCdEfGhIjKlMnOpQr"), out); + }); + + it("keeps the pre-existing redactions working", () => { + assert.match(redactSensitiveErrorText("401: Bearer abc123def456"), /Bearer \[REDACTED\]/); + assert.match( + redactSensitiveErrorText('{"api_key":"secret-value","detail":"bad"}'), + /\[REDACTED\]/ + ); + assert.match( + redactSensitiveErrorText("data:image/png;base64,AAAABBBBCCCC"), + /\[REDACTED_DATA_URL\]/ + ); + }); + + it("does not maul ordinary error prose that merely contains 'sk'", () => { + for (const benign of [ + "Model gpt-5 is not available on this plan", + "risk score too high", + "task sk failed", // short, no credential shape + "Rate limit reached for requests", + ]) { + assert.equal(redactSensitiveErrorText(benign), benign); + } + }); +}); + +describe("the two redaction layers stay in step", () => { + it("every credential shape the passthrough layer refuses is also redacted here", async () => { + // If passthrough REFUSES a body as leaky, the fallback sanitizer is the only + // thing standing between that body and the caller. Anything the first layer + // calls a credential, the second must scrub. + const { shouldPassthroughUpstreamError } = + await import("../../open-sse/utils/upstreamErrorPassthrough.ts"); + for (const body of LEAKY_BODIES) { + const payload = { error: { message: body } }; + const relayedVerbatim = shouldPassthroughUpstreamError(401, payload); + if (relayedVerbatim) continue; // not classified as a leak — nothing to assert + const scrubbed = redactSensitiveErrorText(body); + assert.notEqual( + scrubbed, + body, + `passthrough refused this body as leaky but the sanitizer left it untouched: ${body}` + ); + } + }); +}); diff --git a/tests/unit/search-baseurl-client-override-3f8g.test.ts b/tests/unit/search-baseurl-client-override-3f8g.test.ts new file mode 100644 index 0000000000..17ada75dce --- /dev/null +++ b/tests/unit/search-baseurl-client-override-3f8g.test.ts @@ -0,0 +1,159 @@ +/** + * GHSA-3f8g-pfh9-j687 — a tenant-supplied `provider_options.baseUrl` redirected + * the SaaS search builders, so the OPERATOR's search-provider API key was sent + * to a tenant-chosen host (query string for google-pse/searchapi, header for + * you.com/linkup/nimble/ollama). + * + * This is the same override the GHSA-j7j4-g9qc-q69c fix hardened — and that fix + * only added a block-metadata check, which does nothing here: the attacker + * points at their own PUBLIC host and collects the key. The j7j4 regression test + * missed it because its fixture was `searxng-search`, the one keyless provider + * where redirecting the base URL leaks no credential. + * + * The two override sources have different trust: + * - `providerSpecificData` comes from the stored provider connection + * (`credentials?.providerSpecificData`, search.ts:1510) — OPERATOR config. + * This is how an operator points at their self-hosted searxng, so it keeps + * working on loopback/LAN under the block-metadata policy. + * - `providerOptions` comes straight off the request body + * (`body.provider_options`, v1/search/route.ts:366) — TENANT input. It is + * refused outright: no provider needs a per-request caller-chosen fetch + * target, and honoring one is credential exfiltration for keyed providers + * and SSRF-with-readback for every provider. + * + * Run with: + * node --import tsx/esm --test tests/unit/search-baseurl-client-override-3f8g.test.ts + */ + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; + +import { resolveSearchBaseUrl } from "../../open-sse/handlers/search.ts"; +import type { SearchProviderConfig } from "../../open-sse/config/searchRegistry.ts"; +import { SEARCH_PROVIDERS } from "../../open-sse/config/searchRegistry.ts"; + +const base = { query: "test", searchType: "web", maxResults: 5 }; + +/** Every provider whose builder attaches an operator-held key. */ +const KEYED_PROVIDER_IDS = Object.values(SEARCH_PROVIDERS) + .filter((p) => p.authType === "apikey") + .map((p) => p.id); + +describe("resolveSearchBaseUrl — tenant-supplied baseUrl (GHSA-3f8g-pfh9-j687)", () => { + it("covers every keyed provider in the registry, not one hand-picked fixture", () => { + // The j7j4 test only exercised searxng. Assert the registry actually has + // keyed providers so this suite cannot silently degrade to zero coverage. + assert.ok( + KEYED_PROVIDER_IDS.length >= 5, + `expected keyed providers, got ${KEYED_PROVIDER_IDS.length}` + ); + }); + + for (const id of KEYED_PROVIDER_IDS) { + it(`refuses a tenant baseUrl for the keyed provider ${id}`, () => { + const config = SEARCH_PROVIDERS[id]; + assert.throws( + () => + resolveSearchBaseUrl(config, { + ...base, + providerOptions: { baseUrl: "https://attacker.example" }, + }), + `${id} honored a tenant baseUrl — the operator key would be sent there` + ); + }); + } + + it("the opt-in flag is NEVER set on a provider that carries an operator key", () => { + // The whole invariant in one assertion: if this ever pairs with authType + // "apikey", a caller can redirect the operator's key again. + for (const p of Object.values(SEARCH_PROVIDERS)) { + if (p.allowClientBaseUrlOverride) { + assert.equal(p.authType, "none", `${p.id} opts into a caller baseUrl while holding a key`); + } + } + }); + + it("refuses a caller baseUrl for keyless providers that did NOT opt in", () => { + for (const id of ["context7", "duckduckgo-free"]) { + const config = SEARCH_PROVIDERS[id]; + if (!config || config.allowClientBaseUrlOverride) continue; + assert.throws( + () => + resolveSearchBaseUrl(config, { + ...base, + providerOptions: { baseUrl: "https://attacker.example" }, + }), + `${id} honored a caller baseUrl without opting in` + ); + } + }); + + it("SearXNG keeps its documented self-hosted caller override, IMDS still blocked", () => { + // Opted in: keyless, so no operator credential travels with the request. + // Loopback/LAN is the point of a self-hosted instance (search-route.test.ts + // covers the route-level flow); cloud metadata stays rejected. + const config = SEARCH_PROVIDERS["searxng-search"]; + assert.equal(config.allowClientBaseUrlOverride, true); + assert.equal( + resolveSearchBaseUrl(config, { + ...base, + providerOptions: { baseUrl: "http://127.0.0.1:8888/search" }, + }), + "http://127.0.0.1:8888/search" + ); + assert.throws(() => + resolveSearchBaseUrl(config, { + ...base, + providerOptions: { baseUrl: "http://169.254.169.254/latest/meta-data/" }, + }) + ); + }); + + it("keeps the operator-configured override working, including loopback/LAN", () => { + const config = SEARCH_PROVIDERS["searxng-search"]; + assert.equal( + resolveSearchBaseUrl(config, { + ...base, + providerSpecificData: { baseUrl: "http://127.0.0.1:8888/search" }, + }), + "http://127.0.0.1:8888/search" + ); + assert.equal( + resolveSearchBaseUrl(config, { + ...base, + providerSpecificData: { baseUrl: "http://10.0.0.5:8888" }, + }), + "http://10.0.0.5:8888" + ); + }); + + it("still blocks cloud metadata from the operator source (j7j4 must not regress)", () => { + const config = SEARCH_PROVIDERS["searxng-search"]; + assert.throws(() => + resolveSearchBaseUrl(config, { + ...base, + providerSpecificData: { baseUrl: "http://169.254.169.254/latest/meta-data/" }, + }) + ); + }); + + it("the operator source wins over a tenant one instead of the tenant shadowing it", () => { + const config = SEARCH_PROVIDERS["searxng-search"]; + assert.equal( + resolveSearchBaseUrl(config, { + ...base, + providerOptions: { baseUrl: "https://attacker.example" }, + providerSpecificData: { baseUrl: "http://127.0.0.1:8888" }, + }), + "http://127.0.0.1:8888" + ); + }); + + it("falls back to the catalog baseUrl when neither source supplies one", () => { + const config: SearchProviderConfig = { + ...SEARCH_PROVIDERS["searxng-search"], + baseUrl: "http://localhost:8888/search", + }; + assert.equal(resolveSearchBaseUrl(config, base), "http://localhost:8888/search"); + }); +}); diff --git a/tests/unit/search-baseurl-ssrf-guard.test.ts b/tests/unit/search-baseurl-ssrf-guard.test.ts index f42335edd3..5e4e1a6217 100644 --- a/tests/unit/search-baseurl-ssrf-guard.test.ts +++ b/tests/unit/search-baseurl-ssrf-guard.test.ts @@ -58,18 +58,26 @@ describe("resolveSearchBaseUrl — SSRF guard on client-controlled baseUrl (GHSA }); } - it("still allows a self-hosted loopback/LAN override (block-metadata, not public-only)", () => { + // CORRECTED for GHSA-3f8g-pfh9-j687. This case originally asserted the + // loopback/LAN override via `providerOptions` — i.e. via TENANT input — which + // is the SSRF-with-readback half of that advisory. The self-hosted use case + // this test was written to protect is operator configuration, and it arrives + // on `providerSpecificData` (search.ts reads it from the stored connection's + // `credentials?.providerSpecificData`), so that is where it is asserted now. + // Tenant-supplied overrides are refused outright — see + // search-baseurl-client-override-3f8g.test.ts. + it("still allows a self-hosted loopback/LAN override from OPERATOR config", () => { assert.equal( resolveSearchBaseUrl(config, { ...base, - providerOptions: { baseUrl: "http://127.0.0.1:9999" }, + providerSpecificData: { baseUrl: "http://127.0.0.1:9999" }, }), "http://127.0.0.1:9999" ); assert.equal( resolveSearchBaseUrl(config, { ...base, - providerOptions: { baseUrl: "http://10.0.0.5:8080" }, + providerSpecificData: { baseUrl: "http://10.0.0.5:8080" }, }), "http://10.0.0.5:8080" ); From 239d8fc67d8ff0b30bfa75aee052f21d3556ec05 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 20:56:44 -0300 Subject: [PATCH 057/143] fix(providers): separate MaxAI and UC credential contracts (#12431) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 14 PRs desta campanha de error-boundary sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **120/120** nos 23 arquivos de teste que os PRs trazem. Um ponto que só apareceu no tree combinado: **#12465 e #12466 criam o mesmo arquivo novo** `open-sse/utils/streamReadiness.ts` (que não existe no tip) com desenhos divergentes de cancelamento — `cancelled` + `releaseLock` imediato num, `readInFlight`/`cancelRequested` com `cancelReader` fire-and-forget no outro. Adotei a versão do #12466, que difere e defere o release do lock para quando a leitura em voo termina, e validei a escolha rodando as suítes dos **dois** PRs contra ela: 21/21 no readiness compartilhado e 22/22 incluindo o boundary do Perplexity. --- tests/unit/web-session-credentials.test.ts | 27 ++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/tests/unit/web-session-credentials.test.ts b/tests/unit/web-session-credentials.test.ts index 68c1a64565..12c38a5c06 100644 --- a/tests/unit/web-session-credentials.test.ts +++ b/tests/unit/web-session-credentials.test.ts @@ -104,6 +104,33 @@ test("web session credential metadata identifies cookie, token, and no-auth prov }); }); +test("MaxAI and UC keep independent top-level credential contracts", () => { + assert.deepEqual(webSessionCredentials.getWebSessionCredentialRequirement("maxai"), { + kind: "token", + credentialName: "MaxAI access token (Bearer) + device id", + placeholder: + "Use browser sign-in — OmniRoute mints the MaxAI access token, device id, and user id for you", + acceptsFullCookieHeader: false, + storageKeys: [ + "accessToken", + "access_token", + "maxaiAccessToken", + "deviceId", + "maxaiDeviceId", + "userId", + "maxaiUserId", + ], + }); + + const uc = webSessionCredentials.getWebSessionCredentialRequirement("uc"); + assert.ok(uc && uc.kind === "cookie"); + assert.equal(uc.credentialName, "Clerk __client cookie + session id + user id"); + assert.equal(uc.acceptsFullCookieHeader, true); + assert.ok(uc.storageKeys.includes("__client")); + assert.ok(uc.storageKeys.includes("sid")); + assert.ok(uc.storageKeys.includes("uid")); +}); + test("web session credential validator requires provider-specific non-empty values", () => { assert.equal( webSessionCredentials.hasUsableWebSessionCredential("kimi-web", { token: "kimi-token" }), From a721fc72959f3cc66f49d03457214a8bf3284851 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 20:57:02 -0300 Subject: [PATCH 058/143] fix(adapta): redact streamed upstream errors (#12438) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 14 PRs desta campanha de error-boundary sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **120/120** nos 23 arquivos de teste que os PRs trazem. Um ponto que só apareceu no tree combinado: **#12465 e #12466 criam o mesmo arquivo novo** `open-sse/utils/streamReadiness.ts` (que não existe no tip) com desenhos divergentes de cancelamento — `cancelled` + `releaseLock` imediato num, `readInFlight`/`cancelRequested` com `cancelReader` fire-and-forget no outro. Adotei a versão do #12466, que difere e defere o release do lock para quando a leitura em voo termina, e validei a escolha rodando as suítes dos **dois** PRs contra ela: 21/21 no readiness compartilhado e 22/22 incluindo o boundary do Perplexity. --- open-sse/executors/adapta-web.ts | 6 +- ...dapta-web-stream-error-boundary.fixture.ts | 69 +++++++++++++ .../executor-adapta-web-stream-error.test.ts | 97 +++++++++++++++++++ tests/unit/executor-adapta-web.test.ts | 31 +++--- 4 files changed, 188 insertions(+), 15 deletions(-) create mode 100644 tests/fixtures/adapta-web-stream-error-boundary.fixture.ts create mode 100644 tests/unit/executor-adapta-web-stream-error.test.ts diff --git a/open-sse/executors/adapta-web.ts b/open-sse/executors/adapta-web.ts index da1b836002..83969ed927 100644 --- a/open-sse/executors/adapta-web.ts +++ b/open-sse/executors/adapta-web.ts @@ -5,6 +5,7 @@ import { sanitizeErrorMessage } from "../utils/error.ts"; const ADAPTA_APP_URL = "https://agent.adapta.one"; const ADAPTA_CLERK_URL = "https://clerk.agent.adapta.one"; const ADAPTA_STREAM_URL = `${ADAPTA_APP_URL}/api/chat/stream/v1`; +const ADAPTA_PUBLIC_STREAM_ERROR = `\n\n[Erro: ${sanitizeErrorMessage("Adapta upstream error")}]`; // Default model ID in Adapta's internal system (corresponds to "ONE" / auto-select) const DEFAULT_AI_MODEL_ID = 14; @@ -321,10 +322,9 @@ function transformStream(adaptaStream: ReadableStream, model: string): ReadableS if (event.id === "quick-response") continue; // Real text ended — stream will send more events or close } else if (type === "error") { - const errText = String(event.errorText ?? "Adapta upstream error"); ensureRole(); - // Emit the error as content so the user sees it - chunk({ content: `\n\n[Erro: ${errText}]` }); + // Keep upstream diagnostics private: the transformed SSE is a public HTTP 200 body. + chunk({ content: ADAPTA_PUBLIC_STREAM_ERROR }); finalize(); return; } else if (type === "done" || type === "end") { diff --git a/tests/fixtures/adapta-web-stream-error-boundary.fixture.ts b/tests/fixtures/adapta-web-stream-error-boundary.fixture.ts new file mode 100644 index 0000000000..02c7e6f07e --- /dev/null +++ b/tests/fixtures/adapta-web-stream-error-boundary.fixture.ts @@ -0,0 +1,69 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +assert.ok(process.env.DATA_DIR, "the subprocess fixture requires an isolated DATA_DIR"); +assert.ok( + process.env.OMNIROUTE_PLUGINS_DIR, + "the subprocess fixture requires an isolated OMNIROUTE_PLUGINS_DIR" +); +assert.equal(process.env.HOME, undefined, "the subprocess must not inherit HOME"); +assert.equal(process.env.CODEX_HOME, undefined, "the subprocess must not inherit CODEX_HOME"); + +const { AdaptaWebExecutor } = await import("../../open-sse/executors/adapta-web.ts"); + +const originalFetch = globalThis.fetch; + +test.afterEach(() => { + globalThis.fetch = originalFetch; +}); + +test("terminates an upstream error event without exposing its text in the public SSE", async () => { + const hostileError = + "SQLSTATE 42P01 at /srv/omniroute/private.ts:91 — Authorization: Bearer secret-token"; + const requestedUrls: string[] = []; + const logMessages: string[] = []; + + globalThis.fetch = (async (input: RequestInfo | URL) => { + const url = String(input); + requestedUrls.push(url); + + if (url.endsWith("/v1/client")) { + return Response.json({ + response: { sessions: [{ id: "session-stream-error", status: "active" }] }, + }); + } + + if (url.includes("/tokens")) { + return Response.json({ jwt: "eyJ.test-session.jwt" }); + } + + return new Response(`data: ${JSON.stringify({ type: "error", errorText: hostileError })}\n\n`, { + status: 200, + headers: { "Content-Type": "text/event-stream" }, + }); + }) as typeof fetch; + + const executor = new AdaptaWebExecutor(); + const result = await executor.execute({ + model: "adapta-one", + body: { messages: [{ role: "user", content: "hello" }] }, + stream: true, + credentials: { apiKey: "__client=unique-stream-error-cookie" }, + signal: null, + log: { + info: (_tag, message) => logMessages.push(message), + warn: (_tag, message) => logMessages.push(message), + }, + }); + + assert.equal(result.response.status, 200); + assert.equal(result.response.headers.get("content-type"), "text/event-stream"); + + const publicSse = await result.response.text(); + assert.equal(requestedUrls.length, 3); + assert.match(publicSse, /"content":"\\n\\n\[Erro: Adapta upstream error\]"/); + assert.match(publicSse, /"finish_reason":"stop"/); + assert.match(publicSse, /data: \[DONE\]/); + assert.doesNotMatch(publicSse, /SQLSTATE|\/srv\/omniroute|secret-token/); + assert.doesNotMatch(logMessages.join("\n"), /SQLSTATE|\/srv\/omniroute|secret-token/); +}); diff --git a/tests/unit/executor-adapta-web-stream-error.test.ts b/tests/unit/executor-adapta-web-stream-error.test.ts new file mode 100644 index 0000000000..f17667b92a --- /dev/null +++ b/tests/unit/executor-adapta-web-stream-error.test.ts @@ -0,0 +1,97 @@ +import assert from "node:assert/strict"; +import { spawn } from "node:child_process"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import test from "node:test"; + +const FIXTURE = fileURLToPath( + new URL("../fixtures/adapta-web-stream-error-boundary.fixture.ts", import.meta.url) +); +const REPO_ROOT = fileURLToPath(new URL("../..", import.meta.url)); +const SYNTHETIC_API_KEY_SECRET = + "adapta-stream-boundary-test-secret-00000000000000000000000000000000"; + +type FixtureResult = { + code: number | null; + signal: NodeJS.Signals | null; + stdout: string; + stderr: string; +}; + +function runIsolatedFixture(testRoot: string): Promise { + const dataDir = path.join(testRoot, "data"); + const pluginsDir = path.join(testRoot, "plugins"); + fs.mkdirSync(dataDir, { recursive: true }); + fs.mkdirSync(pluginsDir, { recursive: true }); + + const childEnv: NodeJS.ProcessEnv = { + PATH: process.env.PATH, + NODE_PATH: process.env.NODE_PATH, + LANG: process.env.LANG ?? "C.UTF-8", + LC_ALL: process.env.LC_ALL, + TZ: process.env.TZ ?? "UTC", + TMPDIR: process.env.TMPDIR ?? os.tmpdir(), + NODE_ENV: "test", + APP_LOG_TO_FILE: "false", + API_KEY_SECRET: SYNTHETIC_API_KEY_SECRET, + DATA_DIR: dataDir, + OMNIROUTE_PLUGINS_DIR: pluginsDir, + }; + delete childEnv.NODE_TEST_CONTEXT; + + return new Promise((resolve, reject) => { + const child = spawn(process.execPath, ["--import", "tsx/esm", "--test", FIXTURE], { + cwd: REPO_ROOT, + env: childEnv, + stdio: ["ignore", "pipe", "pipe"], + timeout: 90_000, + }); + let stdout = ""; + let stderr = ""; + child.stdout.setEncoding("utf8").on("data", (chunk: string) => { + stdout += chunk; + }); + child.stderr.setEncoding("utf8").on("data", (chunk: string) => { + stderr += chunk; + }); + child.once("error", reject); + child.once("close", (code, signal) => resolve({ code, signal, stdout, stderr })); + }); +} + +test("Adapta stream errors are sanitized in an isolated executor fixture", async () => { + const testRoot = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-adapta-boundary-parent-")); + const originalDataDir = process.env.DATA_DIR; + const originalPluginsDir = process.env.OMNIROUTE_PLUGINS_DIR; + const originalFetch = globalThis.fetch; + const eventBusOwner = globalThis as { __omnirouteEventBus?: unknown }; + const originalEventBus = eventBusOwner.__omnirouteEventBus; + + try { + const result = await runIsolatedFixture(testRoot); + assert.equal( + result.code, + 0, + `isolated Adapta fixture failed (signal=${result.signal ?? "none"})\n` + + `stdout:\n${result.stdout}\nstderr:\n${result.stderr}` + ); + assert.equal(result.signal, null); + assert.match(result.stdout, /ℹ tests 1/); + assert.match(result.stdout, /ℹ pass 1/); + assert.match(result.stdout, /ℹ fail 0/); + assert.doesNotMatch(result.stdout + result.stderr, new RegExp(SYNTHETIC_API_KEY_SECRET)); + + assert.equal(process.env.DATA_DIR, originalDataDir); + assert.equal(process.env.OMNIROUTE_PLUGINS_DIR, originalPluginsDir); + assert.equal(globalThis.fetch, originalFetch); + assert.equal( + eventBusOwner.__omnirouteEventBus, + originalEventBus, + "the subprocess fixture must not replace the parent event bus singleton" + ); + } finally { + fs.rmSync(testRoot, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + } +}); diff --git a/tests/unit/executor-adapta-web.test.ts b/tests/unit/executor-adapta-web.test.ts index 46105774ee..fc6538f8cf 100644 --- a/tests/unit/executor-adapta-web.test.ts +++ b/tests/unit/executor-adapta-web.test.ts @@ -36,18 +36,25 @@ describe("AdaptaWebExecutor", () => { }); it("execute returns proper result shape on auth failure", async () => { - const executor = new mod.AdaptaWebExecutor(); - const result = await executor.execute({ - model: "adapta-one", - body: { messages: [{ role: "user", content: "hi" }] }, - stream: false, - credentials: { apiKey: "invalid-jwt" }, - signal: null, - }); - assert.ok(result.response instanceof Response); - assert.ok(typeof result.url === "string"); - assert.ok(typeof result.headers === "object"); - assert.ok(result.transformedBody !== undefined); + const originalFetch = globalThis.fetch; + globalThis.fetch = (async () => new Response(null, { status: 401 })) as typeof fetch; + + try { + const executor = new mod.AdaptaWebExecutor(); + const result = await executor.execute({ + model: "adapta-one", + body: { messages: [{ role: "user", content: "hi" }] }, + stream: false, + credentials: { apiKey: "invalid-jwt" }, + signal: null, + }); + assert.ok(result.response instanceof Response); + assert.ok(typeof result.url === "string"); + assert.ok(typeof result.headers === "object"); + assert.ok(result.transformedBody !== undefined); + } finally { + globalThis.fetch = originalFetch; + } }); it("testConnection returns false for invalid credentials", async () => { From 4ef4e25fa7f5cb4ad076a9a5c8e937b5a7eeddf4 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 20:57:21 -0300 Subject: [PATCH 059/143] fix(sse): surface Adapta non-stream SSE errors (#12459) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 14 PRs desta campanha de error-boundary sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **120/120** nos 23 arquivos de teste que os PRs trazem. Um ponto que só apareceu no tree combinado: **#12465 e #12466 criam o mesmo arquivo novo** `open-sse/utils/streamReadiness.ts` (que não existe no tip) com desenhos divergentes de cancelamento — `cancelled` + `releaseLock` imediato num, `readInFlight`/`cancelRequested` com `cancelReader` fire-and-forget no outro. Adotei a versão do #12466, que difere e defere o release do lock para quando a leitura em voo termina, e validei a escolha rodando as suítes dos **dois** PRs contra ela: 21/21 no readiness compartilhado e 22/22 incluindo o boundary do Perplexity. --- .../fixes/adapta-nonstream-sse-error.md | 1 + open-sse/executors/adapta-web.ts | 23 +++- ...ta-web-nonstream-error-boundary.fixture.ts | 129 ++++++++++++++++++ ...dapta-web-nonstream-error-boundary.test.ts | 73 ++++++++++ 4 files changed, 224 insertions(+), 2 deletions(-) create mode 100644 changelog.d/fixes/adapta-nonstream-sse-error.md create mode 100644 tests/fixtures/adapta-web-nonstream-error-boundary.fixture.ts create mode 100644 tests/unit/adapta-web-nonstream-error-boundary.test.ts diff --git a/changelog.d/fixes/adapta-nonstream-sse-error.md b/changelog.d/fixes/adapta-nonstream-sse-error.md new file mode 100644 index 0000000000..fb34658a4b --- /dev/null +++ b/changelog.d/fixes/adapta-nonstream-sse-error.md @@ -0,0 +1 @@ +- **fix(sse):** Treat Adapta Web `type:error` SSE events as sanitized non-stream failures instead of empty HTTP 200 completions. diff --git a/open-sse/executors/adapta-web.ts b/open-sse/executors/adapta-web.ts index 83969ed927..619b952c3c 100644 --- a/open-sse/executors/adapta-web.ts +++ b/open-sse/executors/adapta-web.ts @@ -1,6 +1,6 @@ import { BaseExecutor, type ExecuteInput } from "./base.ts"; import { prepareToolMessages, buildToolAwareResult } from "../translator/webTools.ts"; -import { sanitizeErrorMessage } from "../utils/error.ts"; +import { buildErrorBody, sanitizeErrorMessage } from "../utils/error.ts"; const ADAPTA_APP_URL = "https://agent.adapta.one"; const ADAPTA_CLERK_URL = "https://clerk.agent.adapta.one"; @@ -480,9 +480,10 @@ export class AdaptaWebExecutor extends BaseExecutor { const reader = resp.body!.getReader(); let buf = ""; let fullText = ""; + let upstreamErrorMessage: string | null = null; try { - while (true) { + readLoop: while (true) { const { done, value } = await reader.read(); if (done) break; buf += decoder.decode(value, { stream: true }); @@ -494,6 +495,9 @@ export class AdaptaWebExecutor extends BaseExecutor { const ev = JSON.parse(line.slice(6)); if (ev.type === "text-delta" && ev.id !== "quick-response") { fullText += String(ev.delta ?? ""); + } else if (ev.type === "error") { + upstreamErrorMessage = "Adapta upstream error"; + break readLoop; } } catch { // skip @@ -501,9 +505,24 @@ export class AdaptaWebExecutor extends BaseExecutor { } } } finally { + if (upstreamErrorMessage) { + void reader.cancel("Adapta upstream SSE error").catch(() => undefined); + } reader.releaseLock(); } + if (upstreamErrorMessage) { + return { + response: new Response(JSON.stringify(buildErrorBody(502, upstreamErrorMessage)), { + status: 502, + headers: { "Content-Type": "application/json" }, + }), + url: ADAPTA_STREAM_URL, + headers, + transformedBody: requestPayload, + }; + } + if (hasTools) { const { content, toolCalls, finishReason } = buildToolAwareResult( fullText, diff --git a/tests/fixtures/adapta-web-nonstream-error-boundary.fixture.ts b/tests/fixtures/adapta-web-nonstream-error-boundary.fixture.ts new file mode 100644 index 0000000000..2e58eb87ea --- /dev/null +++ b/tests/fixtures/adapta-web-nonstream-error-boundary.fixture.ts @@ -0,0 +1,129 @@ +// This suite owns process-wide DATA_DIR, plugin, fetch, and DB state. It must run only inside +// the subprocess launched by tests/unit/adapta-web-nonstream-error-boundary.test.ts. +import assert from "node:assert/strict"; +import { mkdirSync, mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { after, afterEach, describe, it } from "node:test"; + +const TEST_DATA_DIR = mkdtempSync(join(tmpdir(), "omniroute-adapta-nonstream-error-")); +const TEST_PLUGINS_DIR = join(TEST_DATA_DIR, "plugins"); +mkdirSync(TEST_PLUGINS_DIR, { recursive: true }); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.OMNIROUTE_PLUGINS_DIR = TEST_PLUGINS_DIR; + +const originalFetch = globalThis.fetch; +const { AdaptaWebExecutor } = await import("../../open-sse/executors/adapta-web.ts"); + +interface ErrorEnvelope { + error?: { + message?: string; + type?: string; + code?: string; + }; + choices?: unknown[]; +} + +interface CompletionEnvelope { + choices?: Array<{ + message?: { + content?: string; + }; + finish_reason?: string; + }>; +} + +function installAdaptaFetch(upstreamBody: string): void { + const mockFetch = async (input: string | URL | Request): Promise => { + const url = + typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url; + + if (url === "https://clerk.agent.adapta.one/v1/client") { + return Response.json({ response: { sessions: [{ id: "sess-fixture", status: "active" }] } }); + } + + if (url === "https://clerk.agent.adapta.one/v1/client/sessions/sess-fixture/tokens") { + return Response.json({ jwt: "eyJ.fixture.signature" }); + } + + if (url === "https://agent.adapta.one/api/chat/stream/v1") { + return new Response(upstreamBody, { + status: 200, + headers: { "Content-Type": "text/event-stream" }, + }); + } + + throw new Error(`Unexpected test fetch URL: ${url}`); + }; + + globalThis.fetch = mockFetch as typeof fetch; +} + +afterEach(() => { + globalThis.fetch = originalFetch; +}); + +after(async () => { + const { resetDbInstance } = await import("../../src/lib/db/core.ts"); + resetDbInstance(); + rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +describe("Adapta Web non-stream error boundary", () => { + it("returns a sanitized 502 when an HTTP 200 SSE body contains type:error", async () => { + installAdaptaFetch( + `data: ${JSON.stringify({ + type: "error", + errorText: + "SQLSTATE 42P01 private detail at /srv/omniroute/open-sse/executors/adapta-web.ts:481:9 Authorization: Bearer secret-token\n at secret (/srv/omniroute/internal.ts:1:1)", + })}\n\n` + ); + + const executor = new AdaptaWebExecutor(); + const result = await executor.execute({ + model: "adapta-one", + body: { messages: [{ role: "user", content: "hello" }] }, + stream: false, + credentials: { apiKey: "fixture-client-error" }, + signal: null, + }); + + assert.equal(result.response.status, 502); + const payload = (await result.response.json()) as ErrorEnvelope; + assert.equal(payload.error?.type, "server_error"); + assert.equal(payload.error?.code, "bad_gateway"); + assert.equal(payload.error?.message, "Adapta upstream error"); + assert.ok(!payload.error?.message?.includes("SQLSTATE")); + assert.ok(!payload.error?.message?.includes("private detail")); + assert.ok(!payload.error?.message?.includes("/srv/omniroute")); + assert.ok(!payload.error?.message?.includes("secret-token")); + assert.ok(!payload.error?.message?.includes("\n")); + assert.equal(payload.choices, undefined); + }); + + it("preserves a normal non-stream completion assembled from text-delta events", async () => { + installAdaptaFetch( + [ + `data: ${JSON.stringify({ type: "text-delta", id: "quick-response", delta: "Loading" })}`, + `data: ${JSON.stringify({ type: "text-delta", id: "answer", delta: "Hello" })}`, + `data: ${JSON.stringify({ type: "text-delta", id: "answer", delta: " world" })}`, + `data: ${JSON.stringify({ type: "done" })}`, + "", + ].join("\n\n") + ); + + const executor = new AdaptaWebExecutor(); + const result = await executor.execute({ + model: "adapta-one", + body: { messages: [{ role: "user", content: "hello" }] }, + stream: false, + credentials: { apiKey: "fixture-client-success" }, + signal: null, + }); + + assert.equal(result.response.status, 200); + const payload = (await result.response.json()) as CompletionEnvelope; + assert.equal(payload.choices?.[0]?.message?.content, "Hello world"); + assert.equal(payload.choices?.[0]?.finish_reason, "stop"); + }); +}); diff --git a/tests/unit/adapta-web-nonstream-error-boundary.test.ts b/tests/unit/adapta-web-nonstream-error-boundary.test.ts new file mode 100644 index 0000000000..ef5c5894f4 --- /dev/null +++ b/tests/unit/adapta-web-nonstream-error-boundary.test.ts @@ -0,0 +1,73 @@ +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import test from "node:test"; + +const REPO_ROOT = fileURLToPath(new URL("../..", import.meta.url)); +const FIXTURE = fileURLToPath( + new URL("../fixtures/adapta-web-nonstream-error-boundary.fixture.ts", import.meta.url) +); + +const CHILD_RUNTIME_ENV_KEYS = [ + "PATH", + "TMPDIR", + "TMP", + "TEMP", + "SystemRoot", + "ComSpec", + "PATHEXT", + "LANG", + "LC_ALL", + "TZ", +] as const; + +function buildFixtureEnv(): NodeJS.ProcessEnv { + const env: NodeJS.ProcessEnv = { + NODE_ENV: "test", + APP_LOG_TO_FILE: "false", + API_KEY_SECRET: "adapta-boundary-fixture-api-key-secret-20260902", + DISABLE_SQLITE_AUTO_BACKUP: "true", + }; + + for (const key of CHILD_RUNTIME_ENV_KEYS) { + const value = process.env[key]; + if (value !== undefined) env[key] = value; + } + + // A nested test runner must receive its own context instead of inheriting the parent's. + delete env.NODE_TEST_CONTEXT; + return env; +} + +test("Adapta Web non-stream error boundaries pass in an isolated process", () => { + const result = spawnSync( + process.execPath, + [ + "--import", + "tsx/esm", + "--import", + "./open-sse/utils/setupPolyfill.ts", + "--test", + "--test-force-exit", + FIXTURE, + ], + { + cwd: REPO_ROOT, + encoding: "utf8", + env: buildFixtureEnv(), + timeout: 60_000, + } + ); + + assert.ifError(result.error); + assert.equal( + result.signal, + null, + `isolated Adapta boundary fixture terminated by ${result.signal}\n${result.stdout}\n${result.stderr}` + ); + assert.equal( + result.status, + 0, + `isolated Adapta boundary fixture failed\n${result.stdout}\n${result.stderr}` + ); +}); From 774e6db3965106304e5bb8f120e1c3b3af5c853c Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 20:57:39 -0300 Subject: [PATCH 060/143] fix(security): redact dashboard failure events (#12469) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 14 PRs desta campanha de error-boundary sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **120/120** nos 23 arquivos de teste que os PRs trazem. Um ponto que só apareceu no tree combinado: **#12465 e #12466 criam o mesmo arquivo novo** `open-sse/utils/streamReadiness.ts` (que não existe no tip) com desenhos divergentes de cancelamento — `cancelled` + `releaseLock` imediato num, `readInFlight`/`cancelRequested` com `cancelReader` fire-and-forget no outro. Adotei a versão do #12466, que difere e defere o release do lock para quando a leitura em voo termina, e validei a escolha rodando as suítes dos **dois** PRs contra ela: 21/21 no readiness compartilhado e 22/22 incluindo o boundary do Perplexity. --- .../dashboard-request-failed-redaction.md | 1 + open-sse/handlers/chatCore/attemptLogging.ts | 5 +- ...ashboard-request-failed-redaction-probe.ts | 125 +++++++++++++++++ ...dashboard-request-failed-redaction.test.ts | 129 ++++++++++++++++++ 4 files changed, 259 insertions(+), 1 deletion(-) create mode 100644 changelog.d/fixes/dashboard-request-failed-redaction.md create mode 100644 tests/fixtures/dashboard-request-failed-redaction-probe.ts create mode 100644 tests/unit/dashboard-request-failed-redaction.test.ts diff --git a/changelog.d/fixes/dashboard-request-failed-redaction.md b/changelog.d/fixes/dashboard-request-failed-redaction.md new file mode 100644 index 0000000000..a9efc53dc7 --- /dev/null +++ b/changelog.d/fixes/dashboard-request-failed-redaction.md @@ -0,0 +1 @@ +- **fix(security):** sanitize `request.failed` diagnostics before publishing them to live dashboard listeners and replay history, while keeping status, model, provider, latency, and internal call-log diagnostics intact. diff --git a/open-sse/handlers/chatCore/attemptLogging.ts b/open-sse/handlers/chatCore/attemptLogging.ts index 07de9f19b2..5ae0876f77 100644 --- a/open-sse/handlers/chatCore/attemptLogging.ts +++ b/open-sse/handlers/chatCore/attemptLogging.ts @@ -19,6 +19,7 @@ import { saveCallLog } from "@/lib/usageDb"; import type { VideoBridgeLogRedactionEntry } from "@/lib/guardrails/videoBridge"; import { FORMATS } from "../../translator/formats.ts"; import { takeEarlyKeepaliveBytes } from "../../utils/earlyKeepaliveByteBuffer.ts"; +import { sanitizeErrorMessage } from "../../utils/error.ts"; import { cloneBoundedChatLogPayload, truncateForLog } from "./logTruncation.ts"; import { attachLogMeta } from "./cacheUsageMeta.ts"; @@ -317,7 +318,9 @@ export function resolveRequestLifecycleEvent(input: { name: "request.failed", payload: { id: traceId, - error: error || `HTTP ${status}`, + // Dashboard listeners and event history cross a public WebSocket boundary. Keep the raw + // diagnostic in the call log/pipeline above, but expose only the canonical safe projection. + error: sanitizeErrorMessage(error || `HTTP ${status}`), statusCode: typeof status === "number" ? status : undefined, latencyMs, model: model || undefined, diff --git a/tests/fixtures/dashboard-request-failed-redaction-probe.ts b/tests/fixtures/dashboard-request-failed-redaction-probe.ts new file mode 100644 index 0000000000..bf740c2220 --- /dev/null +++ b/tests/fixtures/dashboard-request-failed-redaction-probe.ts @@ -0,0 +1,125 @@ +import assert from "node:assert/strict"; + +import type { RequestFailedPayload } from "../../src/lib/events/types.ts"; + +const RESULT_PREFIX = "DASHBOARD_FAILURE_PROBE_RESULT="; + +async function main(): Promise { + assert.ok(process.env.DATA_DIR, "probe requires an isolated DATA_DIR"); + assert.ok(process.env.OMNIROUTE_PLUGINS_DIR, "probe requires an isolated plugins directory"); + assert.ok(process.env.API_KEY_SECRET, "probe requires a synthetic API_KEY_SECRET"); + + const { persistAttemptLogs } = await import("../../open-sse/handlers/chatCore/attemptLogging.ts"); + const eventBus = await import("../../src/lib/events/eventBus.ts"); + const dbCore = await import("../../src/lib/db/core.ts"); + const callLogs = await import("../../src/lib/usage/callLogs.ts"); + + let unsubscribe: (() => void) | undefined; + try { + globalThis.__omnirouteEventBus = undefined; + const hostileError = new Error( + "Provider failed in /srv/omniroute/src/private/provider.ts:42:7 with " + + "api_key='sk-live-dashboard-secret'" + ); + hostileError.stack = + `${hostileError.name}: ${hostileError.message}\n` + + " at dispatch (/srv/omniroute/src/private/transport.ts:91:3)"; + const rawDiagnostic = hostileError.stack; + const traceId = "trace-dashboard-redaction"; + const callLogId = "call-log-dashboard-redaction"; + + const deliveredPromise = new Promise((resolve, reject) => { + const timeout = setTimeout(() => { + unsubscribe?.(); + reject(new Error("timed out waiting for persistAttemptLogs request.failed event")); + }, 10_000); + unsubscribe = eventBus.on("request.failed", (payload) => { + if (payload.id !== traceId) return; + clearTimeout(timeout); + unsubscribe?.(); + unsubscribe = undefined; + resolve(payload); + }); + }); + + persistAttemptLogs( + { + status: 502, + tokens: {}, + responseBody: null, + error: rawDiagnostic, + }, + { + traceId, + provider: "private-provider", + connectionId: null, + model: "private-model", + skillRequestId: "skill-dashboard-redaction", + detailedLoggingEnabled: false, + reqLogger: null, + pendingRequestId: callLogId, + clientRawRequest: { endpoint: "/v1/chat/completions" }, + requestedModel: "private-model", + credentials: null, + startTime: Date.now() - 37, + body: { model: "private-model", messages: [] }, + sourceFormat: "openai", + targetFormat: "openai", + comboName: null, + comboStepId: null, + comboExecutionKey: null, + tokensCompressed: null, + apiKeyInfo: null, + noLogEnabled: false, + correlationId: null, + modelPinned: false, + sessionTag: null, + } + ); + + const delivered = await deliveredPromise; + assert.equal(delivered.id, traceId); + assert.equal(delivered.statusCode, 502); + assert.equal(delivered.model, "private-model"); + assert.equal(delivered.provider, "private-provider"); + assert.ok(delivered.latencyMs >= 0); + assert.equal(delivered.error, "Error: Provider failed in with api_key='[REDACTED]'"); + assert.doesNotMatch(delivered.error, /sk-live-dashboard-secret|\/srv\/omniroute|\n/); + + const replayed = eventBus + .getEventHistory(undefined, 10) + .find( + (entry) => + entry.event === "request.failed" && + (entry.payload as RequestFailedPayload | undefined)?.id === traceId + ); + assert.ok(replayed, "late subscribers must have the safe request.failed history entry"); + assert.deepEqual(replayed.payload, delivered); + + const writerDrained = await callLogs.waitForCallLogSaves(10_000); + assert.equal(writerDrained, true, "call-log write must drain"); + const persisted = await callLogs.getCallLogById(callLogId); + assert.ok(persisted, "failed attempt must still be available to internal diagnostics"); + assert.equal(persisted.error, rawDiagnostic); + + console.log( + RESULT_PREFIX + + JSON.stringify({ + delivered, + replayMatches: JSON.stringify(replayed.payload) === JSON.stringify(delivered), + internalRawPreserved: persisted.error === rawDiagnostic, + writerDrained, + }) + ); + } finally { + unsubscribe?.(); + try { + await callLogs.waitForCallLogSaves(10_000); + await callLogs.closeCallLogSaves(10_000); + } finally { + dbCore.resetDbInstance(); + } + } +} + +await main(); diff --git a/tests/unit/dashboard-request-failed-redaction.test.ts b/tests/unit/dashboard-request-failed-redaction.test.ts new file mode 100644 index 0000000000..e039f7c0dc --- /dev/null +++ b/tests/unit/dashboard-request-failed-redaction.test.ts @@ -0,0 +1,129 @@ +import assert from "node:assert/strict"; +import { execFile } from "node:child_process"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { test } from "node:test"; +import { fileURLToPath } from "node:url"; + +import type { RequestFailedPayload } from "../../src/lib/events/types.ts"; + +const RESULT_PREFIX = "DASHBOARD_FAILURE_PROBE_RESULT="; +const repoRoot = fileURLToPath(new URL("../../", import.meta.url)); +const probePath = fileURLToPath( + new URL("../fixtures/dashboard-request-failed-redaction-probe.ts", import.meta.url) +); + +type ProbeResult = { + delivered: RequestFailedPayload; + replayMatches: boolean; + internalRawPreserved: boolean; + writerDrained: boolean; +}; + +function runProbe(env: NodeJS.ProcessEnv): Promise<{ stdout: string; stderr: string }> { + return new Promise((resolve, reject) => { + execFile( + process.execPath, + ["--import", "tsx/esm", probePath], + { + cwd: repoRoot, + env, + timeout: 30_000, + maxBuffer: 8 * 1024 * 1024, + }, + (error, stdout, stderr) => { + if (error) { + reject( + new Error( + `dashboard failure probe exited unsuccessfully: ${error.message}\n` + + `stdout:\n${stdout}\nstderr:\n${stderr}` + ) + ); + return; + } + resolve({ stdout, stderr }); + } + ); + }); +} + +test("persistAttemptLogs redacts request.failed delivery/replay but keeps its internal log", async () => { + const isolationRoot = fs.mkdtempSync( + path.join(os.tmpdir(), "omniroute-dashboard-failure-redaction-") + ); + const dataDir = path.join(isolationRoot, "data"); + const pluginsDir = path.join(isolationRoot, "plugins"); + fs.mkdirSync(dataDir, { recursive: true }); + fs.mkdirSync(pluginsDir, { recursive: true }); + + try { + // The subprocess receives only process/runtime basics plus synthetic OmniRoute settings: no + // provider credentials are inherited and no parent singleton/env/global state is mutated. + const { stdout, stderr } = await runProbe({ + PATH: process.env.PATH, + NODE_PATH: process.env.NODE_PATH, + LANG: process.env.LANG, + LC_ALL: process.env.LC_ALL, + TZ: process.env.TZ, + TMPDIR: process.env.TMPDIR, + NODE_ENV: "test", + DATA_DIR: dataDir, + OMNIROUTE_PLUGINS_DIR: pluginsDir, + API_KEY_SECRET: "test-dashboard-failure-redaction-secret", + PII_RESPONSE_SANITIZATION: "false", + OMNIROUTE_ENABLE_LIVE_WS: "0", + }); + + assert.doesNotMatch(stderr, /sk-live-dashboard-secret|\/srv\/omniroute/); + const resultLine = stdout.split(/\r?\n/).find((line) => line.startsWith(RESULT_PREFIX)); + assert.ok(resultLine, `probe did not emit its result marker; stdout:\n${stdout}`); + const result = JSON.parse(resultLine.slice(RESULT_PREFIX.length)) as ProbeResult; + + assert.equal(result.delivered.id, "trace-dashboard-redaction"); + assert.equal(result.delivered.statusCode, 502); + assert.equal(result.delivered.model, "private-model"); + assert.equal(result.delivered.provider, "private-provider"); + assert.equal( + result.delivered.error, + "Error: Provider failed in with api_key='[REDACTED]'" + ); + assert.equal(result.replayMatches, true); + assert.equal(result.internalRawPreserved, true); + assert.equal(result.writerDrained, true); + } finally { + // The probe exits only after draining/closing its writer and resetting its DB singleton. + fs.rmSync(isolationRoot, { + recursive: true, + force: true, + maxRetries: 5, + retryDelay: 100, + }); + } +}); + +test("the private LiveWS bridge forwards the already-safe event into its backlog unchanged", () => { + const source = fs.readFileSync( + fileURLToPath(new URL("../../src/server/ws/liveServer.ts", import.meta.url)), + "utf8" + ); + + // publishDashboardEvent/eventHistoryBacklog are module-private. This bounded source-chain + // assertion avoids opening a server while proving the bus payload is what live delivery and + // welcome/backlog replay store. The behavioral safety assertion lives in the subprocess above. + assert.match( + source, + /eventHistoryBacklog\.push\(\{ event, payload, timestamp \}\)/, + "the LiveWS backlog must store the event-bus payload" + ); + assert.match( + source, + /data:\s*h\.payload/, + "welcome replay must forward the stored backlog payload" + ); + assert.match( + source, + /onAny\(\(event:[^\n]+payload:[^\n]+\)\s*=>\s*\{\s*publishDashboardEvent\(event, payload\)/, + "the LiveWS bridge must publish the same event-bus payload" + ); +}); From 406fbd3dcb093b0de56b6d55b6f260864a7ac951 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 20:57:58 -0300 Subject: [PATCH 061/143] fix(huggingchat): sanitize transport failures (#12467) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 14 PRs desta campanha de error-boundary sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **120/120** nos 23 arquivos de teste que os PRs trazem. Um ponto que só apareceu no tree combinado: **#12465 e #12466 criam o mesmo arquivo novo** `open-sse/utils/streamReadiness.ts` (que não existe no tip) com desenhos divergentes de cancelamento — `cancelled` + `releaseLock` imediato num, `readInFlight`/`cancelRequested` com `cancelReader` fire-and-forget no outro. Adotei a versão do #12466, que difere e defere o release do lock para quando a leitura em voo termina, e validei a escolha rodando as suítes dos **dois** PRs contra ela: 21/21 no readiness compartilhado e 22/22 incluindo o boundary do Perplexity. --- ...0-huggingchat-transport-error-redaction.md | 1 + open-sse/executors/huggingchat.ts | 20 +++-- .../huggingchat-transport-error-child.ts | 85 +++++++++++++++++++ ...gingchat-transport-error-redaction.test.ts | 84 ++++++++++++++++++ 4 files changed, 182 insertions(+), 8 deletions(-) create mode 100644 changelog.d/fixes/0000-huggingchat-transport-error-redaction.md create mode 100644 tests/unit/_fixtures/huggingchat-transport-error-child.ts create mode 100644 tests/unit/huggingchat-transport-error-redaction.test.ts diff --git a/changelog.d/fixes/0000-huggingchat-transport-error-redaction.md b/changelog.d/fixes/0000-huggingchat-transport-error-redaction.md new file mode 100644 index 0000000000..3951c9dba8 --- /dev/null +++ b/changelog.d/fixes/0000-huggingchat-transport-error-redaction.md @@ -0,0 +1 @@ +- Sanitize HuggingChat conversation-creation and message-send transport failures before they reach client error bodies or provider logs. diff --git a/open-sse/executors/huggingchat.ts b/open-sse/executors/huggingchat.ts index 7b7557ce02..e75592e5ec 100644 --- a/open-sse/executors/huggingchat.ts +++ b/open-sse/executors/huggingchat.ts @@ -400,13 +400,15 @@ export class HuggingChatExecutor extends BaseExecutor { }; } } catch (err) { - const message = err instanceof Error ? err.message : String(err); + const message = sanitizeErrorMessage(err instanceof Error ? err.message : String(err)); log?.error?.("HUGGINGCHAT", `Conversation creation failed: ${message}`); return { response: new Response( - JSON.stringify({ - error: { message: `HuggingChat connection failed: ${message}`, type: "upstream_error" }, - }), + JSON.stringify( + buildErrorBody(502, `HuggingChat connection failed: ${message}`, undefined, { + type: "upstream_error", + }) + ), { status: 502, headers: { "Content-Type": "application/json" } } ), url: CONVERSATION_URL, @@ -463,13 +465,15 @@ export class HuggingChatExecutor extends BaseExecutor { signal: combinedSignal, }); } catch (err) { - const message = err instanceof Error ? err.message : String(err); + const message = sanitizeErrorMessage(err instanceof Error ? err.message : String(err)); log?.error?.("HUGGINGCHAT", `Message send failed: ${message}`); return { response: new Response( - JSON.stringify({ - error: { message: `HuggingChat connection failed: ${message}`, type: "upstream_error" }, - }), + JSON.stringify( + buildErrorBody(502, `HuggingChat connection failed: ${message}`, undefined, { + type: "upstream_error", + }) + ), { status: 502, headers: { "Content-Type": "application/json" } } ), url: messageUrl, diff --git a/tests/unit/_fixtures/huggingchat-transport-error-child.ts b/tests/unit/_fixtures/huggingchat-transport-error-child.ts new file mode 100644 index 0000000000..2a904451d9 --- /dev/null +++ b/tests/unit/_fixtures/huggingchat-transport-error-child.ts @@ -0,0 +1,85 @@ +import { mkdirSync, mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; + +const CHILD_RESULT_PREFIX = "HUGGINGCHAT_TRANSPORT_RESULT="; +const scenario = process.argv[2]; + +if (scenario !== "conversation-creation" && scenario !== "message-send") { + throw new Error(`Unknown HuggingChat transport scenario: ${String(scenario)}`); +} + +const testRoot = mkdtempSync(join(tmpdir(), "omniroute-huggingchat-transport-child-")); +const testDataDir = join(testRoot, "data"); +const testPluginsDir = join(testRoot, "plugins"); +const testConfigDir = join(testRoot, "config"); + +mkdirSync(testDataDir, { recursive: true }); +mkdirSync(testPluginsDir, { recursive: true }); +mkdirSync(testConfigDir, { recursive: true }); +process.env.DATA_DIR = testDataDir; +process.env.OMNIROUTE_PLUGINS_DIR = testPluginsDir; +process.env.XDG_CONFIG_HOME = testConfigDir; +process.env.APP_LOG_TO_FILE = "false"; +process.env.API_KEY_SECRET = "synthetic-huggingchat-transport-test-key"; + +const hostileTransportMessage = + "TLS request failed at /srv/omniroute/providers/huggingchat/client.ts:44:9 " + + "access_token=transport-secret\n" + + " at sendRequest (/srv/omniroute/runtime/fetch.ts:12:3)"; + +const originalFetch = globalThis.fetch; +let fetchCalls = 0; +const errorLogs: string[] = []; +let childResult: Record | null = null; + +try { + const { HuggingChatExecutor } = await import("../../../open-sse/executors/huggingchat.ts"); + + globalThis.fetch = (async () => { + fetchCalls += 1; + + if (scenario === "conversation-creation") { + throw new Error(hostileTransportMessage); + } + + if (fetchCalls === 1) { + return Response.json({ conversationId: "conversation-test" }); + } + if (fetchCalls === 2) { + return Response.json({ rootMessageId: "root-message-test" }); + } + if (fetchCalls === 3) { + throw new Error(hostileTransportMessage); + } + throw new Error(`Unexpected fetch call ${fetchCalls}`); + }) as typeof globalThis.fetch; + + const result = await new HuggingChatExecutor().execute({ + model: "test/huggingchat-model", + body: { messages: [{ role: "user", content: "hello" }] }, + stream: false, + credentials: { apiKey: "hf-chat=fake-cookie" }, + signal: null, + log: { error: (_tag, message) => errorLogs.push(message) }, + }); + + childResult = { + fetchCalls, + status: result.response.status, + contentType: result.response.headers.get("content-type") || "", + errorLogs, + payload: await result.response.json(), + }; +} finally { + globalThis.fetch = originalFetch; + const coreDb = await import("../../../src/lib/db/core.ts"); + coreDb.resetDbInstance(); + rmSync(testRoot, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +} + +if (!childResult) { + throw new Error(`HuggingChat ${scenario} probe did not produce a result`); +} + +process.stdout.write(`${CHILD_RESULT_PREFIX}${JSON.stringify(childResult)}\n`); diff --git a/tests/unit/huggingchat-transport-error-redaction.test.ts b/tests/unit/huggingchat-transport-error-redaction.test.ts new file mode 100644 index 0000000000..e13253a588 --- /dev/null +++ b/tests/unit/huggingchat-transport-error-redaction.test.ts @@ -0,0 +1,84 @@ +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import { test } from "node:test"; + +const CHILD_RESULT_PREFIX = "HUGGINGCHAT_TRANSPORT_RESULT="; +const childFixture = fileURLToPath( + new URL("./_fixtures/huggingchat-transport-error-child.ts", import.meta.url) +); + +type TransportScenario = "conversation-creation" | "message-send"; + +type TransportFailureResult = { + fetchCalls: number; + status: number; + contentType: string; + errorLogs: string[]; + payload: { + error: { + message: string; + type?: string; + code?: string; + }; + }; +}; + +function runTransportScenario(scenario: TransportScenario): TransportFailureResult { + const result = spawnSync(process.execPath, ["--import", "tsx/esm", childFixture, scenario], { + cwd: process.cwd(), + encoding: "utf8", + timeout: 60_000, + env: { + NODE_ENV: "test", + NO_COLOR: "1", + DISABLE_SQLITE_AUTO_BACKUP: "true", + }, + }); + + assert.equal( + result.status, + 0, + `isolated ${scenario} probe failed: ${String(result.stderr).slice(0, 2_000)}` + ); + + const resultLine = String(result.stdout) + .split("\n") + .findLast((line) => line.startsWith(CHILD_RESULT_PREFIX)); + assert.ok(resultLine, `isolated ${scenario} probe did not emit its result`); + + return JSON.parse(resultLine.slice(CHILD_RESULT_PREFIX.length)) as TransportFailureResult; +} + +function assertPublicFailureIsSanitized(result: TransportFailureResult): void { + assert.equal(result.status, 502); + assert.match(result.contentType, /application\/json/); + assert.equal(result.payload.error.type, "upstream_error"); + assert.match(result.payload.error.message, /^HuggingChat connection failed:/); + assert.match(result.payload.error.message, //); + assert.match(result.payload.error.message, /access_token=\[REDACTED\]/); + + const publicText = JSON.stringify({ payload: result.payload, errorLogs: result.errorLogs }); + assert.doesNotMatch(publicText, /transport-secret/); + assert.doesNotMatch(publicText, /\/srv\/omniroute/); + assert.doesNotMatch(publicText, /sendRequest/); + assert.doesNotMatch(publicText, /\n\s*at /); +} + +test("HuggingChat sanitizes conversation-creation transport failures in body and log", () => { + const result = runTransportScenario("conversation-creation"); + + assert.equal(result.fetchCalls, 1, "the probe must intercept the conversation creation request"); + assert.equal(result.errorLogs.length, 1); + assert.match(result.errorLogs[0], /^Conversation creation failed:/); + assertPublicFailureIsSanitized(result); +}); + +test("HuggingChat sanitizes message-send transport failures in body and log", () => { + const result = runTransportScenario("message-send"); + + assert.equal(result.fetchCalls, 3, "the probe must intercept creation, parent lookup, and send"); + assert.equal(result.errorLogs.length, 1); + assert.match(result.errorLogs[0], /^Message send failed:/); + assertPublicFailureIsSanitized(result); +}); From 855eda16d38de3caeccc03aa1d0400c487e11907 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 20:58:22 -0300 Subject: [PATCH 062/143] fix(zai): treat HTTP 200 stream errors as failures (#12454) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 14 PRs desta campanha de error-boundary sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **120/120** nos 23 arquivos de teste que os PRs trazem. Um ponto que só apareceu no tree combinado: **#12465 e #12466 criam o mesmo arquivo novo** `open-sse/utils/streamReadiness.ts` (que não existe no tip) com desenhos divergentes de cancelamento — `cancelled` + `releaseLock` imediato num, `readInFlight`/`cancelRequested` com `cancelReader` fire-and-forget no outro. Adotei a versão do #12466, que difere e defere o release do lock para quando a leitura em voo termina, e validei a escolha rodando as suítes dos **dois** PRs contra ela: 21/21 no readiness compartilhado e 22/22 incluindo o boundary do Perplexity. --- .../PENDING-zai-web-stream-error-boundary.md | 1 + open-sse/executors/zai-web/stream.ts | 132 +++++++-- .../zai-web-stream-error-boundary.fixture.ts | 253 ++++++++++++++++++ tests/unit/zai-web-silent-empty-repro.test.ts | 67 ++++- .../zai-web-stream-error-boundary.test.ts | 43 +++ 5 files changed, 459 insertions(+), 37 deletions(-) create mode 100644 changelog.d/fixes/PENDING-zai-web-stream-error-boundary.md create mode 100644 tests/fixtures/zai-web-stream-error-boundary.fixture.ts create mode 100644 tests/unit/zai-web-stream-error-boundary.test.ts diff --git a/changelog.d/fixes/PENDING-zai-web-stream-error-boundary.md b/changelog.d/fixes/PENDING-zai-web-stream-error-boundary.md new file mode 100644 index 0000000000..ba4d82521b --- /dev/null +++ b/changelog.d/fixes/PENDING-zai-web-stream-error-boundary.md @@ -0,0 +1 @@ +- **Z.ai Web:** HTTP 200 streams carrying an upstream error now terminate with a structured failure instead of assistant text plus a normal stop, preserving partial output while allowing pre-content combo fallback. diff --git a/open-sse/executors/zai-web/stream.ts b/open-sse/executors/zai-web/stream.ts index 48b99312dd..1c87321bf1 100644 --- a/open-sse/executors/zai-web/stream.ts +++ b/open-sse/executors/zai-web/stream.ts @@ -1,4 +1,4 @@ -import { sanitizeErrorMessage } from "../../utils/error.ts"; +import { buildErrorBody, sanitizeErrorMessage } from "../../utils/error.ts"; export interface ZaiDelta { content: string; @@ -113,22 +113,73 @@ function parseSsePayload(data: string): ZaiDelta | null { } } +type ZaiDeltaSource = { + deltas: AsyncGenerator; + cancel: (reason?: unknown) => void; +}; + +function createZaiDeltaSource(sourceBody: ReadableStream): ZaiDeltaSource { + const decoder = new TextDecoder(); + const reader = sourceBody.getReader(); + const buffer = { text: "" }; + let upstreamDone = false; + let cancelRequested = false; + let readerReleased = false; + + const releaseReader = () => { + if (readerReleased) return; + readerReleased = true; + try { + reader.releaseLock(); + } catch { + // A concurrent read cancellation owns the final release. + } + }; + + const cancel = (reason?: unknown) => { + if (upstreamDone || cancelRequested) return; + cancelRequested = true; + try { + // Do not await an upstream cancel hook: a stalled provider is allowed to + // ignore cancellation, but it must never keep the client cancellation open. + void reader.cancel(reason).catch(() => {}); + } catch { + // The reader may already have closed or released concurrently. + } + }; + + async function* iterate(): AsyncGenerator { + try { + while (true) { + const { done, value } = await reader.read(); + if (done) { + upstreamDone = true; + return; + } + const payloads = extractSseDataPayloads(buffer, decoder.decode(value, { stream: true })); + for (const raw of payloads) { + const delta = parseSsePayload(raw); + if (delta) yield delta; + } + } + } finally { + if (!upstreamDone && !cancelRequested) cancel("Z.ai delta iteration ended"); + releaseReader(); + } + } + + return { deltas: iterate(), cancel }; +} + async function drainSseDeltas( sourceBody: ReadableStream, onDelta: (delta: ZaiDelta) => boolean ): Promise { - const decoder = new TextDecoder(); - const reader = sourceBody.getReader(); - const buffer = { text: "" }; - while (true) { - const { done, value } = await reader.read(); - if (done) return false; - const payloads = extractSseDataPayloads(buffer, decoder.decode(value, { stream: true })); - for (const raw of payloads) { - const delta = parseSsePayload(raw); - if (delta && onDelta(delta)) return true; - } + const { deltas } = createZaiDeltaSource(sourceBody); + for await (const delta of deltas) { + if (onDelta(delta)) return true; } + return false; } function emitDeltaChunks( @@ -137,17 +188,32 @@ function emitDeltaChunks( emitChunk: ZaiChunkEmitter, roleState: { emitted: boolean } ): boolean { - if (!roleState.emitted && (delta.content || delta.reasoning || delta.error)) { + if (delta.error) { + const errorBody = buildErrorBody(502, `Z.ai stream failed: ${delta.error}`, undefined, { + type: "upstream_error", + code: "zai_stream_error", + }); + + if (!roleState.emitted) { + // Keep a pre-content failure as an error-only Chat frame. Stream readiness + // rejects it before response headers are committed, so fallback receives a 502. + controller.enqueue(new TextEncoder().encode(`data: ${JSON.stringify(errorBody)}\n\n`)); + controller.close(); + } else { + // Once content is public, error the protocol-neutral producer. The shared + // pipeline preserves prior chunks, records the failure, and emits the terminal + // error in the client's native Chat, Claude, or Responses wire format. + controller.error(Object.assign(new Error(errorBody.error.message), { statusCode: 502 })); + } + return true; + } + + if (!roleState.emitted && (delta.content || delta.reasoning)) { roleState.emitted = true; emitChunk(controller, { role: "assistant", content: "" }); } if (delta.reasoning) emitChunk(controller, { reasoning_content: delta.reasoning }); if (delta.content) emitChunk(controller, { content: delta.content }); - // Surfaced as visible content, matching the other web executors' mid-stream - // error convention (see zed-hosted's createErrorChunk): the 200 is already on - // the wire, so the status cannot change — but the caller must not be left - // reading an empty success. Any content streamed before the failure is kept. - if (delta.error) emitChunk(controller, { content: `[Z.ai error] ${delta.error}` }); if (delta.done) { emitChunk(controller, {}, "stop"); controller.enqueue(new TextEncoder().encode("data: [DONE]\n\n")); @@ -162,19 +228,32 @@ export function buildZaiStreamingBody( emitChunk: ZaiChunkEmitter, signal: AbortSignal | null | undefined ): ReadableStream { + const deltaSource = createZaiDeltaSource(sourceBody); + const { deltas } = deltaSource; + const roleState = { emitted: false }; + let terminated = false; + return new ReadableStream({ - async start(controller) { - const roleState = { emitted: false }; + async pull(controller) { + if (terminated) return; try { - const ended = await drainSseDeltas(sourceBody, (delta) => - emitDeltaChunks(controller, delta, emitChunk, roleState) - ); - if (ended) return; + const next = await deltas.next(); + if (terminated) return; + if (next.done === false) { + if (emitDeltaChunks(controller, next.value, emitChunk, roleState)) { + terminated = true; + await deltas.return(undefined); + } + return; + } + + terminated = true; if (!roleState.emitted) emitChunk(controller, { role: "assistant", content: "" }); emitChunk(controller, {}, "stop"); controller.enqueue(new TextEncoder().encode("data: [DONE]\n\n")); controller.close(); } catch (error) { + terminated = true; if (!signal?.aborted) { try { controller.error(error); @@ -184,6 +263,11 @@ export function buildZaiStreamingBody( } } }, + cancel(reason) { + terminated = true; + deltaSource.cancel(reason); + void deltas.return(undefined).catch(() => {}); + }, }); } diff --git a/tests/fixtures/zai-web-stream-error-boundary.fixture.ts b/tests/fixtures/zai-web-stream-error-boundary.fixture.ts new file mode 100644 index 0000000000..79afeef0eb --- /dev/null +++ b/tests/fixtures/zai-web-stream-error-boundary.fixture.ts @@ -0,0 +1,253 @@ +import assert from "node:assert/strict"; +import fs from "node:fs"; +import test from "node:test"; + +assert.ok(process.env.DATA_DIR, "the parent wrapper must provide a synthetic DATA_DIR"); +assert.ok( + process.env.OMNIROUTE_PLUGINS_DIR, + "the parent wrapper must provide a synthetic plugin directory" +); +assert.ok(process.env.API_KEY_SECRET, "the parent wrapper must provide a synthetic API secret"); + +fs.mkdirSync(process.env.DATA_DIR, { recursive: true }); +fs.mkdirSync(process.env.OMNIROUTE_PLUGINS_DIR, { recursive: true }); + +const [ + { buildZaiStreamingBody }, + { ensureStreamReadiness }, + { createSSEStream }, + { createStreamController, pipeWithDisconnect }, + { createStreamFailureFinalizers }, + { FORMATS }, + dbCore, + { closeSharedLoggerResource }, +] = await Promise.all([ + import("../../open-sse/executors/zai-web/stream.ts"), + import("../../open-sse/utils/streamReadiness.ts"), + import("../../open-sse/utils/stream.ts"), + import("../../open-sse/utils/streamHandler.ts"), + import("../../open-sse/utils/streamFailureFinalization.ts"), + import("../../open-sse/translator/formats.ts"), + import("../../src/lib/db/core.ts"), + import("../../src/shared/utils/loggerResource.ts"), +]); + +test.after(async () => { + await closeSharedLoggerResource(); + dbCore.resetDbInstance(); +}); + +const encoder = new TextEncoder(); + +function upstreamSse(...payloads: Record[]): ReadableStream { + return new ReadableStream({ + start(controller) { + for (const payload of payloads) { + controller.enqueue(encoder.encode(`data: ${JSON.stringify(payload)}\n\n`)); + } + controller.close(); + }, + }); +} + +function emitOpenAiChunk( + controller: ReadableStreamDefaultController, + delta: Record, + finish: string | null = null +): void { + const chunk = { + id: "chatcmpl-zai-test", + object: "chat.completion.chunk", + created: 1, + model: "glm-5.2", + choices: [{ index: 0, delta, finish_reason: finish }], + }; + controller.enqueue(encoder.encode(`data: ${JSON.stringify(chunk)}\n\n`)); +} + +type PipelineResult = { + output: string; + completions: Array<{ status: number; errorCode?: string | null; error?: string | null }>; + persisted: Array<{ status: number; errorCode?: string }>; + failures: Array<{ status: number; message: string; code?: string; type?: string }>; +}; + +function jsonDataPayloads(output: string): Array> { + return output + .split(/\r?\n/) + .filter((line) => line.startsWith("data: ") && line !== "data: [DONE]") + .map((line) => JSON.parse(line.slice(6)) as Record); +} + +async function runPartialFailurePipeline(clientResponseFormat: string): Promise { + const completions: PipelineResult["completions"] = []; + const persisted: PipelineResult["persisted"] = []; + const failures: PipelineResult["failures"] = []; + const { onPipelineStreamError } = createStreamFailureFinalizers({ + isFailureCompletionRecorded: () => false, + isStreamCompletionRecorded: () => false, + onStreamComplete(payload) { + completions.push({ + status: payload.status, + errorCode: payload.errorCode, + error: payload.error, + }); + }, + persistFailureUsage(status, errorCode) { + persisted.push({ status, errorCode }); + }, + onStreamFailure(failure) { + failures.push(failure); + }, + }); + + const zaiStream = buildZaiStreamingBody( + upstreamSse( + { + type: "chat:completion", + data: { delta_content: "partial answer", phase: "answer" }, + }, + { error: { message: "stream aborted upstream" } } + ), + emitOpenAiChunk, + null + ); + const passthrough = clientResponseFormat === FORMATS.OPENAI; + const transform = createSSEStream({ + mode: passthrough ? "passthrough" : "translate", + targetFormat: FORMATS.OPENAI, + sourceFormat: passthrough ? FORMATS.OPENAI : clientResponseFormat, + clientResponseFormat, + provider: "zai-web", + model: "glm-5.2", + body: { messages: [{ role: "user", content: "hello" }] }, + }); + const streamController = createStreamController({ + provider: "zai-web", + model: "glm-5.2", + clientResponseFormat, + onError: onPipelineStreamError, + }); + const output = await new Response( + pipeWithDisconnect( + new Response(zaiStream, { headers: { "Content-Type": "text/event-stream" } }), + transform, + streamController, + { stallTimeoutMs: 0 } + ) + ).text(); + + return { output, completions, persisted, failures }; +} + +function assertFailureWasPersisted(result: PipelineResult): void { + assert.deepEqual(result.completions, [ + { + status: 502, + errorCode: "stream_pipeline_error", + error: "Z.ai stream failed: stream aborted upstream", + }, + ]); + assert.deepEqual(result.persisted, [{ status: 502, errorCode: "stream_pipeline_error" }]); + assert.deepEqual(result.failures, [ + { + status: 502, + message: "Z.ai stream failed: stream aborted upstream", + code: "stream_pipeline_error", + type: "stream_error", + }, + ]); +} + +test("a pre-content Z.ai error fails stream readiness with a sanitized 502", async () => { + const rawFailure = + 'signature invalid at /srv/omniroute/open-sse/auth.ts:17:9 api_key="sk-private"\n' + + " at verify (/srv/omniroute/open-sse/auth.ts:17:9)"; + const stream = buildZaiStreamingBody( + upstreamSse({ error: { detail: rawFailure } }), + emitOpenAiChunk, + null + ); + + const readiness = await ensureStreamReadiness( + new Response(stream, { headers: { "Content-Type": "text/event-stream" } }), + { timeoutMs: 100, provider: "zai-web", model: "glm-5.2" } + ); + + if (readiness.ok) { + await readiness.response.body?.cancel(); + assert.fail("an error-only Z.ai stream must not be accepted as ready model output"); + } + + assert.equal(readiness.response.status, 502); + assert.equal(readiness.code, "STREAM_EARLY_EOF"); + assert.match(readiness.upstreamDiagnostic ?? "", /Z\.ai stream failed: signature invalid/); + + const publicBody = JSON.stringify(await readiness.response.json()); + assert.doesNotMatch(publicBody, /sk-private|\/srv\/omniroute|auth\.ts/); + assert.match(publicBody, //); +}); + +test("a partial Z.ai failure stays strict Chat and persists as pipeline failure", async () => { + const result = await runPartialFailurePipeline(FORMATS.OPENAI); + const payloads = jsonDataPayloads(result.output); + const terminal = payloads.find((payload) => "error" in payload); + + assert.match(result.output, /partial answer/, "content before the failure is preserved"); + assert.ok(terminal, "the Chat client receives a terminal structured error chunk"); + assert.equal(terminal.object, "chat.completion.chunk"); + assert.deepEqual(Object.keys(terminal).sort(), ["choices", "error", "object"]); + assert.deepEqual(terminal.choices, [{ index: 0, delta: {}, finish_reason: "error" }]); + assert.deepEqual(terminal.error, { + message: "Z.ai stream failed: stream aborted upstream", + type: "server_error", + code: "server_error", + }); + assert.match(result.output, /data: \[DONE\]/); + assert.doesNotMatch(result.output, /response\.failed|event: response\.failed/); + assert.doesNotMatch(result.output, /"finish_reason":"stop"/); + assertFailureWasPersisted(result); +}); + +test("a partial Z.ai failure is translated to Claude and persists as failure", async () => { + const result = await runPartialFailurePipeline(FORMATS.CLAUDE); + + assert.match(result.output, /partial answer/, "translated partial content is preserved"); + assert.match(result.output, /event: error\r?\n/); + assert.match(result.output, /"type":"error"/); + assert.match(result.output, /"message":"Z\.ai stream failed: stream aborted upstream"/); + assert.match(result.output, /event: message_stop\r?\n/); + assert.doesNotMatch(result.output, /response\.failed|event: response\.failed/); + assert.doesNotMatch(result.output, /finish_reason|data: \[DONE\]/); + assertFailureWasPersisted(result); +}); + +test("client cancellation stays non-blocking and cancels a stalled Z.ai body", async () => { + let markPullStarted: (() => void) | null = null; + const pullStarted = new Promise((resolve) => { + markPullStarted = resolve; + }); + let upstreamCancelCalls = 0; + const stalledUpstream = new ReadableStream({ + pull() { + markPullStarted?.(); + return new Promise(() => {}); + }, + cancel() { + upstreamCancelCalls += 1; + return new Promise(() => {}); + }, + }); + const reader = buildZaiStreamingBody(stalledUpstream, emitOpenAiChunk, null).getReader(); + const pendingRead = reader.read(); + void pendingRead.catch(() => {}); + await pullStarted; + + const outcome = await Promise.race([ + reader.cancel("client closed").then(() => "resolved" as const), + new Promise<"timeout">((resolve) => setTimeout(() => resolve("timeout"), 100)), + ]); + + assert.equal(outcome, "resolved", "consumer cancel cannot wait for a stalled upstream body"); + assert.equal(upstreamCancelCalls, 1, "the locked upstream reader receives one cancel request"); +}); diff --git a/tests/unit/zai-web-silent-empty-repro.test.ts b/tests/unit/zai-web-silent-empty-repro.test.ts index 60e2be4d64..01fb742455 100644 --- a/tests/unit/zai-web-silent-empty-repro.test.ts +++ b/tests/unit/zai-web-silent-empty-repro.test.ts @@ -1,9 +1,8 @@ import test from "node:test"; import assert from "node:assert/strict"; -const { buildZaiStreamingBody, parseZaiFrame, collectZaiNonStreaming } = await import( - "../../open-sse/executors/zai-web/stream.ts" -); +const { buildZaiStreamingBody, parseZaiFrame, collectZaiNonStreaming } = + await import("../../open-sse/executors/zai-web/stream.ts"); /** * Hard Rule #6 — "never silently swallow errors in SSE streams". @@ -47,6 +46,23 @@ async function readAll(stream: ReadableStream): Promise { return out; } +async function readUntilError(stream: ReadableStream): Promise<{ output: string; error: unknown }> { + const reader = stream.getReader(); + const decoder = new TextDecoder(); + let output = ""; + try { + for (;;) { + const { done, value } = await reader.read(); + if (done) return { output, error: null }; + output += decoder.decode(value as Uint8Array, { stream: true }); + } + } catch (error) { + return { output, error }; + } finally { + reader.releaseLock(); + } +} + const emitChunk = ( controller: ReadableStreamDefaultController, delta: Record, @@ -61,6 +77,14 @@ const emitChunk = ( const contentOf = (sse: string) => [...sse.matchAll(/"content":"([^"]*)"/g)].map((m) => m[1]).join(""); +function errorPayloads(sse: string): Array> { + return sse + .split(/\r?\n/) + .filter((line) => line.startsWith("data: ") && line !== "data: [DONE]") + .map((line) => JSON.parse(line.slice(6)) as Record) + .filter((payload) => "error" in payload); +} + test("parseZaiFrame classifies an error-shaped frame instead of discarding it", () => { assert.equal(parseZaiFrame({ error: "captcha expired" })?.error, "captcha expired"); assert.equal( @@ -88,25 +112,42 @@ test("REGRESSION GUARD: contentless frames are still skipped, not reported as er assert.equal(parseZaiFrame("not-an-object"), null); }); -test("a 200 stream carrying an error frame surfaces it instead of finishing empty", async () => { +test("a 200 stream carrying an error frame emits a terminal error instead of false success", async () => { const upstream = sseStream(JSON.stringify({ error: { detail: "signature invalid" } })); const out = await readAll(buildZaiStreamingBody(upstream, emitChunk, null)); - assert.match(contentOf(out), /signature invalid/, "the upstream's diagnosis must reach the caller"); - assert.match(contentOf(out), /\[Z\.ai error\]/, "tagged like the other web executors"); - assert.ok(out.includes('"finish_reason":"stop"')); - assert.ok(out.includes("[DONE]"), "the stream still terminates cleanly for the client"); + assert.equal(contentOf(out), "", "an upstream failure must not become assistant content"); + assert.deepEqual(errorPayloads(out), [ + { + error: { + message: "Z.ai stream failed: signature invalid", + type: "upstream_error", + code: "zai_stream_error", + }, + }, + ]); + assert.ok(!out.includes("response.failed"), "Chat streams cannot emit Responses events"); + assert.ok(!out.includes('"finish_reason":"stop"'), "a failure must not report a normal stop"); + assert.ok(!out.includes("[DONE]"), "readiness must see an error-only pre-content stream"); }); -test("an error frame after partial content still surfaces, keeping what was streamed", async () => { +test("an error after partial content preserves it, then errors the producer stream", async () => { const upstream = sseStream( - JSON.stringify({ type: "chat:completion", data: { delta_content: "partial", phase: "answer" } }), + JSON.stringify({ + type: "chat:completion", + data: { delta_content: "partial", phase: "answer" }, + }), JSON.stringify({ error: "stream aborted upstream" }) ); - const out = await readAll(buildZaiStreamingBody(upstream, emitChunk, null)); + const { output, error } = await readUntilError(buildZaiStreamingBody(upstream, emitChunk, null)); - assert.match(contentOf(out), /partial/, "already-streamed content is preserved"); - assert.match(contentOf(out), /stream aborted upstream/, "and the failure is appended, not dropped"); + assert.match(contentOf(output), /partial/, "already-streamed content is preserved"); + assert.match(String(error), /Z\.ai stream failed: stream aborted upstream/); + assert.ok(!output.includes("response.failed"), "the producer stays protocol-neutral"); + assert.ok( + !output.includes('"finish_reason":"stop"'), + "partial output does not make failure success" + ); }); test("control: a well-formed stream is untouched", async () => { diff --git a/tests/unit/zai-web-stream-error-boundary.test.ts b/tests/unit/zai-web-stream-error-boundary.test.ts new file mode 100644 index 0000000000..ee0aa5a52d --- /dev/null +++ b/tests/unit/zai-web-stream-error-boundary.test.ts @@ -0,0 +1,43 @@ +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; +import { fileURLToPath } from "node:url"; + +const fixture = fileURLToPath( + new URL("../fixtures/zai-web-stream-error-boundary.fixture.ts", import.meta.url) +); + +test("Z.ai stream error boundaries pass in a process-isolated fixture", () => { + const testRoot = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-zai-stream-boundary-")); + const childEnv: NodeJS.ProcessEnv = { + API_KEY_SECRET: "zai-stream-boundary-test-only-secret", + DATA_DIR: path.join(testRoot, "data"), + OMNIROUTE_PLUGINS_DIR: path.join(testRoot, "plugins"), + }; + + // The parent itself is a node:test process. Never forward its runner identity to the child; + // `node --test` owns the child context and creates a fresh value for its fixture process. + delete childEnv.NODE_TEST_CONTEXT; + + try { + const result = spawnSync(process.execPath, ["--import", "tsx/esm", "--test", fixture], { + cwd: fileURLToPath(new URL("../..", import.meta.url)), + encoding: "utf8", + env: childEnv, + timeout: 60_000, + }); + const diagnostics = [result.stdout, result.stderr].filter(Boolean).join("\n"); + + assert.equal(result.error, undefined, diagnostics); + assert.equal(result.signal, null, diagnostics); + assert.equal(result.status, 0, diagnostics); + assert.match(result.stdout, /tests 4\b/); + assert.match(result.stdout, /pass 4\b/); + assert.match(result.stdout, /fail 0\b/); + } finally { + fs.rmSync(testRoot, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + } +}); From 9d92d71014017de5878d86c024ed6cee68a2d38d Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 20:58:39 -0300 Subject: [PATCH 063/143] fix(sse): fail Zed streams without false success (#12455) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 14 PRs desta campanha de error-boundary sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **120/120** nos 23 arquivos de teste que os PRs trazem. Um ponto que só apareceu no tree combinado: **#12465 e #12466 criam o mesmo arquivo novo** `open-sse/utils/streamReadiness.ts` (que não existe no tip) com desenhos divergentes de cancelamento — `cancelled` + `releaseLock` imediato num, `readInFlight`/`cancelRequested` com `cancelReader` fire-and-forget no outro. Adotei a versão do #12466, que difere e defere o release do lock para quando a leitura em voo termina, e validei a escolha rodando as suítes dos **dois** PRs contra ela: 21/21 no readiness compartilhado e 22/22 incluindo o boundary do Perplexity. --- .../fixes/zed-hosted-stream-error-boundary.md | 1 + open-sse/executors/zed-hosted.ts | 172 +++++++-- .../zed-hosted-stream-error-boundary-child.ts | 341 ++++++++++++++++++ .../zed-hosted-stream-error-boundary.test.ts | 84 +++++ 4 files changed, 572 insertions(+), 26 deletions(-) create mode 100644 changelog.d/fixes/zed-hosted-stream-error-boundary.md create mode 100644 tests/fixtures/zed-hosted-stream-error-boundary-child.ts create mode 100644 tests/unit/zed-hosted-stream-error-boundary.test.ts diff --git a/changelog.d/fixes/zed-hosted-stream-error-boundary.md b/changelog.d/fixes/zed-hosted-stream-error-boundary.md new file mode 100644 index 0000000000..e15742f092 --- /dev/null +++ b/changelog.d/fixes/zed-hosted-stream-error-boundary.md @@ -0,0 +1 @@ +- **fix(providers):** Zed Hosted streaming failures now trigger fallback before content and end partial streams with a sanitized structured error instead of fake assistant text and a normal-success stop. diff --git a/open-sse/executors/zed-hosted.ts b/open-sse/executors/zed-hosted.ts index ef1ae4ade6..ba66706f10 100644 --- a/open-sse/executors/zed-hosted.ts +++ b/open-sse/executors/zed-hosted.ts @@ -44,6 +44,8 @@ import { zedLlmFetch, type ZedCredentials, } from "../shared/zedAuth.ts"; +import { buildErrorBody } from "../utils/error.ts"; +import { hasUsefulStreamContent } from "../utils/streamReadiness.ts"; import { resolveSuppressThinkClose, THINKING_MARKER_HEADER } from "../utils/thinkCloseMarker.ts"; // Wire values for the `provider` field of POST /completions. These are NOT @@ -122,37 +124,72 @@ function convertProviderEvent( return event; } -function createErrorChunk(model: string, message: string): Record { - return { - id: `chatcmpl-zed-error-${Date.now()}`, - object: "chat.completion.chunk", - created: Math.floor(Date.now() / 1000), - model, - choices: [{ index: 0, delta: { content: `[Zed error] ${message}` }, finish_reason: "stop" }], - }; +const MAX_ZED_FAILURE_MESSAGE_LENGTH = 512; +const MAX_PENDING_ZED_OUTPUT_LENGTH = 64 * 1024; +const ZED_STREAM_FAILURE_PUBLIC_MESSAGE = "Zed upstream stream failed"; + +function boundedFailureText(value: unknown): string | null { + if (typeof value !== "string" && typeof value !== "number") return null; + const text = String(value).trim(); + return text ? text.slice(0, MAX_ZED_FAILURE_MESSAGE_LENGTH) : null; +} + +function extractZedFailureMessage(failed: Record): string { + const nestedError = + failed.error && typeof failed.error === "object" && !Array.isArray(failed.error) + ? (failed.error as Record) + : null; + const candidates = [ + failed.message, + nestedError?.message, + typeof failed.error === "object" ? undefined : failed.error, + failed.code, + nestedError?.code, + ]; + for (const candidate of candidates) { + const text = boundedFailureText(candidate); + if (text) return text; + } + return "request failed"; +} + +function createErrorChunk(message: string): ReturnType { + return buildErrorBody(502, `Zed stream failed: ${message}`, undefined, { + type: "upstream_error", + code: "ZED_STREAM_FAILED", + }); } /** - * The single controller capability these SSE helpers use. They only ever enqueue — - * never `close()`, never read `desiredSize` — so typing them by that one method lets - * the same code serve both stream kinds. The wider + * The controller capabilities these SSE helpers use. Normal frames only enqueue; + * terminal failures also terminate so they do not depend on the upstream socket + * eventually reaching EOF. Narrow controller types keep the helpers honest. The wider * `ReadableStreamDefaultController` annotation rejected every call site, because the * helpers are driven from a TransformStream and `TransformStreamDefaultController` * has no `close()`. */ type SseEnqueueTarget = Pick, "enqueue">; +type SseProcessTarget = Pick, "enqueue" | "terminate">; + +function serializeSseObject(chunk: unknown): string { + if (!chunk) return ""; + let serialized = ""; + const items = Array.isArray(chunk) ? chunk : [chunk]; + for (const item of items) { + if (!item) continue; + serialized += `data: ${JSON.stringify(item)}\n\n`; + } + return serialized; +} function enqueueSseObject( controller: SseEnqueueTarget, encoder: TextEncoder, chunk: unknown ): void { - if (!chunk) return; - const items = Array.isArray(chunk) ? chunk : [chunk]; - for (const item of items) { - if (!item) continue; - controller.enqueue(encoder.encode(`data: ${JSON.stringify(item)}\n\n`)); - } + const serialized = serializeSseObject(chunk); + if (!serialized) return; + controller.enqueue(encoder.encode(serialized)); } type ZedLine = { done?: true; status?: unknown; event?: unknown } | null; @@ -226,16 +263,47 @@ function wrapZedCompletionStream( } let buffer = ""; let done = false; + let providerOutputForwarded = false; + let pendingProviderOutput = ""; + let pendingFailure: (Error & { statusCode: number }) | null = null; + + const forwardProviderOutput = (controller: SseEnqueueTarget, chunk: unknown) => { + const serialized = serializeSseObject(chunk); + if (!serialized) return; + if (providerOutputForwarded) { + controller.enqueue(encoder.encode(serialized)); + return; + } + + // A role/bootstrap-only chunk makes ensureStreamReadiness release the response before any + // model output exists. If the next chunk is status.failed, downstream read-ahead can discard + // the first real content while propagating the error. Hold structural frames until the first + // substantive text/reasoning/tool delta, then release them atomically with that output. + const outputWithBootstrap = pendingProviderOutput + serialized; + if (!hasUsefulStreamContent(outputWithBootstrap)) { + pendingProviderOutput = + outputWithBootstrap.length <= MAX_PENDING_ZED_OUTPUT_LENGTH + ? outputWithBootstrap + : serialized.length <= MAX_PENDING_ZED_OUTPUT_LENGTH + ? serialized + : ""; + return; + } + controller.enqueue(encoder.encode(outputWithBootstrap)); + pendingProviderOutput = ""; + providerOutputForwarded = true; + }; const finish = (controller: SseEnqueueTarget) => { if (done) return; const finalChunk = convertProviderEvent(provider, null, state); - enqueueSseObject(controller, encoder, finalChunk); - controller.enqueue(encoder.encode("data: [DONE]\n\n")); + const finalOutput = `${pendingProviderOutput}${serializeSseObject(finalChunk)}data: [DONE]\n\n`; + pendingProviderOutput = ""; + controller.enqueue(encoder.encode(finalOutput)); done = true; }; - const processLine = (line: string, controller: SseEnqueueTarget) => { + const processLine = (line: string, controller: SseProcessTarget) => { if (done) return; const payload = unwrapZedLine(line); if (!payload) return; @@ -246,17 +314,29 @@ function wrapZedCompletionStream( if (payload.status) { const status = normalizeStatus(payload.status); if (status?.type === "failed" || status?.failed) { - const failed = (status.failed as Record) || status; - const message = String(failed.message || failed.error || failed.code || "request failed"); - enqueueSseObject(controller, encoder, createErrorChunk(model, message)); - finish(controller); + const failed = + status.failed && typeof status.failed === "object" && !Array.isArray(status.failed) + ? (status.failed as Record) + : status; + if (providerOutputForwarded) { + pendingFailure = Object.assign(new Error(ZED_STREAM_FAILURE_PUBLIC_MESSAGE), { + statusCode: 502, + }); + done = true; + controller.terminate(); + return; + } + pendingProviderOutput = ""; + enqueueSseObject(controller, encoder, createErrorChunk(extractZedFailureMessage(failed))); + done = true; + controller.terminate(); } else if (status?.type === "stream_ended" || status === ("stream_ended" as unknown)) { finish(controller); } return; } const converted = convertProviderEvent(provider, payload.event, state); - enqueueSseObject(controller, encoder, converted); + forwardProviderOutput(controller, converted); }; const transformed = response.body.pipeThrough( @@ -281,7 +361,47 @@ function wrapZedCompletionStream( }) ); - return new Response(transformed, { + // `TransformStreamDefaultController.error()` discards already-enqueued output. A failed + // status can share one upstream network chunk with the last content delta, so erroring the + // transform immediately would erase that partial answer. Drain the transformed chunks through + // a backpressure-aware reader first, then reject the next read with the fixed public error. + // The normal chat pipeline turns that rejection into its client-format terminal frame and + // records the 502 through the existing failure finalizers. + const transformedReader = transformed.getReader(); + let guardedStreamCancelled = false; + const cancelTransformedReader = (reason: unknown) => { + if (guardedStreamCancelled) return; + guardedStreamCancelled = true; + // Client cancellation must settle independently of an upstream body whose cancel hook hangs. + // Request cancellation once, but do not await provider cleanup on the client-facing boundary. + void transformedReader.cancel(reason).catch(() => { + console.debug("[ZED] upstream stream cancellation rejected"); + }); + }; + const guardedStream = new ReadableStream({ + async pull(controller) { + try { + const next = await transformedReader.read(); + if (guardedStreamCancelled) return; + if (!next.done) { + controller.enqueue(next.value); + return; + } + if (pendingFailure) { + controller.error(pendingFailure); + return; + } + controller.close(); + } catch (error) { + if (!guardedStreamCancelled) controller.error(error); + } + }, + cancel(reason) { + cancelTransformedReader(reason); + }, + }); + + return new Response(guardedStream, { status: response.status, statusText: response.statusText, headers: { diff --git a/tests/fixtures/zed-hosted-stream-error-boundary-child.ts b/tests/fixtures/zed-hosted-stream-error-boundary-child.ts new file mode 100644 index 0000000000..c5c8f8352d --- /dev/null +++ b/tests/fixtures/zed-hosted-stream-error-boundary-child.ts @@ -0,0 +1,341 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +// This file is executed only by the process-isolated unit-test wrapper. State +// mutations and repository imports must remain here, never in the parent test. +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-zed-stream-data-")); +const TEST_PLUGINS_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-zed-stream-plugins-")); +const originalDataDir = process.env.DATA_DIR; +const originalPluginsDir = process.env.OMNIROUTE_PLUGINS_DIR; +const originalFetch = globalThis.fetch; +let networkCalls = 0; + +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.OMNIROUTE_PLUGINS_DIR = TEST_PLUGINS_DIR; +globalThis.fetch = async () => { + networkCalls += 1; + throw new Error("Unexpected network access in Zed stream boundary test"); +}; + +const core = await import("../../src/lib/db/core.ts"); +const loggerResource = await import("../../src/shared/utils/loggerResource.ts"); +const { __test__ } = await import("../../open-sse/executors/zed-hosted.ts"); +const { FORMATS } = await import("../../open-sse/translator/formats.ts"); +const { assembleStreamingPipeline } = + await import("../../open-sse/handlers/chatCore/streamingPipeline.ts"); +const { createPassthroughStreamWithLogger } = await import("../../open-sse/utils/stream.ts"); +const { createStreamFailureFinalizers } = + await import("../../open-sse/utils/streamFailureFinalization.ts"); +const { createStreamController } = await import("../../open-sse/utils/streamHandler.ts"); +const { ensureStreamReadiness } = await import("../../open-sse/utils/streamReadiness.ts"); +const { wrapZedCompletionStream } = __test__; + +type StreamCompletionEvent = Parameters< + Parameters[0]["onStreamComplete"] +>[0]; + +const RAW_FAILURE = "Bearer TOP_SECRET /srv/omniroute/zed-handler.ts:42 api_key=zed-secret"; +const TEST_MODEL = "grok-test-zed-stream-boundary"; +const TEST_CONNECTION_ID = "zed-stream-boundary-partial-connection"; + +function failedStatusLine(): string { + return JSON.stringify({ status: { failed: { message: RAW_FAILURE } } }); +} + +function nestedFailedStatusLine(): string { + return JSON.stringify({ + status: { + type: "failed", + error: { message: `${RAW_FAILURE} ${"x".repeat(2_000)}` }, + }, + }); +} + +function wrapOpenNdjson(lines: unknown[]): Response { + const encoder = new TextEncoder(); + const body = lines + .map((line) => (typeof line === "string" ? line : JSON.stringify(line))) + .join("\n"); + const stream = new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(`${body}\n`)); + // Keep the upstream open: status.failed must terminate the wrapped stream itself. + }, + cancel() {}, + }); + return wrapZedCompletionStream( + new Response(stream, { + status: 200, + headers: { "Content-Type": "application/x-ndjson" }, + }), + "x_ai", + TEST_MODEL + ); +} + +function wrapOpenFailedNdjson(): Response { + const encoder = new TextEncoder(); + const body = new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(`${failedStatusLine()}\n`)); + // Deliberately stay open: status.failed is terminal by itself and must not + // depend on the upstream socket eventually reaching EOF. + }, + cancel() {}, + }); + return wrapZedCompletionStream( + new Response(body, { + status: 200, + headers: { "Content-Type": "application/x-ndjson" }, + }), + "x_ai", + TEST_MODEL + ); +} + +function wrapStalledNdjson(onCancel: () => void): Response { + const body = new ReadableStream({ + cancel() { + onCancel(); + return new Promise(() => {}); + }, + }); + return wrapZedCompletionStream( + new Response(body, { + status: 200, + headers: { "Content-Type": "application/x-ndjson" }, + }), + "x_ai", + TEST_MODEL + ); +} + +async function resolvesWithin(promise: Promise, timeoutMs: number): Promise { + let timeout: ReturnType | undefined; + try { + await Promise.race([ + promise, + new Promise((_, reject) => { + timeout = setTimeout( + () => reject(new Error(`operation exceeded ${timeoutMs}ms`)), + timeoutMs + ); + }), + ]); + } finally { + if (timeout) clearTimeout(timeout); + } +} + +async function waitFor(predicate: () => boolean, timeoutMs: number): Promise { + const deadline = Date.now() + timeoutMs; + while (!predicate() && Date.now() < deadline) { + await new Promise((resolve) => setTimeout(resolve, 1)); + } + assert.equal(predicate(), true, `condition was not met within ${timeoutMs}ms`); +} + +function parseSsePayloads(text: string): Array> { + return text + .split(/\r?\n/) + .filter((line) => line.startsWith("data: ") && line.slice(6) !== "[DONE]") + .map((line) => JSON.parse(line.slice(6)) as Record); +} + +function assertNoSensitiveFailureText(text: string): void { + assert.doesNotMatch(text, /TOP_SECRET|zed-secret|\/srv\/omniroute\/zed-handler\.ts/); +} + +test.after(async () => { + core.resetDbInstance(); + await loggerResource.closeSharedLoggerResource(); + globalThis.fetch = originalFetch; + if (originalDataDir === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = originalDataDir; + if (originalPluginsDir === undefined) delete process.env.OMNIROUTE_PLUGINS_DIR; + else process.env.OMNIROUTE_PLUGINS_DIR = originalPluginsDir; + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + fs.rmSync(TEST_PLUGINS_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("zed-hosted pre-content status.failed becomes a sanitized 502 readiness failure", async () => { + const readiness = await ensureStreamReadiness(wrapOpenFailedNdjson(), { + timeoutMs: 100, + provider: "zed-hosted", + model: TEST_MODEL, + }); + + assert.equal(readiness.ok, false, "the structured error must remain eligible for fallback"); + if (readiness.ok) assert.fail("pre-content Zed failure must not make the stream ready"); + assert.equal(readiness.response.status, 502); + assert.equal(readiness.code, "STREAM_EARLY_EOF"); + + const bodyText = await readiness.response.text(); + const body = JSON.parse(bodyText) as { + error: { message: string; type: string; code: string }; + upstream_details?: { error?: { message?: string } }; + }; + assert.equal(body.error.type, "stream_early_eof"); + assert.equal(body.error.code, "STREAM_EARLY_EOF"); + assert.match(body.upstream_details?.error?.message ?? "", /Zed stream failed/i); + assertNoSensitiveFailureText(bodyText); + assert.equal(networkCalls, 0); +}); + +test("zed-hosted partial failure reaches stream finalization and persistence as 502", async () => { + const roleChunk = { + event: { + id: "chatcmpl-zed-partial", + object: "chat.completion.chunk", + choices: [{ index: 0, delta: { role: "assistant" }, finish_reason: null }], + }, + }; + const contentChunk = { + event: { + id: "chatcmpl-zed-partial", + object: "chat.completion.chunk", + choices: [{ index: 0, delta: { content: "partial answer" }, finish_reason: null }], + }, + }; + const readiness = await ensureStreamReadiness( + wrapOpenNdjson([ + roleChunk, + contentChunk, + nestedFailedStatusLine(), + { event: { ignored: "after failure" } }, + ]), + { + timeoutMs: 100, + provider: "zed-hosted", + model: TEST_MODEL, + } + ); + + assert.equal(readiness.ok, true, "partial model output must remain deliverable"); + const completionEvents: StreamCompletionEvent[] = []; + const persistedFailures: Array<{ + connectionId: string; + model: string; + status: number; + code?: string; + }> = []; + const streamFailures: Array<{ status: number; message: string; code?: string; type?: string }> = + []; + const pipelineErrors: Array<{ message: string; statusCode: number }> = []; + let streamCompletionRecorded = false; + let failureCompletionRecorded = false; + + const recordCompletion = (payload: StreamCompletionEvent): void => { + if (streamCompletionRecorded) return; + streamCompletionRecorded = true; + if (payload.status !== 200) failureCompletionRecorded = true; + completionEvents.push(payload); + }; + const finalizers = createStreamFailureFinalizers({ + isFailureCompletionRecorded: () => failureCompletionRecorded, + isStreamCompletionRecorded: () => streamCompletionRecorded, + onStreamComplete: recordCompletion, + persistFailureUsage: (status, code) => + persistedFailures.push({ + connectionId: TEST_CONNECTION_ID, + model: TEST_MODEL, + status, + code, + }), + onStreamFailure: (failure) => streamFailures.push(failure), + }); + const streamController = createStreamController({ + onError: (event) => { + pipelineErrors.push({ message: event.message, statusCode: event.statusCode }); + return finalizers.onPipelineStreamError(event); + }, + provider: "zed-hosted", + model: TEST_MODEL, + connectionId: TEST_CONNECTION_ID, + clientResponseFormat: FORMATS.OPENAI, + }); + const transformStream = createPassthroughStreamWithLogger( + "zed-hosted", + null, + null, + TEST_MODEL, + TEST_CONNECTION_ID, + { messages: [{ role: "user", content: "test" }] }, + recordCompletion, + null, + finalizers.handleStreamFailure, + FORMATS.OPENAI + ); + const responseHeaders: Record = {}; + const finalStream = assembleStreamingPipeline({ + providerResponse: readiness.response, + transformStream, + streamController, + createPiiTransform: null, + clientRawRequestHeaders: null, + clientResponseFormat: FORMATS.OPENAI, + echoModel: null, + responseHeaders, + }); + const text = await new Response(finalStream, { headers: responseHeaders }).text(); + const payloads = parseSsePayloads(text); + const errorPayload = payloads.find((payload) => "error" in payload) as + { error: { message: string; type: string; code: string } } | undefined; + + assert.match(text, /partial answer/); + assert.ok(errorPayload, "the stream handler must emit its format-safe terminal error"); + assert.equal(errorPayload.error.type, "server_error"); + assert.equal(errorPayload.error.code, "server_error"); + assert.equal(errorPayload.error.message, "Zed upstream stream failed"); + assert.match(text, /"finish_reason":"error"/); + assert.doesNotMatch(text, /\[Zed error\]|"finish_reason":"stop"|response\.failed/); + assert.doesNotMatch(text, /"ignored":"after failure"/); + assertNoSensitiveFailureText(text); + + assert.equal(completionEvents.length, 1, "the failure must finalize exactly once"); + assert.equal(completionEvents[0].status, 502); + assert.equal(completionEvents[0].error, "Zed upstream stream failed"); + assert.equal(completionEvents[0].errorCode, "stream_pipeline_error"); + assert.deepEqual(persistedFailures, [ + { + connectionId: TEST_CONNECTION_ID, + model: TEST_MODEL, + status: 502, + code: "stream_pipeline_error", + }, + ]); + assert.deepEqual(streamFailures, [ + { + status: 502, + message: "Zed upstream stream failed", + code: "stream_pipeline_error", + type: "stream_error", + }, + ]); + assert.deepEqual(pipelineErrors, [{ message: "Zed upstream stream failed", statusCode: 502 }]); + assertNoSensitiveFailureText(JSON.stringify(completionEvents)); + assert.equal(networkCalls, 0); +}); + +test("zed-hosted client cancellation does not await a stalled upstream cancel hook", async () => { + let upstreamCancelCalls = 0; + const response = wrapStalledNdjson(() => { + upstreamCancelCalls += 1; + }); + assert.ok(response.body); + + const reader = response.body.getReader(); + const pendingRead = reader.read(); + await resolvesWithin(reader.cancel("client disconnected"), 100); + const readResult = await pendingRead; + assert.equal(readResult.done, true); + await waitFor(() => upstreamCancelCalls === 1, 100); + + await resolvesWithin(reader.cancel("duplicate cancel"), 100); + await new Promise((resolve) => setTimeout(resolve, 10)); + assert.equal(upstreamCancelCalls, 1, "the upstream cancel hook must be requested exactly once"); + assert.equal(networkCalls, 0); +}); diff --git a/tests/unit/zed-hosted-stream-error-boundary.test.ts b/tests/unit/zed-hosted-stream-error-boundary.test.ts new file mode 100644 index 0000000000..1565c1c1bd --- /dev/null +++ b/tests/unit/zed-hosted-stream-error-boundary.test.ts @@ -0,0 +1,84 @@ +import assert from "node:assert/strict"; +import { spawn } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import test from "node:test"; + +const repoRoot = fileURLToPath(new URL("../../", import.meta.url)); +const fixturePath = fileURLToPath( + new URL("../fixtures/zed-hosted-stream-error-boundary-child.ts", import.meta.url) +); + +type FixtureResult = { + code: number | null; + signal: NodeJS.Signals | null; + stdout: string; + stderr: string; +}; + +function runFixture(): Promise { + // Keep the parent process pristine: the fast unit suite can run files with + // --test-isolation=none, so all stateful imports and mutations live in the child. + const childEnv: NodeJS.ProcessEnv = { + PATH: process.env.PATH, + NODE_PATH: process.env.NODE_PATH, + LANG: process.env.LANG, + LC_ALL: process.env.LC_ALL, + TZ: process.env.TZ, + TMPDIR: process.env.TMPDIR, + NODE_ENV: "test", + API_KEY_SECRET: "zed-boundary-test-only-secret-with-32-plus-characters", + DISABLE_SQLITE_AUTO_BACKUP: "true", + NO_COLOR: "1", + }; + // Inheriting this marker makes Node silently skip the nested --test run. + delete childEnv.NODE_TEST_CONTEXT; + + return new Promise((resolve, reject) => { + const child = spawn(process.execPath, ["--import", "tsx/esm", "--test", fixturePath], { + cwd: repoRoot, + env: childEnv, + stdio: ["ignore", "pipe", "pipe"], + }); + let stdout = ""; + let stderr = ""; + let timedOut = false; + + child.stdout.setEncoding("utf8"); + child.stderr.setEncoding("utf8"); + child.stdout.on("data", (chunk: string) => { + stdout += chunk; + }); + child.stderr.on("data", (chunk: string) => { + stderr += chunk; + }); + + const timeout = setTimeout(() => { + timedOut = true; + child.kill("SIGKILL"); + }, 120_000); + + child.once("error", (error) => { + clearTimeout(timeout); + reject(error); + }); + child.once("close", (code, signal) => { + clearTimeout(timeout); + if (timedOut) { + reject(new Error("Zed stream error boundary fixture timed out after 120 seconds")); + return; + } + resolve({ code, signal, stdout, stderr }); + }); + }); +} + +test("Zed stream error boundary passes in a process-isolated runtime", async () => { + const result = await runFixture(); + const output = `${result.stdout}\n${result.stderr}`; + + assert.equal(result.signal, null, output.slice(-12_000)); + assert.equal(result.code, 0, output.slice(-12_000)); + assert.match(output, /(?:^|\s)tests\s+3(?:\s|$)/m); + assert.match(output, /(?:^|\s)pass\s+3(?:\s|$)/m); + assert.match(output, /(?:^|\s)fail\s+0(?:\s|$)/m); +}); From 9469fa5f594593d0320974248a7939e4b8aebf86 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 20:58:56 -0300 Subject: [PATCH 064/143] fix(streaming): sanitize generic stream failure boundaries (#12457) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 14 PRs desta campanha de error-boundary sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **120/120** nos 23 arquivos de teste que os PRs trazem. Um ponto que só apareceu no tree combinado: **#12465 e #12466 criam o mesmo arquivo novo** `open-sse/utils/streamReadiness.ts` (que não existe no tip) com desenhos divergentes de cancelamento — `cancelled` + `releaseLock` imediato num, `readInFlight`/`cancelRequested` com `cancelReader` fire-and-forget no outro. Adotei a versão do #12466, que difere e defere o release do lock para quando a leitura em voo termina, e validei a escolha rodando as suítes dos **dois** PRs contra ela: 21/21 no readiness compartilhado e 22/22 incluindo o boundary do Perplexity. --- CHANGELOG.md | 4 + open-sse/utils/streamHandler.ts | 14 +- ...m-handler-public-error-boundary.fixture.ts | 211 ++++++++++++++++++ ...ream-handler-public-error-boundary.test.ts | 62 +++++ tests/unit/stream-handler.test.ts | 17 +- 5 files changed, 296 insertions(+), 12 deletions(-) create mode 100644 tests/fixtures/stream-handler-public-error-boundary.fixture.ts create mode 100644 tests/unit/stream-handler-public-error-boundary.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index bfcad514f4..80abce64dc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -98,6 +98,10 @@ _Living section — cycle opened at the v3.8.50 freeze (parallel-cycle model). B ### 🐛 Bug Fixes +- **security(streaming):** sanitize generic mid-stream error messages before emitting OpenAI, + Responses, or Claude SSE failure frames and before diagnostic logging, while preserving raw + failures for internal classification and keeping client disconnects out of provider failure state. + ### 📝 Maintenance --- diff --git a/open-sse/utils/streamHandler.ts b/open-sse/utils/streamHandler.ts index 7776f2e5e9..841346bcfe 100644 --- a/open-sse/utils/streamHandler.ts +++ b/open-sse/utils/streamHandler.ts @@ -1,6 +1,7 @@ import { trackPendingRequest } from "@/lib/usageDb"; import { STREAM_IDLE_TIMEOUT_MS } from "../config/constants.ts"; import { FORMATS } from "../translator/formats.ts"; +import { buildErrorBody } from "./error.ts"; import { PENDING_REQUEST_CLEARED_MARKER } from "./stream.ts"; import { createCompletedResponsesToolHandoffWatcher } from "./responsesToolHandoff.ts"; import { createStreamContentWatcher, type StreamContentWatcher } from "./streamReadiness.ts"; @@ -187,6 +188,10 @@ function getErrorStatusCode(error: unknown): number { return 502; } +function getPublicErrorMessage(errorMsg: string, statusCode: number): string { + return buildErrorBody(statusCode, errorMsg).error.message; +} + function isDeadlineAbortReason(reason: unknown): reason is Error { return ( reason instanceof Error && @@ -406,7 +411,7 @@ export function createStreamController({ } if (error instanceof Error) { - logStream(`error: ${error.message}`); + logStream(`error: ${getPublicErrorMessage(error.message, getErrorStatusCode(error))}`); return; } logStream("error: unknown"); @@ -452,6 +457,7 @@ export function buildStreamErrorChunks( clientResponseFormat?: string | null ) { const statusMapping = getStreamErrorStatusMapping(statusCode); + const publicErrorMessage = getPublicErrorMessage(errorMsg, statusCode); if (isResponsesClientFormat(clientResponseFormat)) { const errorEvent = { @@ -460,7 +466,7 @@ export function buildStreamErrorChunks( id: null, status: "failed", error: { - message: errorMsg, + message: publicErrorMessage, type: statusMapping.responses.type, code: statusMapping.responses.code, }, @@ -475,7 +481,7 @@ export function buildStreamErrorChunks( type: "error", error: { type: statusMapping.claude.type, - message: errorMsg, + message: publicErrorMessage, }, }; @@ -498,7 +504,7 @@ export function buildStreamErrorChunks( }, ], error: { - message: errorMsg, + message: publicErrorMessage, type: statusMapping.responses.type, code: statusMapping.responses.code, }, diff --git a/tests/fixtures/stream-handler-public-error-boundary.fixture.ts b/tests/fixtures/stream-handler-public-error-boundary.fixture.ts new file mode 100644 index 0000000000..de48e9bdf4 --- /dev/null +++ b/tests/fixtures/stream-handler-public-error-boundary.fixture.ts @@ -0,0 +1,211 @@ +// This suite owns process-wide DATA_DIR, plugin, logger, and DB state. It must run only inside +// the subprocess launched by tests/unit/stream-handler-public-error-boundary.test.ts. +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; + +const originalDataDir = process.env.DATA_DIR; +const originalPluginsDir = process.env.OMNIROUTE_PLUGINS_DIR; +const testRoot = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-stream-public-error-")); +const TEST_DATA_DIR = path.join(testRoot, "data"); +const TEST_PLUGINS_DIR = path.join(testRoot, "plugins"); +fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +fs.mkdirSync(TEST_PLUGINS_DIR, { recursive: true }); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.OMNIROUTE_PLUGINS_DIR = TEST_PLUGINS_DIR; + +const [core, callLogs, artifactWriter, loggerResource, streamHandler, { FORMATS }] = + await Promise.all([ + import("../../src/lib/db/core.ts"), + import("../../src/lib/usage/callLogs.ts"), + import("../../src/lib/usage/callLogArtifactWriter.ts"), + import("../../src/shared/utils/loggerResource.ts"), + import("../../open-sse/utils/streamHandler.ts"), + import("../../open-sse/translator/formats.ts"), + ]); +const { createStreamController, pipeWithDisconnect } = streamHandler; + +const SECRET = "sk-live-streamhandler-secret-123456"; +const API_KEY = "provider-key-streamhandler-654321"; +const PRIVATE_PATH = "/srv/omniroute/private/provider.ts:42:9"; +const RAW_MESSAGE = + `Upstream failed at ${PRIVATE_PATH} Authorization: Bearer ${SECRET} api_key=${API_KEY}` + + `\n at dispatch (/srv/omniroute/private/dispatcher.ts:88:3)`; + +test.after(async () => { + assert.equal(await callLogs.waitForCallLogSaves(3_000), true); + await artifactWriter.closeCallLogArtifactWriter(); + core.resetDbInstance(); + await loggerResource.closeSharedLoggerResource(); + + if (originalDataDir === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = originalDataDir; + if (originalPluginsDir === undefined) delete process.env.OMNIROUTE_PLUGINS_DIR; + else process.env.OMNIROUTE_PLUGINS_DIR = originalPluginsDir; + + fs.rmSync(testRoot, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("fixture binds all persistent state to its process-owned directories", () => { + assert.equal(core.DATA_DIR, TEST_DATA_DIR); + assert.equal(core.SQLITE_FILE, path.join(TEST_DATA_DIR, "storage.sqlite")); + assert.equal(process.env.DATA_DIR, TEST_DATA_DIR); + assert.equal(process.env.OMNIROUTE_PLUGINS_DIR, TEST_PLUGINS_DIR); + assert.equal(fs.existsSync(TEST_DATA_DIR), true); + assert.equal(fs.existsSync(TEST_PLUGINS_DIR), true); +}); + +test("OpenAI stream failures keep raw diagnostics internal and sanitize the public wire", async () => { + const upstreamError = Object.assign(new Error(RAW_MESSAGE), { statusCode: 502 }); + const source = new ReadableStream({ + start(controller) { + controller.error(upstreamError); + }, + }); + let internalMessage = ""; + + const stream = pipeWithDisconnect( + new Response(source), + new TransformStream(), + createStreamController({ + clientResponseFormat: FORMATS.OPENAI, + onError(event) { + internalMessage = event.message; + return true; + }, + }), + { stallTimeoutMs: 0 } + ); + const publicWire = await new Response(stream).text(); + + assert.equal(internalMessage, RAW_MESSAGE, "failure classification must retain the raw message"); + assert.match(publicWire, /"finish_reason":"error"/); + assert.match(publicWire, /"code":"server_error"/); + assert.match(publicWire, /\[DONE\]/); + assert.doesNotMatch(publicWire, new RegExp(SECRET)); + assert.doesNotMatch(publicWire, new RegExp(API_KEY)); + assert.doesNotMatch(publicWire, /\/srv\/omniroute\/private/); + assert.doesNotMatch(publicWire, /dispatcher\.ts/); + assert.match(publicWire, /Authorization: \[REDACTED\]/); + assert.match(publicWire, //); +}); + +test("Responses stream failures preserve the failure event shape without leaking diagnostics", async () => { + const upstreamError = Object.assign(new Error(RAW_MESSAGE), { statusCode: 429 }); + const source = new ReadableStream({ + start(controller) { + controller.error(upstreamError); + }, + }); + let internalError: unknown; + + const stream = pipeWithDisconnect( + new Response(source), + new TransformStream(), + createStreamController({ + clientResponseFormat: FORMATS.OPENAI_RESPONSES, + onError(event) { + internalError = event.error; + return true; + }, + }), + { stallTimeoutMs: 0 } + ); + const publicWire = await new Response(stream).text(); + + assert.equal(internalError, upstreamError, "the original error object must reach classification"); + assert.match(publicWire, /event: response\.failed/); + assert.match(publicWire, /"type":"response\.failed"/); + assert.match(publicWire, /"type":"rate_limit_error"/); + assert.match(publicWire, /"code":"rate_limit_exceeded"/); + assert.doesNotMatch(publicWire, new RegExp(SECRET)); + assert.doesNotMatch(publicWire, new RegExp(API_KEY)); + assert.doesNotMatch(publicWire, /\/srv\/omniroute\/private/); + assert.doesNotMatch(publicWire, /dispatcher\.ts/); + assert.match(publicWire, /Authorization: \[REDACTED\]/); + assert.match(publicWire, //); +}); + +test("Claude stream failures preserve error and stop events without leaking diagnostics", async () => { + const upstreamError = Object.assign(new Error(RAW_MESSAGE), { statusCode: 403 }); + const source = new ReadableStream({ + start(controller) { + controller.error(upstreamError); + }, + }); + let internalStatusCode = 0; + + const stream = pipeWithDisconnect( + new Response(source), + new TransformStream(), + createStreamController({ + clientResponseFormat: FORMATS.CLAUDE, + onError(event) { + internalStatusCode = event.statusCode; + return true; + }, + }), + { stallTimeoutMs: 0 } + ); + const publicWire = await new Response(stream).text(); + + assert.equal(internalStatusCode, 403); + assert.match(publicWire, /event: error/); + assert.match(publicWire, /"type":"permission_error"/); + assert.match(publicWire, /event: message_stop/); + assert.doesNotMatch(publicWire, new RegExp(SECRET)); + assert.doesNotMatch(publicWire, new RegExp(API_KEY)); + assert.doesNotMatch(publicWire, /\/srv\/omniroute\/private/); + assert.doesNotMatch(publicWire, /dispatcher\.ts/); + assert.match(publicWire, /Authorization: \[REDACTED\]/); + assert.match(publicWire, //); +}); + +test("stream diagnostics sanitize logs while callbacks retain the original failure", () => { + const upstreamError = Object.assign(new Error(RAW_MESSAGE), { statusCode: 502 }); + const originalLog = console.log; + const logLines: string[] = []; + let internalError: unknown; + console.log = (...args: unknown[]) => { + logLines.push(args.map(String).join(" ")); + }; + + try { + createStreamController({ + provider: "test-provider", + model: "test-model", + onError(event) { + internalError = event.error; + return true; + }, + }).handleError(upstreamError); + } finally { + console.log = originalLog; + } + + const logs = logLines.join("\n"); + assert.equal(internalError, upstreamError); + assert.match(logs, /error: Upstream failed at /); + assert.match(logs, /Authorization: \[REDACTED\]/); + assert.doesNotMatch(logs, new RegExp(SECRET)); + assert.doesNotMatch(logs, new RegExp(API_KEY)); + assert.doesNotMatch(logs, /\/srv\/omniroute\/private/); + assert.doesNotMatch(logs, /dispatcher\.ts/); +}); + +test("client disconnects stay outside the provider-failure callback", () => { + let providerFailureRecorded = false; + const controller = createStreamController({ + onError() { + providerFailureRecorded = true; + return true; + }, + }); + + controller.handleError(new DOMException("request_signal_aborted", "AbortError")); + + assert.equal(providerFailureRecorded, false); + assert.equal(controller.signal.aborted, false); +}); diff --git a/tests/unit/stream-handler-public-error-boundary.test.ts b/tests/unit/stream-handler-public-error-boundary.test.ts new file mode 100644 index 0000000000..41aa796b19 --- /dev/null +++ b/tests/unit/stream-handler-public-error-boundary.test.ts @@ -0,0 +1,62 @@ +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import test from "node:test"; + +const REPO_ROOT = fileURLToPath(new URL("../..", import.meta.url)); +const FIXTURE = fileURLToPath( + new URL("../fixtures/stream-handler-public-error-boundary.fixture.ts", import.meta.url) +); + +const CHILD_RUNTIME_ENV_KEYS = [ + "PATH", + "TMPDIR", + "TMP", + "TEMP", + "SystemRoot", + "ComSpec", + "PATHEXT", + "LANG", + "LC_ALL", + "TZ", +] as const; + +function buildFixtureEnv(): NodeJS.ProcessEnv { + const env: NodeJS.ProcessEnv = { + NODE_ENV: "test", + APP_LOG_TO_FILE: "false", + API_KEY_SECRET: "stream-handler-boundary-fixture-secret-20260902", + DISABLE_SQLITE_AUTO_BACKUP: "true", + NO_COLOR: "1", + }; + + for (const key of CHILD_RUNTIME_ENV_KEYS) { + const value = process.env[key]; + if (value !== undefined) env[key] = value; + } + + // Nested test runners must not inherit the parent runner's recursion marker. + delete env.NODE_TEST_CONTEXT; + return env; +} + +test("generic stream public error boundaries pass in an isolated process", () => { + const result = spawnSync( + process.execPath, + ["--import", "tsx/esm", "--import", "./open-sse/utils/setupPolyfill.ts", "--test", FIXTURE], + { + cwd: REPO_ROOT, + encoding: "utf8", + env: buildFixtureEnv(), + timeout: 120_000, + } + ); + const output = `${result.stdout}\n${result.stderr}`; + + assert.ifError(result.error); + assert.equal(result.signal, null, output.slice(-12_000)); + assert.equal(result.status, 0, output.slice(-12_000)); + assert.match(output, /(?:^|\s)tests\s+6(?:\s|$)/m); + assert.match(output, /(?:^|\s)pass\s+6(?:\s|$)/m); + assert.match(output, /(?:^|\s)fail\s+0(?:\s|$)/m); +}); diff --git a/tests/unit/stream-handler.test.ts b/tests/unit/stream-handler.test.ts index 8c19c08802..c358ff7c2a 100644 --- a/tests/unit/stream-handler.test.ts +++ b/tests/unit/stream-handler.test.ts @@ -256,7 +256,8 @@ test("createDisconnectAwareStream emits Responses API failure events for Respons assert.match(text, /event: response\.failed/); assert.match(text, /"type":"response\.failed"/); - assert.match(text, /"message":"responses stream\\ndied"/); + assert.match(text, /"message":"responses stream"/); + assert.doesNotMatch(text, /died/); assert.match(text, /"type":"server_error"/); assert.match(text, /"code":"server_error"/); assert.doesNotMatch(text, /chat\.completion\.chunk/); @@ -264,7 +265,7 @@ test("createDisconnectAwareStream emits Responses API failure events for Respons assert.doesNotMatch(text, /\[DONE\]/); }); -test("createDisconnectAwareStream keeps newlines escaped inside SSE data fields", async () => { +test("createDisconnectAwareStream strips multiline diagnostic tails from Responses errors", async () => { const upstreamError = Object.assign(new Error("line one\nline two\rline three"), { statusCode: 400, }); @@ -290,9 +291,9 @@ test("createDisconnectAwareStream keeps newlines escaped inside SSE data fields" const text = await readStreamText(stream); assert.match(text, /^event: response\.failed\ndata: \{"type":"response\.failed"/); - assert.match(text, /"message":"line one\\nline two\\rline three"/); - assert.doesNotMatch(text, /^line two/m); - assert.doesNotMatch(text, /^line three/m); + assert.match(text, /"message":"line one"/); + assert.doesNotMatch(text, /line two/); + assert.doesNotMatch(text, /line three/); }); test("createDisconnectAwareStream treats legacy OpenAI response format alias as Responses", async () => { @@ -360,7 +361,7 @@ test("createDisconnectAwareStream emits Claude SSE errors for Claude clients", a assert.doesNotMatch(text, /\[DONE\]/); }); -test("createDisconnectAwareStream keeps newlines escaped for Claude SSE errors", async () => { +test("createDisconnectAwareStream strips multiline diagnostic tails from Claude errors", async () => { const upstreamError = Object.assign(new Error("claude line one\nclaude line two"), { statusCode: 502, }); @@ -386,8 +387,8 @@ test("createDisconnectAwareStream keeps newlines escaped for Claude SSE errors", const text = await readStreamText(stream); assert.match(text, /^event: error\ndata: \{"type":"error"/); - assert.match(text, /"message":"claude line one\\nclaude line two"/); - assert.doesNotMatch(text, /^claude line two/m); + assert.match(text, /"message":"claude line one"/); + assert.doesNotMatch(text, /claude line two/); }); // #7699/#7816 — heuristic is scoped to FORMATS.CLAUDE (/v1/messages); a From 2f6fdf16c75f0a2a0a7beb5bb30617aef2f97d90 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 20:59:14 -0300 Subject: [PATCH 065/143] fix(codex): close response failure boundary (#12444) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 14 PRs desta campanha de error-boundary sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **120/120** nos 23 arquivos de teste que os PRs trazem. Um ponto que só apareceu no tree combinado: **#12465 e #12466 criam o mesmo arquivo novo** `open-sse/utils/streamReadiness.ts` (que não existe no tip) com desenhos divergentes de cancelamento — `cancelled` + `releaseLock` imediato num, `readInFlight`/`cancelRequested` com `cancelReader` fire-and-forget no outro. Adotei a versão do #12466, que difere e defere o release do lock para quando a leitura em voo termina, e validei a escolha rodando as suítes dos **dois** PRs contra ela: 21/21 no readiness compartilhado e 22/22 incluindo o boundary do Perplexity. --- open-sse/executors/codex.ts | 10 +- open-sse/utils/codexPublicError.ts | 110 +++++ open-sse/vendor/codex-chatgpt-web/bridge.ts | 21 +- .../codex-response-failed-boundary.fixture.ts | 376 ++++++++++++++++++ .../codex-response-failed-boundary.test.ts | 73 ++++ 5 files changed, 582 insertions(+), 8 deletions(-) create mode 100644 open-sse/utils/codexPublicError.ts create mode 100644 tests/fixtures/codex-response-failed-boundary.fixture.ts create mode 100644 tests/unit/codex-response-failed-boundary.test.ts diff --git a/open-sse/executors/codex.ts b/open-sse/executors/codex.ts index 4701f175c2..1194fe4dee 100644 --- a/open-sse/executors/codex.ts +++ b/open-sse/executors/codex.ts @@ -38,6 +38,7 @@ import { applyReasoningInputPolicy } from "../services/reasoningInputPolicy.ts"; import { normalizeCodexVerbosity } from "../services/codexVerbosity.ts"; import { getThinkingBudgetConfig, ThinkingMode } from "../services/thinkingBudget.ts"; import { CORS_HEADERS } from "../utils/cors.ts"; +import { projectCodexPublicError } from "../utils/codexPublicError.ts"; import { errorResponse } from "../utils/error.ts"; import { normalizeCodexResponsesInput } from "../utils/responsesInputNormalization.ts"; import * as prl from "../utils/providerRequestLogging.ts"; @@ -493,7 +494,6 @@ function toCodexResponseFailedEvent(parsed: Record): Record = { code, message }; const explicitStatus = toStatusCode(parsed.status_code) ?? toStatusCode(parsed.status) ?? @@ -503,8 +503,10 @@ function toCodexResponseFailedEvent(parsed: Record): Record = { + ...projectCodexPublicError({ status: statusCode, code, type }), + }; - if (type) error.type = type; if (statusCode !== null) error.status_code = statusCode; return { @@ -955,7 +957,7 @@ export class CodexExecutor extends BaseExecutor { } }; - const failController = (code: string, message: string) => { + const failController = (code: string, _message: string) => { if (closed) return; const controller = streamController; const payload = JSON.stringify({ @@ -963,7 +965,7 @@ export class CodexExecutor extends BaseExecutor { response: { id: null, status: "failed", - error: { code, message }, + error: projectCodexPublicError({ status: 502, code, type: "provider_error" }), }, }); try { diff --git a/open-sse/utils/codexPublicError.ts b/open-sse/utils/codexPublicError.ts new file mode 100644 index 0000000000..6a8c92209c --- /dev/null +++ b/open-sse/utils/codexPublicError.ts @@ -0,0 +1,110 @@ +import { sanitizeErrorMessage } from "./error.ts"; + +export const CODEX_PUBLIC_ERROR_MESSAGE = sanitizeErrorMessage("Codex provider request failed"); + +export interface CodexPublicError { + message: string; + type: string; + code: string; +} + +interface CodexPublicErrorInput { + status?: number | null; + type?: unknown; + code?: unknown; +} + +interface CodexPublicErrorRule { + type: string; + allowsStatus: (status: number) => boolean; +} + +const exactStatuses = + (...statuses: number[]) => + (status: number): boolean => + statuses.includes(status); + +const CODEX_PUBLIC_ERROR_RULES = new Map([ + ["browser_stream_inconsistent", { type: "server_error", allowsStatus: exactStatuses(502) }], + ["chatgpt_session_expired", { type: "authentication_error", allowsStatus: exactStatuses(401) }], + ["chatgpt_submission_ambiguous", { type: "server_error", allowsStatus: exactStatuses(502) }], + ["chatgpt_submitted_turn_failed", { type: "server_error", allowsStatus: exactStatuses(502) }], + ["chatgpt_subscription_unavailable", { type: "server_error", allowsStatus: exactStatuses(503) }], + ["client_cancelled", { type: "invalid_request_error", allowsStatus: exactStatuses(499) }], + ["client_closed_request", { type: "invalid_request_error", allowsStatus: exactStatuses(499) }], + ["codex_app_server_turn_failed", { type: "provider_error", allowsStatus: exactStatuses(502) }], + [ + "compaction_control_unavailable", + { type: "invalid_request_error", allowsStatus: exactStatuses(409) }, + ], + [ + "compaction_handoff_failed", + { type: "invalid_request_error", allowsStatus: exactStatuses(409) }, + ], + [ + "compaction_source_unavailable", + { type: "invalid_request_error", allowsStatus: exactStatuses(409) }, + ], + ["connector_not_found", { type: "connector_error", allowsStatus: exactStatuses(424) }], + [ + "context_length_exceeded", + { type: "invalid_request_error", allowsStatus: exactStatuses(400, 413) }, + ], + ["insufficient_quota", { type: "insufficient_quota", allowsStatus: exactStatuses(429) }], + ["invalid_api_key", { type: "authentication_error", allowsStatus: exactStatuses(401) }], + ["invalid_output_schema", { type: "invalid_request_error", allowsStatus: exactStatuses(400) }], + ["invalid_request_error", { type: "invalid_request_error", allowsStatus: exactStatuses(400) }], + ["multipart_protocol_violation", { type: "server_error", allowsStatus: exactStatuses(502) }], + ["origin_rejected", { type: "invalid_request_error", allowsStatus: exactStatuses(403) }], + ["permission_denied", { type: "permission_error", allowsStatus: exactStatuses(403) }], + ["prompt_attachment_integrity", { type: "server_error", allowsStatus: exactStatuses(502) }], + ["rate_limit_exceeded", { type: "rate_limit_error", allowsStatus: exactStatuses(429) }], + ["server_is_overloaded", { type: "server_error", allowsStatus: exactStatuses(503) }], + [ + "structured_output_validation_failed", + { type: "server_error", allowsStatus: exactStatuses(502) }, + ], + ["subscription_required", { type: "permission_error", allowsStatus: exactStatuses(403) }], + [ + "upstream_server_error", + { + type: "server_error", + allowsStatus: (status) => status >= 500 && status <= 599 && status !== 503, + }, + ], + [ + "upstream_websocket_connect_failed", + { type: "provider_error", allowsStatus: exactStatuses(502) }, + ], + ["upstream_websocket_error", { type: "provider_error", allowsStatus: exactStatuses(502) }], + ["usage_limit_reached", { type: "rate_limit_error", allowsStatus: exactStatuses(429) }], +]); + +function defaultPublicClassification(status: number): Pick { + if (status === 429) return { type: "rate_limit_error", code: "rate_limit_exceeded" }; + if (status === 401) return { type: "authentication_error", code: "invalid_api_key" }; + if (status === 403) return { type: "permission_error", code: "permission_denied" }; + if (status === 499) return { type: "invalid_request_error", code: "client_closed_request" }; + if (status === 503) return { type: "server_error", code: "server_is_overloaded" }; + if (status >= 500) return { type: "server_error", code: "upstream_server_error" }; + return { type: "invalid_request_error", code: "invalid_request_error" }; +} + +/** + * Project an internally classified Codex failure onto its public Responses contract. + * + * Upstream message, code, and type fields are untrusted. The public message is fixed, + * while code/type retain only closed, protocol-level identifiers already produced by + * OmniRoute. Everything else falls back to the HTTP status classification. + */ +export function projectCodexPublicError(input: CodexPublicErrorInput): CodexPublicError { + const status = + typeof input.status === "number" && Number.isInteger(input.status) ? input.status : 502; + const fallback = defaultPublicClassification(status); + const rule = + typeof input.code === "string" ? CODEX_PUBLIC_ERROR_RULES.get(input.code) : undefined; + if (!rule || !rule.allowsStatus(status)) { + return { message: CODEX_PUBLIC_ERROR_MESSAGE, ...fallback }; + } + return { message: CODEX_PUBLIC_ERROR_MESSAGE, type: rule.type, code: input.code as string }; +} diff --git a/open-sse/vendor/codex-chatgpt-web/bridge.ts b/open-sse/vendor/codex-chatgpt-web/bridge.ts index 38eacd167c..a6f6244969 100644 --- a/open-sse/vendor/codex-chatgpt-web/bridge.ts +++ b/open-sse/vendor/codex-chatgpt-web/bridge.ts @@ -5,6 +5,7 @@ import type { CodexProviderContinuationState, CodexUsage, } from "./types"; +import { projectCodexPublicError } from "../../utils/codexPublicError"; import { adapterFailureFromMessage, classifyError, type CodexErrorPayload } from "./lib/errors"; import { encodeCompactionSummary } from "./responses/compaction"; import { encodeReasoningEnvelope, type ReasoningEnvelope } from "./responses/reasoning-envelope"; @@ -46,7 +47,8 @@ function responsesUsage(usage: CodexUsage | undefined): Record } function responseError(status: number, type: string, message: string): CodexErrorPayload { - return classifyError(status, type, message); + const classified = classifyError(status, type, message); + return projectCodexPublicError({ status, type: classified.type, code: classified.code }); } function adapterFailureFromEvent(event: Extract): { @@ -54,14 +56,25 @@ function adapterFailureFromEvent(event: Extract error: CodexErrorPayload; } { if (event.status === undefined && event.errorType === undefined && event.code === undefined) { - return adapterFailureFromMessage(event.message); + const fallback = adapterFailureFromMessage(event.message); + return { + httpStatus: fallback.httpStatus, + error: projectCodexPublicError({ + status: fallback.httpStatus, + type: fallback.error.type, + code: fallback.error.code, + }), + }; } const fallback = adapterFailureFromMessage(event.message); const httpStatus = event.status ?? fallback.httpStatus; const error = classifyError(httpStatus, event.errorType ?? fallback.error.type, event.message); if (event.errorType !== undefined) error.type = event.errorType; if (event.code !== undefined) error.code = event.code; - return { httpStatus, error }; + return { + httpStatus, + error: projectCodexPublicError({ status: httpStatus, type: error.type, code: error.code }), + }; } export { adapterFailureFromMessage } from "./lib/errors"; @@ -1314,7 +1327,7 @@ export function buildResponseJSON( } export function formatErrorResponse(status: number, type: string, message: string): Response { - return new Response(JSON.stringify({ error: classifyError(status, type, message) }), { + return new Response(JSON.stringify({ error: responseError(status, type, message) }), { status, headers: { "Content-Type": "application/json" }, }); diff --git a/tests/fixtures/codex-response-failed-boundary.fixture.ts b/tests/fixtures/codex-response-failed-boundary.fixture.ts new file mode 100644 index 0000000000..eba763b40a --- /dev/null +++ b/tests/fixtures/codex-response-failed-boundary.fixture.ts @@ -0,0 +1,376 @@ +// This suite intentionally owns process-wide DATA_DIR, plugin, and DB state. It must run only +// inside the subprocess launched by tests/unit/codex-response-failed-boundary.test.ts. +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; + +import type { CodexWreqWebSocket } from "../../open-sse/executors/codex/appServerClient.ts"; +import type { AdapterEvent } from "../../open-sse/vendor/codex-chatgpt-web/types.ts"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-codex-boundary-data-")); +const TEST_PLUGINS_DIR = fs.mkdtempSync( + path.join(os.tmpdir(), "omniroute-codex-boundary-plugins-") +); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.OMNIROUTE_PLUGINS_DIR = TEST_PLUGINS_DIR; +process.env.APP_LOG_TO_FILE = "false"; + +const { CodexExecutor, __setCodexWebSocketTransportForTesting, encodeResponseSseEvent } = + await import("../../open-sse/executors/codex.ts"); +const { CodexAppServerExecutor } = await import("../../open-sse/executors/codex-app-server.ts"); +const { bridgeToResponsesSSE, buildResponseJSON } = + await import("../../open-sse/vendor/codex-chatgpt-web/bridge.ts"); +const { resetDbInstance } = await import("../../src/lib/db/core.ts"); + +const PUBLIC_MESSAGE = "Codex provider request failed"; +const HOSTILE_MESSAGE = + "token=codex-secret-value at /srv/omniroute/private/config.json\nforged-log: admin=true"; + +type FailedPayload = { + type: "response.failed"; + response: { + error: { + code: string | null; + message: string; + status_code?: number; + type?: string; + }; + }; +}; + +function responseFailedPayload(sse: string): FailedPayload { + for (const line of sse.split("\n")) { + if (!line.startsWith("data: ") || line === "data: [DONE]") continue; + const parsed = JSON.parse(line.slice("data: ".length)) as Record; + if (parsed.type === "response.failed") return parsed as FailedPayload; + } + assert.fail(`response.failed frame missing from: ${sse}`); +} + +function assertPublicFailure( + payload: FailedPayload, + expected: { code: string; type: string; statusCode?: number } +): void { + assert.equal(payload.response.error.message, PUBLIC_MESSAGE); + assert.equal(payload.response.error.code, expected.code); + assert.equal(payload.response.error.type, expected.type); + if (expected.statusCode !== undefined) { + assert.equal(payload.response.error.status_code, expected.statusCode); + } + assert.ok(!JSON.stringify(payload).includes(HOSTILE_MESSAGE)); + assert.ok(!JSON.stringify(payload).includes("codex-secret-value")); + assert.ok(!JSON.stringify(payload).includes("/srv/omniroute/private")); +} + +async function executeCodexWebSocketFailure( + websocket: Parameters[0] +): Promise { + __setCodexWebSocketTransportForTesting(websocket); + try { + const result = await new CodexExecutor().execute({ + model: "gpt-5.5", + body: { model: "gpt-5.5", input: "hello" }, + stream: true, + credentials: { + accessToken: "test-token", + providerSpecificData: { codexTransport: "websocket" }, + }, + }); + return await result.response.text(); + } finally { + __setCodexWebSocketTransportForTesting(undefined); + } +} + +async function executeAppServerFailure(stream: boolean): Promise { + const socket: CodexWreqWebSocket = { + send(data: string) { + const frame = JSON.parse(data) as Record; + if (frame.id == null || typeof frame.method !== "string") return; + queueMicrotask(() => { + if (frame.method === "thread/start") { + socket.onmessage?.({ + data: JSON.stringify({ + jsonrpc: "2.0", + id: frame.id, + result: { thread: { id: "thread-public-boundary" } }, + }), + }); + return; + } + if (frame.method === "turn/start") { + socket.onmessage?.({ + data: JSON.stringify({ + jsonrpc: "2.0", + id: frame.id, + result: { turn: { id: "turn-public-boundary", status: "inProgress" } }, + }), + }); + setTimeout(() => { + socket.onmessage?.({ + data: JSON.stringify({ + jsonrpc: "2.0", + method: "error", + params: { error: { message: HOSTILE_MESSAGE } }, + }), + }); + }, 0); + return; + } + socket.onmessage?.({ + data: JSON.stringify({ jsonrpc: "2.0", id: frame.id, result: {} }), + }); + }); + }, + close() {}, + onmessage: null, + onerror: null, + onclose: null, + }; + const executor = new CodexAppServerExecutor({ websocketFn: async () => socket }); + const result = await executor.execute({ + model: "gpt-5.5", + body: { input: "hello" }, + stream, + credentials: { + providerSpecificData: { + codexTransport: "app-server", + codexAppServerUrl: "ws://codex-app-server.test:1456", + codexAppServerToken: "test-app-server-token", + }, + }, + }); + return result.response; +} + +test.after(() => { + __setCodexWebSocketTransportForTesting(undefined); + resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + fs.rmSync(TEST_PLUGINS_DIR, { recursive: true, force: true }); +}); + +test("Codex same-format error event emits only a fixed public failure contract", () => { + const result = encodeResponseSseEvent( + JSON.stringify({ + type: "error", + status_code: 502, + error: { + code: "secret_backend_code_9182", + type: "secret_backend_type_7731", + message: HOSTILE_MESSAGE, + }, + }) + ); + + assert.equal(result.terminal, true); + assertPublicFailure(responseFailedPayload(result.sse), { + code: "upstream_server_error", + type: "server_error", + statusCode: 502, + }); +}); + +test("Codex same-format quota classification survives while its raw message does not", () => { + const result = encodeResponseSseEvent( + JSON.stringify({ + type: "response.failed", + response: { + status: "failed", + error: { code: "usage_limit_reached", message: HOSTILE_MESSAGE }, + }, + }) + ); + + assertPublicFailure(responseFailedPayload(result.sse), { + code: "usage_limit_reached", + type: "rate_limit_error", + statusCode: 429, + }); +}); + +test("Codex same-format failures reject contradictory allowlisted status, code and type", () => { + const wrongStatus = responseFailedPayload( + encodeResponseSseEvent( + JSON.stringify({ + type: "response.failed", + status_code: 502, + response: { + status: "failed", + error: { + code: "invalid_api_key", + type: "rate_limit_error", + message: HOSTILE_MESSAGE, + }, + }, + }) + ).sse + ); + assertPublicFailure(wrongStatus, { + code: "upstream_server_error", + type: "server_error", + statusCode: 502, + }); + + const wrongType = responseFailedPayload( + encodeResponseSseEvent( + JSON.stringify({ + type: "response.failed", + status_code: 401, + response: { + status: "failed", + error: { + code: "invalid_api_key", + type: "rate_limit_error", + message: HOSTILE_MESSAGE, + }, + }, + }) + ).sse + ); + assertPublicFailure(wrongType, { + code: "invalid_api_key", + type: "authentication_error", + statusCode: 401, + }); +}); + +test("Codex WebSocket in-flight error event cannot expose transport details", async () => { + const socket = { + send() { + queueMicrotask(() => socket.onerror?.({ message: HOSTILE_MESSAGE })); + }, + close() {}, + onmessage: null as ((event: { data: unknown }) => void) | null, + onerror: null as ((event: { message?: string }) => void) | null, + onclose: null as (() => void) | null, + }; + const sse = await executeCodexWebSocketFailure(async () => socket); + + assertPublicFailure(responseFailedPayload(sse), { + code: "upstream_websocket_error", + type: "provider_error", + }); +}); + +test("Codex WebSocket connection failure cannot expose exception details", async () => { + const sse = await executeCodexWebSocketFailure(async () => { + throw new Error(HOSTILE_MESSAGE); + }); + + assertPublicFailure(responseFailedPayload(sse), { + code: "upstream_websocket_connect_failed", + type: "provider_error", + }); +}); + +test("Codex App Server streaming failure is projected before the HTTP 200 SSE boundary", async () => { + const response = await executeAppServerFailure(true); + assert.equal(response.status, 200); + + assertPublicFailure(responseFailedPayload(await response.text()), { + code: "codex_app_server_turn_failed", + type: "provider_error", + }); +}); + +test("Codex App Server non-streaming failure is projected before the HTTP 200 JSON boundary", async () => { + const response = await executeAppServerFailure(false); + assert.equal(response.status, 200); + const body = (await response.json()) as FailedPayload["response"] & { status: string }; + + assert.equal(body.status, "failed"); + assertPublicFailure( + { type: "response.failed", response: body }, + { + code: "codex_app_server_turn_failed", + type: "provider_error", + } + ); +}); + +test("ChatGPT Web Playwright adapter failures keep safe routing metadata without raw text", async () => { + async function* browserEvents(): AsyncGenerator { + yield { + type: "error", + message: HOSTILE_MESSAGE, + status: 502, + errorType: "server_error", + code: "chatgpt_submission_ambiguous", + retryable: false, + }; + } + + const sse = await new Response(bridgeToResponsesSSE(browserEvents(), "gpt-5.5")).text(); + assertPublicFailure(responseFailedPayload(sse), { + code: "chatgpt_submission_ambiguous", + type: "server_error", + }); +}); + +test("Codex bridge projects message-only adapter failures before SSE serialization", async () => { + async function* messageOnlyEvents(): AsyncGenerator { + yield { type: "error", message: HOSTILE_MESSAGE }; + } + + const sse = await new Response(bridgeToResponsesSSE(messageOnlyEvents(), "gpt-5.5")).text(); + assertPublicFailure(responseFailedPayload(sse), { + code: "upstream_server_error", + type: "server_error", + }); +}); + +test("Codex batch bridge projects message-only adapter failures before JSON serialization", () => { + const body = buildResponseJSON( + [{ type: "error", message: HOSTILE_MESSAGE }], + "gpt-5.5" + ) as FailedPayload["response"] & { status: string }; + + assert.equal(body.status, "failed"); + assertPublicFailure( + { type: "response.failed", response: body }, + { + code: "upstream_server_error", + type: "server_error", + } + ); +}); + +test("Codex bridge exceptions cannot serialize raw exception messages", async () => { + async function* throwingEvents(): AsyncGenerator { + throw new Error(HOSTILE_MESSAGE); + } + + const sse = await new Response(bridgeToResponsesSSE(throwingEvents(), "gpt-5.5")).text(); + assertPublicFailure(responseFailedPayload(sse), { + code: "upstream_server_error", + type: "server_error", + }); +}); + +test("Codex batch bridge applies the same public failure projector", () => { + const body = buildResponseJSON( + [ + { + type: "error", + message: HOSTILE_MESSAGE, + status: 502, + errorType: "server_error", + code: "chatgpt_submitted_turn_failed", + retryable: false, + }, + ], + "gpt-5.5" + ) as FailedPayload["response"] & { status: string }; + + assert.equal(body.status, "failed"); + assertPublicFailure( + { type: "response.failed", response: body }, + { + code: "chatgpt_submitted_turn_failed", + type: "server_error", + } + ); +}); diff --git a/tests/unit/codex-response-failed-boundary.test.ts b/tests/unit/codex-response-failed-boundary.test.ts new file mode 100644 index 0000000000..51f8ffe59e --- /dev/null +++ b/tests/unit/codex-response-failed-boundary.test.ts @@ -0,0 +1,73 @@ +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import test from "node:test"; + +const REPO_ROOT = fileURLToPath(new URL("../..", import.meta.url)); +const FIXTURE = fileURLToPath( + new URL("../fixtures/codex-response-failed-boundary.fixture.ts", import.meta.url) +); + +const CHILD_RUNTIME_ENV_KEYS = [ + "PATH", + "TMPDIR", + "TMP", + "TEMP", + "SystemRoot", + "ComSpec", + "PATHEXT", + "LANG", + "LC_ALL", + "TZ", +] as const; + +function buildFixtureEnv(): NodeJS.ProcessEnv { + const env: NodeJS.ProcessEnv = { + NODE_ENV: "test", + APP_LOG_TO_FILE: "false", + API_KEY_SECRET: "codex-boundary-fixture-api-key-secret-20260902", + DISABLE_SQLITE_AUTO_BACKUP: "true", + }; + + for (const key of CHILD_RUNTIME_ENV_KEYS) { + const value = process.env[key]; + if (value !== undefined) env[key] = value; + } + + // A nested test runner must receive its own context instead of inheriting the parent's. + delete env.NODE_TEST_CONTEXT; + return env; +} + +test("Codex public failure boundaries pass in an isolated process", () => { + const result = spawnSync( + process.execPath, + [ + "--import", + "tsx/esm", + "--import", + "./open-sse/utils/setupPolyfill.ts", + "--test", + "--test-force-exit", + FIXTURE, + ], + { + cwd: REPO_ROOT, + encoding: "utf8", + env: buildFixtureEnv(), + timeout: 60_000, + } + ); + + assert.ifError(result.error); + assert.equal( + result.signal, + null, + `isolated Codex boundary fixture terminated by ${result.signal}\n${result.stdout}\n${result.stderr}` + ); + assert.equal( + result.status, + 0, + `isolated Codex boundary fixture failed\n${result.stdout}\n${result.stderr}` + ); +}); From d63d25f96c9d642a5f5f425d03d765d09f20ccb7 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 20:59:31 -0300 Subject: [PATCH 066/143] fix(huggingchat): surface HTTP 200 JSONL failures (#12456) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 14 PRs desta campanha de error-boundary sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **120/120** nos 23 arquivos de teste que os PRs trazem. Um ponto que só apareceu no tree combinado: **#12465 e #12466 criam o mesmo arquivo novo** `open-sse/utils/streamReadiness.ts` (que não existe no tip) com desenhos divergentes de cancelamento — `cancelled` + `releaseLock` imediato num, `readInFlight`/`cancelRequested` com `cancelReader` fire-and-forget no outro. Adotei a versão do #12466, que difere e defere o release do lock para quando a leitura em voo termina, e validei a escolha rodando as suítes dos **dois** PRs contra ela: 21/21 no readiness compartilhado e 22/22 incluindo o boundary do Perplexity. --- ...nding-huggingchat-stream-error-boundary.md | 1 + open-sse/executors/huggingchat.ts | 108 +++- open-sse/executors/huggingchat/jsonlStream.ts | 63 +- ...ggingchat-stream-error-boundary.fixture.ts | 592 ++++++++++++++++++ .../huggingchat-stream-error-boundary.test.ts | 85 +++ 5 files changed, 822 insertions(+), 27 deletions(-) create mode 100644 changelog.d/fixes/pending-huggingchat-stream-error-boundary.md create mode 100644 tests/unit/_fixtures/huggingchat-stream-error-boundary.fixture.ts create mode 100644 tests/unit/huggingchat-stream-error-boundary.test.ts diff --git a/changelog.d/fixes/pending-huggingchat-stream-error-boundary.md b/changelog.d/fixes/pending-huggingchat-stream-error-boundary.md new file mode 100644 index 0000000000..7285276d10 --- /dev/null +++ b/changelog.d/fixes/pending-huggingchat-stream-error-boundary.md @@ -0,0 +1 @@ +- HuggingChat now turns HTTP 200 JSONL generation failures into a sanitized 502 before content, or a fixed public stream failure after partial output, so fallback and request persistence no longer record a false successful stop. diff --git a/open-sse/executors/huggingchat.ts b/open-sse/executors/huggingchat.ts index e75592e5ec..30ec7da0ea 100644 --- a/open-sse/executors/huggingchat.ts +++ b/open-sse/executors/huggingchat.ts @@ -27,7 +27,11 @@ import { import { FETCH_TIMEOUT_MS } from "../config/constants.ts"; import { buildErrorBody, sanitizeErrorMessage } from "../utils/error.ts"; import { normalizeSessionCookieHeader } from "@/lib/providers/webCookieAuth"; -import { streamJsonlToOpenAi, readJsonlResponse } from "./huggingchat/jsonlStream.ts"; +import { + HuggingChatStreamError, + readJsonlResponse, + streamJsonlToOpenAi, +} from "./huggingchat/jsonlStream.ts"; const HUGGINGFACE_BASE = "https://huggingface.co"; const CONVERSATION_URL = `${HUGGINGFACE_BASE}/chat/conversation`; @@ -38,6 +42,7 @@ const USER_AGENT = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36"; const DEFAULT_MODEL = "baidu/ERNIE-4.5-VL-424B-A47B-Base-PT"; +const HUGGINGCHAT_PUBLIC_STREAM_ERROR = "HuggingChat generation failed"; // -- Helpers ----------------------------------------------------------------- @@ -527,25 +532,80 @@ export class HuggingChatExecutor extends BaseExecutor { if (stream) { const encoder = new TextEncoder(); + const streamCancellationController = new AbortController(); const jsonlStream = streamJsonlToOpenAi( upstreamResponse.body, resolvedModel, id, created, - signal + signal, + streamCancellationController.signal ); - const sseStream = new ReadableStream({ - async start(controller) { - try { - for await (const chunk of jsonlStream) { - controller.enqueue(encoder.encode(chunk)); - } - } catch (err) { - log?.error?.("HUGGINGCHAT", `Stream error: ${err}`); - } finally { - controller.close(); + const primedChunks: string[] = []; + try { + const first = await jsonlStream.next(); + if (!first.done) { + primedChunks.push(first.value); + if (first.value.includes('"role":"assistant"')) { + const content = await jsonlStream.next(); + if (!content.done) primedChunks.push(content.value); } + } + } catch (err) { + if (!(err instanceof HuggingChatStreamError)) throw err; + const message = err instanceof Error ? err.message : String(err); + const safeMessage = sanitizeErrorMessage(message); + log?.error?.("HUGGINGCHAT", `Stream failed before content: ${safeMessage}`); + return { + response: new Response( + JSON.stringify( + buildErrorBody(502, message, undefined, { + type: "upstream_error", + code: "huggingchat_generation_error", + }) + ), + { status: 502, headers: { "Content-Type": "application/json" } } + ), + url: messageUrl, + headers: baseHeaders, + transformedBody: sendDataPayload, + }; + } + + let primedChunkIndex = 0; + let streamCancelled = false; + const sseStream = new ReadableStream({ + async pull(controller) { + if (streamCancelled) return; + if (primedChunkIndex < primedChunks.length) { + controller.enqueue(encoder.encode(primedChunks[primedChunkIndex])); + primedChunkIndex += 1; + return; + } + + try { + const chunk = await jsonlStream.next(); + if (streamCancelled) return; + if (chunk.done) { + controller.close(); + return; + } + controller.enqueue(encoder.encode(chunk.value)); + } catch (err) { + if (streamCancelled) return; + const message = err instanceof Error ? err.message : String(err); + const safeMessage = sanitizeErrorMessage(message); + log?.error?.("HUGGINGCHAT", `Stream error: ${safeMessage}`); + controller.error( + Object.assign(new Error(HUGGINGCHAT_PUBLIC_STREAM_ERROR), { statusCode: 502 }) + ); + } + }, + cancel() { + streamCancelled = true; + streamCancellationController.abort(); + void jsonlStream.return(undefined).catch(() => undefined); }, }); @@ -564,7 +624,29 @@ export class HuggingChatExecutor extends BaseExecutor { }; } - const fullText = await readJsonlResponse(upstreamResponse.body, signal); + let fullText: string; + try { + fullText = await readJsonlResponse(upstreamResponse.body, signal); + } catch (err) { + if (!(err instanceof HuggingChatStreamError)) throw err; + const message = err instanceof Error ? err.message : String(err); + const safeMessage = sanitizeErrorMessage(message); + log?.error?.("HUGGINGCHAT", `Generation error: ${safeMessage}`); + return { + response: new Response( + JSON.stringify( + buildErrorBody(502, message, undefined, { + type: "upstream_error", + code: "huggingchat_generation_error", + }) + ), + { status: 502, headers: { "Content-Type": "application/json" } } + ), + url: messageUrl, + headers: baseHeaders, + transformedBody: sendDataPayload, + }; + } const completionTokens = estimateTokens(fullText); return { diff --git a/open-sse/executors/huggingchat/jsonlStream.ts b/open-sse/executors/huggingchat/jsonlStream.ts index b09bcb2c30..3d4980aebb 100644 --- a/open-sse/executors/huggingchat/jsonlStream.ts +++ b/open-sse/executors/huggingchat/jsonlStream.ts @@ -1,5 +1,36 @@ // Pure JSONL stream translation (HuggingChat NDJSON -> OpenAI SSE). Verbatim from huggingchat.ts. +export class HuggingChatStreamError extends Error { + constructor(message: string) { + super(message); + this.name = "HuggingChatStreamError"; + } +} + +function cancelReader(reader: ReadableStreamDefaultReader): void { + try { + void reader.cancel().catch(() => undefined); + } catch { + // The error event is authoritative; transport cleanup is best effort. + } +} + +function bindReaderCancellation( + reader: ReadableStreamDefaultReader, + signal?: AbortSignal | null +): () => void { + if (!signal) return () => undefined; + + const cancel = () => cancelReader(reader); + if (signal.aborted) { + cancel(); + return () => undefined; + } + + signal.addEventListener("abort", cancel, { once: true }); + return () => signal.removeEventListener("abort", cancel); +} + export function sseChunk(data: unknown): string { return `data: ${JSON.stringify(data)}\n\n`; } @@ -42,9 +73,11 @@ export async function* streamJsonlToOpenAi( model: string, id: string, created: number, - signal?: AbortSignal | null + signal?: AbortSignal | null, + cancellationSignal?: AbortSignal | null ): AsyncGenerator { const reader = body.getReader(); + const unbindReaderCancellation = bindReaderCancellation(reader, cancellationSignal); const decoder = new TextDecoder(); let buffer = ""; let emittedRole = false; @@ -70,16 +103,8 @@ export async function* streamJsonlToOpenAi( const parsed = parseJsonlLine(trimmed); if (parsed.error) { - yield sseChunk({ - id, - object: "chat.completion.chunk", - created, - model, - choices: [{ index: 0, delta: {}, finish_reason: "stop" }], - }); - yield "data: [DONE]\n\n"; - finished = true; - return; + cancelReader(reader); + throw new HuggingChatStreamError(parsed.error); } if (parsed.token) { @@ -140,6 +165,9 @@ export async function* streamJsonlToOpenAi( if (!finished && buffer.trim()) { const parsed = parseJsonlLine(buffer.trim()); + if (parsed.error) { + throw new HuggingChatStreamError(parsed.error); + } if (parsed.token && !signal?.aborted) { if (!emittedRole) { emittedRole = true; @@ -161,10 +189,11 @@ export async function* streamJsonlToOpenAi( } } } finally { + unbindReaderCancellation(); reader.releaseLock(); } - if (!signal?.aborted) { + if (!signal?.aborted && !cancellationSignal?.aborted) { yield sseChunk({ id, object: "chat.completion.chunk", @@ -172,7 +201,9 @@ export async function* streamJsonlToOpenAi( model, choices: [{ index: 0, delta: {}, finish_reason: "stop" }], }); - yield "data: [DONE]\n\n"; + if (!signal?.aborted && !cancellationSignal?.aborted) { + yield "data: [DONE]\n\n"; + } } } @@ -204,7 +235,10 @@ export async function readJsonlResponse( const parsed = parseJsonlLine(trimmed); if (parsed.token) fullText += parsed.token; if (parsed.text) return parsed.text; - if (parsed.error) throw new Error(parsed.error); + if (parsed.error) { + cancelReader(reader); + throw new HuggingChatStreamError(parsed.error); + } } } @@ -212,6 +246,7 @@ export async function readJsonlResponse( const parsed = parseJsonlLine(buffer.trim()); if (parsed.text) return parsed.text; if (parsed.token) fullText += parsed.token; + if (parsed.error) throw new HuggingChatStreamError(parsed.error); } } finally { reader.releaseLock(); diff --git a/tests/unit/_fixtures/huggingchat-stream-error-boundary.fixture.ts b/tests/unit/_fixtures/huggingchat-stream-error-boundary.fixture.ts new file mode 100644 index 0000000000..fc901496d2 --- /dev/null +++ b/tests/unit/_fixtures/huggingchat-stream-error-boundary.fixture.ts @@ -0,0 +1,592 @@ +import assert from "node:assert/strict"; +import { isAbsolute, relative } from "node:path"; +import { after, test } from "node:test"; + +function requiredEnv(name: string): string { + const value = process.env[name]; + assert.ok(value, `${name} must be supplied by the isolated parent wrapper`); + return value; +} + +const testRoot = requiredEnv("OMNIROUTE_HUGGINGCHAT_TEST_ROOT"); +const fixtureRunId = requiredEnv("OMNIROUTE_HUGGINGCHAT_TEST_RUN_ID"); +const testDataDir = requiredEnv("DATA_DIR"); +const testPluginsDir = requiredEnv("OMNIROUTE_PLUGINS_DIR"); +const xdgConfigDir = requiredEnv("XDG_CONFIG_HOME"); + +for (const [name, candidate] of [ + ["DATA_DIR", testDataDir], + ["OMNIROUTE_PLUGINS_DIR", testPluginsDir], + ["XDG_CONFIG_HOME", xdgConfigDir], +] as const) { + const fromRoot = relative(testRoot, candidate); + assert.equal( + isAbsolute(fromRoot) || fromRoot.startsWith(".."), + false, + `${name} escaped test root` + ); +} +assert.equal( + process.env.NODE_TEST_CONTEXT, + undefined, + "nested node:test state must not be inherited" +); +assert.equal(process.env.HOME, undefined, "the child must not inherit the operator HOME"); +assert.equal(process.env.CODEX_HOME, undefined, "the child must not inherit CODEX_HOME"); +assert.match(requiredEnv("API_KEY_SECRET"), /^[0-9a-f]{64}$/); + +const [ + { HuggingChatExecutor }, + { HuggingChatStreamError, streamJsonlToOpenAi }, + { createPassthroughStreamWithLogger }, + { createStreamController, pipeWithDisconnect }, + { createStreamFailureFinalizers, finalizeStreamRequestLog }, + { ensureStreamReadiness }, + { FORMATS }, + usageHistory, + coreDb, + callLogs, + callLogArtifactWriter, + loggerResource, +] = await Promise.all([ + import("../../../open-sse/executors/huggingchat.ts"), + import("../../../open-sse/executors/huggingchat/jsonlStream.ts"), + import("../../../open-sse/utils/stream.ts"), + import("../../../open-sse/utils/streamHandler.ts"), + import("../../../open-sse/utils/streamFailureFinalization.ts"), + import("../../../open-sse/utils/streamReadiness.ts"), + import("../../../open-sse/translator/formats.ts"), + import("../../../src/lib/usage/usageHistory.ts"), + import("../../../src/lib/db/core.ts"), + import("../../../src/lib/usage/callLogs.ts"), + import("../../../src/lib/usage/callLogArtifactWriter.ts"), + import("../../../src/shared/utils/loggerResource.ts"), +]); + +after(async () => { + assert.equal( + await callLogs.waitForCallLogSaves(10_000), + true, + "all asynchronous call-log writes must drain before DB teardown" + ); + await callLogArtifactWriter.closeCallLogArtifactWriter(); + usageHistory.clearPendingRequests(); + await loggerResource.closeSharedLoggerResource(); + coreDb.resetDbInstance(); +}); + +function jsonlBody(lines: string[], trailingNewline = true): ReadableStream { + const encoded = new TextEncoder().encode(`${lines.join("\n")}${trailingNewline ? "\n" : ""}`); + return new ReadableStream({ + start(controller) { + controller.enqueue(encoded); + controller.close(); + }, + }); +} + +async function collectStream(body: ReadableStream): Promise { + const chunks: string[] = []; + for await (const chunk of streamJsonlToOpenAi( + body, + "test/huggingchat-model", + "chatcmpl-huggingchat-test", + 1_725_000_000 + )) { + chunks.push(chunk); + } + return chunks.join(""); +} + +test("HuggingChat turns a pre-content JSONL generation error into a sanitized 502", async () => { + const rawError = + "generation failed at /srv/omniroute/providers/huggingchat.ts:44:9 api_key=super-secret\n" + + " at provider (/srv/omniroute/runtime.ts:1:1)"; + const realFetch = globalThis.fetch; + let callCount = 0; + const errorLogs: string[] = []; + + globalThis.fetch = (async () => { + callCount += 1; + if (callCount === 1) { + return Response.json({ conversationId: "conversation-test" }); + } + if (callCount === 2) { + return Response.json({ rootMessageId: "root-message-test" }); + } + if (callCount === 3) { + return new Response( + jsonlBody( + [ + JSON.stringify({ type: "status", status: "started" }), + JSON.stringify({ type: "status", status: "error", message: rawError }), + ], + false + ), + { status: 200, headers: { "Content-Type": "application/jsonl" } } + ); + } + throw new Error(`Unexpected fetch call ${callCount}`); + }) as typeof globalThis.fetch; + + try { + const result = await new HuggingChatExecutor().execute({ + model: "test/huggingchat-model", + body: { messages: [{ role: "user", content: "hello" }] }, + stream: true, + credentials: { apiKey: "hf-chat=fake-cookie" }, + signal: null, + log: { error: (_tag, message) => errorLogs.push(message) }, + }); + + assert.equal(callCount, 3, "the test must intercept every HuggingChat request"); + assert.equal(result.response.status, 502); + assert.match(result.response.headers.get("content-type") || "", /application\/json/); + + const payload = (await result.response.json()) as { + error: { message: string; type?: string; code?: string }; + }; + assert.equal(payload.error.type, "upstream_error"); + assert.equal(payload.error.code, "huggingchat_generation_error"); + assert.match(payload.error.message, /generation failed/); + assert.doesNotMatch(payload.error.message, /\/srv\/omniroute/); + assert.doesNotMatch(payload.error.message, /super-secret/); + assert.doesNotMatch(payload.error.message, /\n\s*at /); + assert.equal(errorLogs.length, 1); + assert.doesNotMatch(errorLogs[0], /\/srv\/omniroute/); + assert.doesNotMatch(errorLogs[0], /super-secret/); + assert.doesNotMatch(errorLogs[0], /\n\s*at /); + } finally { + globalThis.fetch = realFetch; + } +}); + +test("HuggingChat turns a terminal non-stream JSONL error into a sanitized 502", async () => { + const rawError = + "generation failed at /srv/omniroute/providers/huggingchat.ts:55:2 cookie=super-secret\n" + + " at provider (/srv/omniroute/runtime.ts:1:1)"; + const realFetch = globalThis.fetch; + let callCount = 0; + + globalThis.fetch = (async () => { + callCount += 1; + if (callCount === 1) return Response.json({ conversationId: "conversation-test" }); + if (callCount === 2) return Response.json({ rootMessageId: "root-message-test" }); + if (callCount === 3) { + return new Response( + jsonlBody([JSON.stringify({ type: "status", status: "error", message: rawError })], false), + { status: 200, headers: { "Content-Type": "application/jsonl" } } + ); + } + throw new Error(`Unexpected fetch call ${callCount}`); + }) as typeof globalThis.fetch; + + try { + const result = await new HuggingChatExecutor().execute({ + model: "test/huggingchat-model", + body: { messages: [{ role: "user", content: "hello" }] }, + stream: false, + credentials: { apiKey: "hf-chat=fake-cookie" }, + signal: null, + }); + + assert.equal(callCount, 3, "the test must intercept every HuggingChat request"); + assert.equal(result.response.status, 502); + const payload = (await result.response.json()) as { + error: { message: string; type?: string; code?: string }; + }; + assert.equal(payload.error.type, "upstream_error"); + assert.equal(payload.error.code, "huggingchat_generation_error"); + assert.doesNotMatch(payload.error.message, /\/srv\/omniroute/); + assert.doesNotMatch(payload.error.message, /super-secret/); + assert.doesNotMatch(payload.error.message, /\n\s*at /); + } finally { + globalThis.fetch = realFetch; + } +}); + +test("HuggingChat rejects its JSONL generator after partial content instead of faking success", async () => { + const rawError = + "generation failed at /srv/omniroute/providers/huggingchat.ts:44:9 access_token=super-secret\n" + + " at provider (/srv/omniroute/runtime.ts:1:1)"; + const stream = streamJsonlToOpenAi( + jsonlBody([ + JSON.stringify({ type: "stream", token: "partial answer" }), + JSON.stringify({ type: "status", status: "error", message: rawError }), + ]), + "test/huggingchat-model", + "chatcmpl-huggingchat-test", + 1_725_000_000 + ); + + const roleChunk = await stream.next(); + const contentChunk = await stream.next(); + + assert.equal(roleChunk.done, false); + assert.match(roleChunk.value || "", /"role":"assistant"/); + assert.equal(contentChunk.done, false); + assert.match(contentChunk.value || "", /partial answer/); + await assert.rejects(() => stream.next(), HuggingChatStreamError); +}); + +test("HuggingChat partial failures reach stream finalization, persistence, and fallback", async () => { + const model = `test/huggingchat-model-${fixtureRunId}`; + const provider = "huggingchat"; + const connectionId = `huggingchat-stream-error-boundary-${fixtureRunId}`; + const publicErrorMessage = "HuggingChat generation failed"; + const rawError = + "generation failed at /srv/omniroute/providers/huggingchat.ts:44:9 access_token=super-secret\n" + + " at provider (/srv/omniroute/runtime.ts:1:1)"; + const realFetch = globalThis.fetch; + let callCount = 0; + const errorLogs: string[] = []; + + globalThis.fetch = (async () => { + callCount += 1; + if (callCount === 1) return Response.json({ conversationId: "conversation-test" }); + if (callCount === 2) return Response.json({ rootMessageId: "root-message-test" }); + if (callCount === 3) { + return new Response( + jsonlBody([ + JSON.stringify({ type: "stream", token: "partial answer" }), + JSON.stringify({ type: "status", status: "error", message: rawError }), + ]), + { status: 200, headers: { "Content-Type": "application/jsonl" } } + ); + } + throw new Error(`Unexpected fetch call ${callCount}`); + }) as typeof globalThis.fetch; + + usageHistory.clearPendingRequests(); + assert.equal(usageHistory.getPendingById().size, 0, "the child must start without pending state"); + assert.equal( + usageHistory.getCompletedDetails().size, + 0, + "the child must start without completed state" + ); + const previousPersistence = coreDb + .getDbInstance() + .prepare("SELECT COUNT(*) AS count FROM call_logs WHERE connection_id = ? AND model = ?") + .get(connectionId, model) as { count: number }; + assert.equal(previousPersistence.count, 0, "the child must not reuse a prior persisted identity"); + const requestId = usageHistory.trackPendingRequest(model, provider, connectionId, true); + assert.ok(requestId, "the full-pipeline test must own a real pending request"); + + type CompletionPayload = { + status: number; + usage: unknown; + providerPayload?: unknown; + clientPayload?: unknown; + error?: string | null; + errorCode?: string | null; + }; + type FailurePayload = { + status: number; + message: string; + code?: string; + type?: string; + }; + + let completionPayload: CompletionPayload | null = null; + let streamCompletionRecorded = false; + let streamFailureCompletionRecorded = false; + const persistedFailures: Array<{ status: number; errorCode?: string }> = []; + const fallbackFailures: FailurePayload[] = []; + + const onStreamComplete = (payload: CompletionPayload) => { + const normalizedStatus = payload.status || 200; + if (streamCompletionRecorded) return; + streamCompletionRecorded = true; + if (normalizedStatus !== 200) { + if (streamFailureCompletionRecorded) return; + streamFailureCompletionRecorded = true; + } + completionPayload = payload; + finalizeStreamRequestLog({ + pendingRequestId: requestId, + model, + provider, + connectionId, + providerResponse: payload.providerPayload, + clientResponse: payload.clientPayload, + status: normalizedStatus, + error: payload.error, + errorCode: payload.errorCode, + }); + }; + + const { handleStreamFailure, onPipelineStreamError } = createStreamFailureFinalizers({ + isFailureCompletionRecorded: () => streamFailureCompletionRecorded, + isStreamCompletionRecorded: () => streamCompletionRecorded, + onStreamComplete, + persistFailureUsage: (status, errorCode) => { + persistedFailures.push({ status, errorCode }); + }, + onStreamFailure: (failure) => { + fallbackFailures.push(failure); + }, + }); + + try { + const result = await new HuggingChatExecutor().execute({ + model, + body: { messages: [{ role: "user", content: "hello" }] }, + stream: true, + credentials: { apiKey: "hf-chat=fake-cookie" }, + signal: null, + log: { error: (_tag, message) => errorLogs.push(message) }, + }); + + assert.equal(callCount, 3, "the test must intercept every HuggingChat request"); + assert.equal(result.response.status, 200, "partial output has already committed HTTP 200"); + const readiness = await ensureStreamReadiness(result.response, { + timeoutMs: 1_000, + provider, + model, + }); + if (!readiness.ok) assert.fail(`unexpected readiness failure: ${readiness.reason}`); + + const transform = createPassthroughStreamWithLogger( + provider, + null, + null, + model, + connectionId, + { messages: [{ role: "user", content: "hello" }] }, + onStreamComplete, + null, + handleStreamFailure, + FORMATS.OPENAI + ); + const streamController = createStreamController({ + onError: onPipelineStreamError, + provider, + model, + connectionId, + clientResponseFormat: FORMATS.OPENAI, + }); + const clientStream = pipeWithDisconnect(readiness.response, transform, streamController, { + stallTimeoutMs: 0, + }); + const wire = await new Response(clientStream).text(); + + assert.match(wire, /partial answer/); + assert.match(wire, /"finish_reason":"error"/); + assert.match(wire, new RegExp(publicErrorMessage)); + assert.match(wire, /data: \[DONE\]/); + assert.doesNotMatch(wire, /"finish_reason":"stop"/); + assert.doesNotMatch(wire, /\/srv\/omniroute/); + assert.doesNotMatch(wire, /super-secret/); + assert.equal(errorLogs.length, 1); + assert.doesNotMatch(errorLogs[0], /\/srv\/omniroute/); + assert.doesNotMatch(errorLogs[0], /super-secret/); + assert.doesNotMatch(errorLogs[0], /\n\s*at /); + + assert.ok(completionPayload, "the pipeline must record a terminal failure"); + assert.equal(completionPayload.status, 502); + assert.equal(completionPayload.error, publicErrorMessage); + assert.equal(completionPayload.errorCode, "stream_pipeline_error"); + assert.deepEqual(persistedFailures, [{ status: 502, errorCode: "stream_pipeline_error" }]); + assert.deepEqual(fallbackFailures, [ + { + status: 502, + message: publicErrorMessage, + code: "stream_pipeline_error", + type: "stream_error", + }, + ]); + + assert.equal(usageHistory.getPendingById().has(requestId), false); + const completedDetail = usageHistory.getCompletedDetails().get(requestId); + assert.ok(completedDetail, "failure finalization must persist the completed request detail"); + assert.equal(completedDetail.status, 502); + assert.equal(completedDetail.error, publicErrorMessage); + assert.equal(completedDetail.errorCode, "stream_pipeline_error"); + assert.doesNotMatch(JSON.stringify(completedDetail), /\/srv\/omniroute|super-secret/); + assert.deepEqual( + [...usageHistory.getCompletedDetails().keys()], + [requestId], + "only this child run may own completed usage state" + ); + } finally { + globalThis.fetch = realFetch; + usageHistory.clearPendingRequests(); + } +}); + +test("HuggingChat reports an authoritative error without waiting for transport cancellation", async () => { + let cancelCalled = false; + const encoded = new TextEncoder().encode( + `${JSON.stringify({ type: "status", status: "error", message: "provider failed" })}\n` + ); + const body = new ReadableStream({ + start(controller) { + controller.enqueue(encoded); + }, + cancel() { + cancelCalled = true; + return new Promise(() => undefined); + }, + }); + const stream = streamJsonlToOpenAi( + body, + "test/huggingchat-model", + "chatcmpl-huggingchat-test", + 1_725_000_000 + ); + + const outcome = await Promise.race([ + stream.next().then( + () => ({ kind: "resolved" as const }), + (error: unknown) => ({ kind: "rejected" as const, error }) + ), + new Promise<{ kind: "hung" }>((resolve) => { + setImmediate(() => resolve({ kind: "hung" })); + }), + ]); + + assert.equal(cancelCalled, true); + assert.equal(outcome.kind, "rejected", "transport cleanup must not delay error delivery"); + assert.ok( + outcome.kind === "rejected" && outcome.error instanceof HuggingChatStreamError, + "the authoritative HuggingChat error must remain classifiable" + ); +}); + +test("HuggingChat cancellation suppresses final chunks after a pending JSONL read", async () => { + let upstreamCancelCalled = false; + let upstreamPullCount = 0; + const cancellationController = new AbortController(); + const token = new TextEncoder().encode( + `${JSON.stringify({ type: "stream", token: "partial answer" })}\n` + ); + const body = new ReadableStream({ + pull(controller) { + upstreamPullCount += 1; + if (upstreamPullCount === 1) { + controller.enqueue(token); + return; + } + return new Promise(() => undefined); + }, + cancel() { + upstreamCancelCalled = true; + return new Promise(() => undefined); + }, + }); + const stream = streamJsonlToOpenAi( + body, + "test/huggingchat-model", + "chatcmpl-huggingchat-test", + 1_725_000_000, + null, + cancellationController.signal + ); + + assert.match((await stream.next()).value || "", /"role":"assistant"/); + assert.match((await stream.next()).value || "", /partial answer/); + const pendingNext = stream.next(); + await new Promise((resolve) => setImmediate(resolve)); + cancellationController.abort(); + + const outcome = await Promise.race([ + pendingNext.then((result) => ({ kind: "settled" as const, result })), + new Promise<{ kind: "hung" }>((resolve) => setImmediate(() => resolve({ kind: "hung" }))), + ]); + + assert.equal(upstreamCancelCalled, true); + assert.equal(outcome.kind, "settled", "cancellation must settle the pending generator read"); + assert.equal( + outcome.kind === "settled" ? outcome.result.done : false, + true, + "a cancelled generator must not emit stop or [DONE]" + ); + void stream.return(undefined).catch(() => undefined); +}); + +test("HuggingChat client cancellation reaches a blocked upstream reader without waiting", async () => { + const realFetch = globalThis.fetch; + let callCount = 0; + let upstreamCancelCalled = false; + let upstreamPullCount = 0; + const errorLogs: string[] = []; + const token = new TextEncoder().encode( + `${JSON.stringify({ type: "stream", token: "partial answer" })}\n` + ); + const blockedBody = new ReadableStream({ + pull(controller) { + upstreamPullCount += 1; + if (upstreamPullCount === 1) { + controller.enqueue(token); + return; + } + return new Promise(() => undefined); + }, + cancel() { + upstreamCancelCalled = true; + return new Promise(() => undefined); + }, + }); + + globalThis.fetch = (async () => { + callCount += 1; + if (callCount === 1) return Response.json({ conversationId: "conversation-test" }); + if (callCount === 2) return Response.json({ rootMessageId: "root-message-test" }); + if (callCount === 3) { + return new Response(blockedBody, { + status: 200, + headers: { "Content-Type": "application/jsonl" }, + }); + } + throw new Error(`Unexpected fetch call ${callCount}`); + }) as typeof globalThis.fetch; + + try { + const result = await new HuggingChatExecutor().execute({ + model: "test/huggingchat-model", + body: { messages: [{ role: "user", content: "hello" }] }, + stream: true, + credentials: { apiKey: "hf-chat=fake-cookie" }, + signal: null, + log: { error: (_tag, message) => errorLogs.push(message) }, + }); + + assert.equal(callCount, 3, "the test must intercept every HuggingChat request"); + assert.ok(result.response.body); + const reader = result.response.body.getReader(); + const roleChunk = await reader.read(); + const contentChunk = await reader.read(); + assert.match(new TextDecoder().decode(roleChunk.value), /"role":"assistant"/); + assert.match(new TextDecoder().decode(contentChunk.value), /partial answer/); + + const blockedRead = reader.read(); + await new Promise((resolve) => setImmediate(resolve)); + const cancelOutcome = await Promise.race([ + reader.cancel("client disconnected").then(() => "resolved" as const), + new Promise<"hung">((resolve) => setImmediate(() => resolve("hung"))), + ]); + void blockedRead.catch(() => undefined); + + assert.equal(cancelOutcome, "resolved", "downstream cancellation must remain non-blocking"); + assert.equal(upstreamCancelCalled, true, "cancellation must reach the locked upstream reader"); + await new Promise((resolve) => setImmediate(resolve)); + assert.deepEqual(errorLogs, [], "client cancellation must not log a provider stream failure"); + } finally { + globalThis.fetch = realFetch; + } +}); + +test("HuggingChat keeps the normal JSONL completion contract unchanged", async () => { + const output = await collectStream( + jsonlBody([ + JSON.stringify({ type: "stream", token: "complete answer" }), + JSON.stringify({ type: "status", status: "finished" }), + ]) + ); + + assert.match(output, /"role":"assistant"/); + assert.match(output, /complete answer/); + assert.match(output, /"finish_reason":"stop"/); + assert.match(output, /data: \[DONE\]/); + assert.doesNotMatch(output, /"error":\{/); +}); diff --git a/tests/unit/huggingchat-stream-error-boundary.test.ts b/tests/unit/huggingchat-stream-error-boundary.test.ts new file mode 100644 index 0000000000..5fe1e9f545 --- /dev/null +++ b/tests/unit/huggingchat-stream-error-boundary.test.ts @@ -0,0 +1,85 @@ +import assert from "node:assert/strict"; +import { spawn } from "node:child_process"; +import { mkdirSync, mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { basename, join } from "node:path"; +import { fileURLToPath } from "node:url"; +import test from "node:test"; + +const REPO_ROOT = fileURLToPath(new URL("../..", import.meta.url)); +const FIXTURE = fileURLToPath( + new URL("./_fixtures/huggingchat-stream-error-boundary.fixture.ts", import.meta.url) +); +const SYNTHETIC_API_KEY_SECRET = "0".repeat(64); + +type ChildResult = { + code: number | null; + signal: NodeJS.Signals | null; + stdout: string; + stderr: string; +}; + +function runFixture(testRoot: string): Promise { + const dataDir = join(testRoot, "data"); + const pluginsDir = join(testRoot, "plugins"); + // Keep config fallbacks inside the fixture root without inheriting or repurposing HOME. + const xdgConfigDir = join(testRoot, "xdg-config"); + + for (const dir of [dataDir, pluginsDir, xdgConfigDir]) { + mkdirSync(dir, { recursive: true }); + } + + const childEnv: NodeJS.ProcessEnv = { + API_KEY_SECRET: SYNTHETIC_API_KEY_SECRET, + APP_LOG_TO_FILE: "false", + DATA_DIR: dataDir, + DISABLE_SQLITE_AUTO_BACKUP: "true", + FORCE_COLOR: "0", + LANG: "C.UTF-8", + NODE_ENV: "test", + OMNIROUTE_HUGGINGCHAT_TEST_ROOT: testRoot, + OMNIROUTE_HUGGINGCHAT_TEST_RUN_ID: basename(testRoot), + OMNIROUTE_PLUGINS_DIR: pluginsDir, + TZ: "UTC", + XDG_CONFIG_HOME: xdgConfigDir, + }; + if (process.env.PATH) childEnv.PATH = process.env.PATH; + + return new Promise((resolve, reject) => { + const child = spawn(process.execPath, ["--import", "tsx/esm", FIXTURE], { + cwd: REPO_ROOT, + env: childEnv, + stdio: ["ignore", "pipe", "pipe"], + }); + let stdout = ""; + let stderr = ""; + child.stdout.setEncoding("utf8").on("data", (chunk) => (stdout += chunk)); + child.stderr.setEncoding("utf8").on("data", (chunk) => (stderr += chunk)); + child.once("error", reject); + child.once("close", (code, signal) => resolve({ code, signal, stdout, stderr })); + }); +} + +function childDiagnostics(result: ChildResult): string { + return [ + `exit=${String(result.code)} signal=${String(result.signal)}`, + "--- stdout ---", + result.stdout, + "--- stderr ---", + result.stderr, + ].join("\n"); +} + +test("HuggingChat stream error boundaries stay isolated from shared DB and usage state", async () => { + const testRoot = mkdtempSync(join(tmpdir(), "omniroute-huggingchat-boundary-child-")); + try { + const result = await runFixture(testRoot); + assert.equal(result.signal, null, childDiagnostics(result)); + assert.equal(result.code, 0, childDiagnostics(result)); + assert.match(result.stdout, /(?:#|ℹ) pass 8\b/, childDiagnostics(result)); + assert.match(result.stdout, /(?:#|ℹ) fail 0\b/, childDiagnostics(result)); + assert.doesNotMatch(result.stdout + result.stderr, /super-secret/); + } finally { + rmSync(testRoot, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + } +}); From ca2edfdca812c2b7e53212e264be644e4582926c Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 20:59:56 -0300 Subject: [PATCH 067/143] fix(grok-web): stop streaming errors from reporting false success (#12458) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 14 PRs desta campanha de error-boundary sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **120/120** nos 23 arquivos de teste que os PRs trazem. Um ponto que só apareceu no tree combinado: **#12465 e #12466 criam o mesmo arquivo novo** `open-sse/utils/streamReadiness.ts` (que não existe no tip) com desenhos divergentes de cancelamento — `cancelled` + `releaseLock` imediato num, `readInFlight`/`cancelRequested` com `cancelReader` fire-and-forget no outro. Adotei a versão do #12466, que difere e defere o release do lock para quando a leitura em voo termina, e validei a escolha rodando as suítes dos **dois** PRs contra ela: 21/21 no readiness compartilhado e 22/22 incluindo o boundary do Perplexity. --- .../pending-grok-web-stream-error-boundary.md | 4 + open-sse/executors/grok-web.ts | 216 +++++--- .../grok-web-stream-error-boundary-child.ts | 517 ++++++++++++++++++ .../grok-web-stream-error-boundary.test.ts | 86 +++ 4 files changed, 757 insertions(+), 66 deletions(-) create mode 100644 changelog.d/fixes/pending-grok-web-stream-error-boundary.md create mode 100644 tests/fixtures/grok-web-stream-error-boundary-child.ts create mode 100644 tests/unit/grok-web-stream-error-boundary.test.ts diff --git a/changelog.d/fixes/pending-grok-web-stream-error-boundary.md b/changelog.d/fixes/pending-grok-web-stream-error-boundary.md new file mode 100644 index 0000000000..28e4292504 --- /dev/null +++ b/changelog.d/fixes/pending-grok-web-stream-error-boundary.md @@ -0,0 +1,4 @@ +- **fix(grok-web):** treat upstream streaming failures as failures instead of successful + assistant text: error-only streams now fail readiness with HTTP 502, while failures after + legitimate content preserve that partial output and terminate through the sanitized stream + failure path without a normal `stop` completion. diff --git a/open-sse/executors/grok-web.ts b/open-sse/executors/grok-web.ts index 939a86903c..011e81c32b 100644 --- a/open-sse/executors/grok-web.ts +++ b/open-sse/executors/grok-web.ts @@ -19,7 +19,7 @@ import { type ExecuteInput, type ExecutorLog, } from "./base.ts"; -import { FETCH_TIMEOUT_MS } from "../config/constants.ts"; +import { FETCH_TIMEOUT_MS, STREAM_READINESS_TIMEOUT_MS } from "../config/constants.ts"; import { buildGrokCookieHeader } from "@/lib/providers/webCookieAuth"; import { tlsFetchGrok, @@ -27,7 +27,8 @@ import { isCloudflareChallenge, type TlsFetchResult, } from "../services/grokTlsClient.ts"; -import { sanitizeErrorMessage } from "../utils/error.ts"; +import { buildErrorBody, sanitizeErrorMessage } from "../utils/error.ts"; +import { ensureStreamReadiness } from "../utils/streamReadiness.ts"; import { shouldUseGrokBrowserBacked, acquireFreshGrokClearance, @@ -119,12 +120,29 @@ async function* readGrokNdjsonEvents( const reader = body.getReader(); const decoder = new TextDecoder(); let buffer = ""; + let reachedEnd = false; + let cancelRequested = false; + + const requestReaderCancel = (reason?: unknown) => { + if (cancelRequested || reachedEnd) return; + cancelRequested = true; + // Cancellation must release the upstream promptly even when a provider's + // underlying cancel promise never settles. + void reader.cancel(reason).catch(() => {}); + }; + const handleAbort = () => requestReaderCancel(signal?.reason); + + if (signal?.aborted) requestReaderCancel(signal.reason); + else signal?.addEventListener("abort", handleAbort, { once: true }); try { while (true) { if (signal?.aborted) return; const { value, done } = await reader.read(); - if (done) break; + if (done) { + reachedEnd = true; + break; + } buffer += decoder.decode(value, { stream: true }); while (true) { @@ -142,6 +160,8 @@ async function* readGrokNdjsonEvents( } } + if (signal?.aborted) return; + // Flush remaining buffer buffer += decoder.decode(); const remaining = buffer.trim(); @@ -153,7 +173,11 @@ async function* readGrokNdjsonEvents( } } } finally { - reader.releaseLock(); + signal?.removeEventListener("abort", handleAbort); + if (!reachedEnd) requestReaderCancel(signal?.reason ?? "Grok stream reader closed early"); + try { + reader.releaseLock(); + } catch {} } } @@ -271,6 +295,8 @@ async function* extractContent( } } + if (signal?.aborted) return; + const trailingThinking = suppressThinkingAfterVisibleContent && emittedVisibleContent ? "" : thinkingFilter.flush(); if (trailingThinking) { @@ -292,6 +318,25 @@ function sseChunk(data: unknown): string { return `data: ${JSON.stringify(data)}\n\n`; } +const GROK_STREAM_FAILURE_MESSAGE = "Grok upstream stream failed"; +const GROK_STREAM_FAILURE_CODE = "GROK_STREAM_ERROR"; + +function grokStreamErrorChunk(): string { + return sseChunk( + buildErrorBody(502, GROK_STREAM_FAILURE_MESSAGE, undefined, { + type: "upstream_error", + code: GROK_STREAM_FAILURE_CODE, + }) + ); +} + +function grokStreamFailure(): Error & { statusCode: number; code: string } { + return Object.assign(new Error(GROK_STREAM_FAILURE_MESSAGE), { + statusCode: 502, + code: GROK_STREAM_FAILURE_CODE, + }); +} + function enqueueStreamingToolCalls( controller: ReadableStreamDefaultController, encoder: TextEncoder, @@ -349,63 +394,77 @@ function buildStreamingResponse( signal?: AbortSignal | null ): ReadableStream { const encoder = new TextEncoder(); + const streamAbortController = new AbortController(); + const requestStreamCancel = (reason?: unknown) => { + if (!streamAbortController.signal.aborted) streamAbortController.abort(reason); + }; + const handleParentAbort = () => requestStreamCancel(signal?.reason); + + if (signal?.aborted) requestStreamCancel(signal.reason); + else signal?.addEventListener("abort", handleParentAbort, { once: true }); return new ReadableStream( { async start(controller) { + let roleSent = false; + let firstOutputHandedOff = false; try { - // Initial role chunk - controller.enqueue( - encoder.encode( - sseChunk({ - id: cid, - object: "chat.completion.chunk", - created, - model, - system_fingerprint: null, - choices: [ - { index: 0, delta: { role: "assistant" }, finish_reason: null, logprobs: null }, - ], - }) - ) - ); - let fp = ""; let buffered = ""; + const enqueueRole = () => { + if (roleSent) return; + controller.enqueue( + encoder.encode( + sseChunk({ + id: cid, + object: "chat.completion.chunk", + created, + model, + system_fingerprint: fp || null, + choices: [ + { + index: 0, + delta: { role: "assistant" }, + finish_reason: null, + logprobs: null, + }, + ], + }) + ) + ); + roleSent = true; + }; + + const handOffFirstOutput = async () => { + if (firstOutputHandedOff) return; + firstOutputHandedOff = true; + // Give readiness/finalization wrappers one turn to attach before a later + // upstream failure errors the stream and invalidates queued chunks. + await new Promise((resolve) => setImmediate(resolve)); + }; + for await (const chunk of extractContent( eventStream, isThinkingModel, toolRegistry, - signal, + streamAbortController.signal, true )) { if (chunk.fingerprint) fp = chunk.fingerprint; if (chunk.error) { - controller.enqueue( - encoder.encode( - sseChunk({ - id: cid, - object: "chat.completion.chunk", - created, - model, - system_fingerprint: fp || null, - choices: [ - { - index: 0, - delta: { content: `[Error: ${chunk.error}]` }, - finish_reason: null, - logprobs: null, - }, - ], - }) - ) - ); - break; + if (roleSent) { + controller.error(grokStreamFailure()); + return; + } + controller.enqueue(encoder.encode(grokStreamErrorChunk())); + controller.enqueue(encoder.encode("data: [DONE]\n\n")); + return; } if (chunk.thinking) { + enqueueRole(); controller.enqueue( encoder.encode( sseChunk({ @@ -425,10 +484,12 @@ function buildStreamingResponse( }) ) ); + await handOffFirstOutput(); continue; } if (chunk.toolCalls) { + enqueueRole(); enqueueStreamingToolCalls(controller, encoder, { id: cid, created, @@ -444,6 +505,7 @@ function buildStreamingResponse( if (chunk.fullMessage) { const toolCalls = parseClientToolCallMarkup(chunk.fullMessage, toolRegistry); if (toolCalls) { + enqueueRole(); enqueueStreamingToolCalls(controller, encoder, { id: cid, created, @@ -453,6 +515,30 @@ function buildStreamingResponse( }); return; } + if (!buffered) { + enqueueRole(); + buffered = chunk.fullMessage; + controller.enqueue( + encoder.encode( + sseChunk({ + id: cid, + object: "chat.completion.chunk", + created, + model, + system_fingerprint: fp || null, + choices: [ + { + index: 0, + delta: { content: chunk.fullMessage }, + finish_reason: null, + logprobs: null, + }, + ], + }) + ) + ); + await handOffFirstOutput(); + } } if (chunk.delta) { @@ -469,6 +555,7 @@ function buildStreamingResponse( return; } if (hasOpenToolCallMarkup(buffered)) continue; + enqueueRole(); controller.enqueue( encoder.encode( sseChunk({ @@ -488,10 +575,13 @@ function buildStreamingResponse( }) ) ); + await handOffFirstOutput(); } } - // Stop chunk + if (streamAbortController.signal.aborted || !roleSent) return; + + // Stop chunk — only after legitimate content/reasoning/tool output. controller.enqueue( encoder.encode( sseChunk({ @@ -505,37 +595,24 @@ function buildStreamingResponse( ) ); controller.enqueue(encoder.encode("data: [DONE]\n\n")); - } catch (err) { - controller.enqueue( - encoder.encode( - sseChunk({ - id: cid, - object: "chat.completion.chunk", - created, - model, - system_fingerprint: null, - choices: [ - { - index: 0, - delta: { - content: sanitizeErrorMessage( - `[Stream error: ${err instanceof Error ? err.message : String(err)}]` - ), - }, - finish_reason: "stop", - logprobs: null, - }, - ], - }) - ) - ); + } catch { + if (streamAbortController.signal.aborted) return; + if (roleSent) { + controller.error(grokStreamFailure()); + return; + } + controller.enqueue(encoder.encode(grokStreamErrorChunk())); controller.enqueue(encoder.encode("data: [DONE]\n\n")); } finally { + signal?.removeEventListener("abort", handleParentAbort); try { controller.close(); } catch {} } }, + cancel(reason) { + requestStreamCancel(reason); + }, }, { highWaterMark: 16384 } ); @@ -1026,6 +1103,13 @@ export class GrokWebExecutor extends BaseExecutor { "X-Accel-Buffering": "no", }, }); + const readiness = await ensureStreamReadiness(finalResponse, { + timeoutMs: STREAM_READINESS_TIMEOUT_MS, + provider: this.provider, + model, + log, + }); + finalResponse = readiness.response; } else { finalResponse = await buildNonStreamingResponse( tlsResult.body, diff --git a/tests/fixtures/grok-web-stream-error-boundary-child.ts b/tests/fixtures/grok-web-stream-error-boundary-child.ts new file mode 100644 index 0000000000..b3ad226d52 --- /dev/null +++ b/tests/fixtures/grok-web-stream-error-boundary-child.ts @@ -0,0 +1,517 @@ +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; + +// This file is executed only by the process-isolated unit-test wrapper. State +// mutations and repository imports must remain here, never in the parent test. +const originalDataDir = process.env.DATA_DIR; +const originalPluginsDir = process.env.OMNIROUTE_PLUGINS_DIR; +const originalFetch = globalThis.fetch; +const testRoot = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-grok-web-stream-error-")); + +process.env.DATA_DIR = path.join(testRoot, "data"); +process.env.OMNIROUTE_PLUGINS_DIR = path.join(testRoot, "plugins"); +fs.mkdirSync(process.env.DATA_DIR, { recursive: true }); +fs.mkdirSync(process.env.OMNIROUTE_PLUGINS_DIR, { recursive: true }); +globalThis.fetch = async () => { + throw new Error("Unexpected network request in Grok stream error boundary test"); +}; + +const [ + { GrokWebExecutor }, + { __setTlsFetchOverrideForTesting }, + dbCore, + settingsDb, + callLogs, + artifactWriter, + { handleChatCore }, + usageHistory, + accountSemaphore, + requestDedup, + accountFallback, + loggerResource, +] = await Promise.all([ + import("../../open-sse/executors/grok-web.ts"), + import("../../open-sse/services/grokTlsClient.ts"), + import("../../src/lib/db/core.ts"), + import("../../src/lib/db/settings.ts"), + import("../../src/lib/usage/callLogs.ts"), + import("../../src/lib/usage/callLogArtifactWriter.ts"), + import("../../open-sse/handlers/chatCore.ts"), + import("../../src/lib/usage/usageHistory.ts"), + import("../../open-sse/services/accountSemaphore.ts"), + import("../../open-sse/services/requestDedup.ts"), + import("../../open-sse/services/accountFallback.ts"), + import("../../src/shared/utils/loggerResource.ts"), +]); + +function grokEventStream(events: unknown[]): ReadableStream { + const encoder = new TextEncoder(); + return new ReadableStream({ + start(controller) { + controller.enqueue( + encoder.encode(`${events.map((event) => JSON.stringify(event)).join("\n")}\n`) + ); + controller.close(); + }, + }); +} + +function stalledGrokEventStream( + events: unknown[], + onCancel: () => void +): ReadableStream { + const encoder = new TextEncoder(); + return new ReadableStream({ + start(controller) { + controller.enqueue( + encoder.encode(`${events.map((event) => JSON.stringify(event)).join("\n")}\n`) + ); + }, + pull() { + return new Promise(() => {}); + }, + cancel() { + onCancel(); + return new Promise(() => {}); + }, + }); +} + +type TestExecutorLog = { + debug?: (tag: string, message: string) => void; + info?: (tag: string, message: string) => void; + warn?: (tag: string, message: string) => void; + error?: (tag: string, message: string) => void; +}; + +async function executeStreamingBody( + upstreamBody: ReadableStream, + requestBody: Record = { + messages: [{ role: "user", content: "hello" }], + stream: true, + }, + options: { log?: TestExecutorLog | null; signal?: AbortSignal | null } = {} +): Promise { + __setTlsFetchOverrideForTesting(async () => ({ + status: 200, + headers: new Headers({ "Content-Type": "application/x-ndjson" }), + text: null, + body: upstreamBody, + })); + + const result = await new GrokWebExecutor().execute({ + model: "grok-4.1-fast", + body: requestBody, + stream: true, + credentials: { apiKey: "sso=test-only-cookie" }, + signal: options.signal ?? AbortSignal.timeout(10_000), + log: options.log ?? null, + }); + return result.response; +} + +function executeStreaming(events: unknown[]): Promise { + return executeStreamingBody(grokEventStream(events)); +} + +function parseSseData(text: string): unknown[] { + return text + .split(/\r?\n/) + .filter((line) => line.startsWith("data: ") && line !== "data: [DONE]") + .map((line) => JSON.parse(line.slice("data: ".length)) as unknown); +} + +async function readUntilFailure(response: Response): Promise<{ text: string; error: unknown }> { + assert.ok(response.body, "expected a streaming response body"); + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let text = ""; + try { + while (true) { + const { done, value } = await reader.read(); + if (done) return { text, error: null }; + text += decoder.decode(value, { stream: true }); + } + } catch (error) { + text += decoder.decode(); + return { text, error }; + } +} + +async function waitFor(read: () => Promise, timeoutMs = 3_000): Promise { + const startedAt = Date.now(); + while (Date.now() - startedAt < timeoutMs) { + const value = await read(); + if (value) return value; + await new Promise((resolve) => setTimeout(resolve, 25)); + } + return null; +} + +async function settlesWithin(promise: Promise, timeoutMs = 500): Promise { + let timeout: ReturnType | undefined; + const settled = await Promise.race([ + promise.then(() => true), + new Promise((resolve) => { + timeout = setTimeout(() => resolve(false), timeoutMs); + }), + ]); + if (timeout) clearTimeout(timeout); + return settled; +} + +test.afterEach(() => { + __setTlsFetchOverrideForTesting(null); + usageHistory.clearPendingRequests(); + accountSemaphore.resetAll(); + requestDedup.clearInflight(); + accountFallback.clearModelLock(); +}); + +test.after(async () => { + __setTlsFetchOverrideForTesting(null); + assert.equal(await callLogs.waitForCallLogSaves(3_000), true); + await artifactWriter.closeCallLogArtifactWriter(); + usageHistory.clearPendingRequests(); + accountSemaphore.resetAll(); + requestDedup.clearInflight(); + accountFallback.clearModelLock(); + dbCore.resetDbInstance(); + await loggerResource.closeSharedLoggerResource(); + globalThis.fetch = originalFetch; + + if (originalDataDir === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = originalDataDir; + if (originalPluginsDir === undefined) delete process.env.OMNIROUTE_PLUGINS_DIR; + else process.env.OMNIROUTE_PLUGINS_DIR = originalPluginsDir; + + fs.rmSync(testRoot, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("Grok Web rejects an error-only upstream stream before advertising HTTP 200 success", async () => { + const response = await executeStreaming([ + { + error: { + code: "UPSTREAM_PRIVATE_CODE", + message: + "UPSTREAM_PRIVATE_DETAIL Bearer top-secret-token /srv/grok/handler.ts:42\n" + + " at internal (/srv/grok/handler.ts:42:7)", + }, + }, + ]); + + assert.equal(response.status, 502); + assert.match(response.headers.get("Content-Type") ?? "", /application\/json/); + + const body = (await response.json()) as { + error: { message: string; type?: string; code?: string }; + upstream_details?: { error?: { message?: string } }; + }; + assert.equal(body.error.code, "STREAM_EARLY_EOF"); + assert.equal(body.error.type, "stream_early_eof"); + assert.equal(body.upstream_details?.error?.message, "Grok upstream stream failed"); + + const publicBody = JSON.stringify(body); + assert.doesNotMatch(publicBody, /UPSTREAM_PRIVATE/); + assert.doesNotMatch(publicBody, /top-secret-token/); + assert.doesNotMatch(publicBody, /\/srv\/grok/); + assert.doesNotMatch(publicBody, /\bat internal\b/); +}); + +test("Grok Web preserves partial content then rejects with a fixed public error", async () => { + let upstreamCancelCalls = 0; + const response = await executeStreamingBody( + stalledGrokEventStream( + [ + { result: { response: { token: "partial answer" } } }, + { + error: { + code: "UPSTREAM_PRIVATE_CODE", + message: "UPSTREAM_PRIVATE_DETAIL secret=never-public /srv/grok/stream.ts:99", + }, + }, + ], + () => { + upstreamCancelCalls += 1; + } + ) + ); + + assert.equal(response.status, 200); + const { text, error } = await readUntilFailure(response); + assert.ok(error instanceof Error); + assert.equal(error.message, "Grok upstream stream failed"); + const payloads = parseSseData(text) as Array>; + const content = payloads.find((payload) => { + const choices = payload.choices as Array<{ delta?: { content?: string } }> | undefined; + return choices?.[0]?.delta?.content === "partial answer"; + }); + assert.ok(content, "the valid content preceding the upstream failure must be retained"); + + assert.doesNotMatch(text, /UPSTREAM_PRIVATE/); + assert.doesNotMatch(text, /never-public/); + assert.doesNotMatch(text, /\/srv\/grok/); + assert.doesNotMatch(text, /\[Error:/); + assert.doesNotMatch(text, /"finish_reason":"stop"/); + assert.equal(upstreamCancelCalls, 1); +}); + +test("Grok Web converts a reader failure after content into the same safe terminal error", async () => { + const encoder = new TextEncoder(); + const upstreamBody = new ReadableStream({ + start(controller) { + controller.enqueue( + encoder.encode(`${JSON.stringify({ result: { response: { token: "kept" } } })}\n`) + ); + setTimeout(() => { + controller.error( + new Error("READER_PRIVATE_DETAIL Bearer stream-token /srv/grok/reader.ts:12") + ); + }, 0); + }, + }); + + const response = await executeStreamingBody(upstreamBody); + assert.equal(response.status, 200); + const { text, error } = await readUntilFailure(response); + assert.ok(error instanceof Error); + assert.equal(error.message, "Grok upstream stream failed"); + const payloads = parseSseData(text) as Array>; + assert.ok( + payloads.some((payload) => { + const choices = payload.choices as Array<{ delta?: { content?: string } }> | undefined; + return choices?.[0]?.delta?.content === "kept"; + }) + ); + assert.doesNotMatch(text, /READER_PRIVATE/); + assert.doesNotMatch(text, /stream-token/); + assert.doesNotMatch(text, /\/srv\/grok/); + assert.doesNotMatch(text, /"finish_reason":"stop"/); +}); + +test("Grok Web propagates downstream cancellation once without awaiting a stuck upstream", async () => { + const encoder = new TextEncoder(); + let upstreamCancelCalls = 0; + const upstreamBody = new ReadableStream({ + start(controller) { + controller.enqueue( + encoder.encode( + `${JSON.stringify({ result: { response: { token: "cancel-safe partial" } } })}\n` + ) + ); + }, + pull() { + return new Promise(() => {}); + }, + cancel() { + upstreamCancelCalls += 1; + return new Promise(() => {}); + }, + }); + const logMessages: string[] = []; + const recordLog = (tag: string, message: string) => { + logMessages.push(`${tag}: ${message}`); + }; + + const response = await executeStreamingBody(upstreamBody, undefined, { + log: { debug: recordLog, info: recordLog, warn: recordLog, error: recordLog }, + }); + assert.equal(response.status, 200); + assert.ok(response.body); + + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let text = ""; + while (!text.includes("cancel-safe partial")) { + const { done, value } = await reader.read(); + assert.equal(done, false); + if (value) text += decoder.decode(value, { stream: true }); + } + const logCountBeforeCancel = logMessages.length; + + assert.equal(await settlesWithin(reader.cancel("client stopped reading")), true); + assert.equal(await settlesWithin(reader.cancel("duplicate cancel")), true); + await new Promise((resolve) => setImmediate(resolve)); + + assert.equal(upstreamCancelCalls, 1); + assert.doesNotMatch(text, /"finish_reason":"stop"|data: \[DONE\]/); + assert.equal(logMessages.length, logCountBeforeCancel); +}); + +test("chatCore returns a pre-content Grok failure to the outer fallback contract", async () => { + const streamFailures: Array> = []; + let requestSucceeded = false; + const requestBody = { + model: "grok-4.1-fast", + messages: [{ role: "user", content: "fallback proof" }], + stream: true, + }; + + __setTlsFetchOverrideForTesting(async () => ({ + status: 200, + headers: new Headers({ "Content-Type": "application/x-ndjson" }), + text: null, + body: grokEventStream([ + { + error: { + code: "FALLBACK_PRIVATE_CODE", + message: "FALLBACK_PRIVATE_DETAIL secret=never-public /srv/grok/fallback.ts:5", + }, + }, + ]), + })); + + const result = await handleChatCore({ + body: structuredClone(requestBody), + modelInfo: { provider: "grok-web", model: "grok-4.1-fast", extendedContext: false }, + credentials: { apiKey: "sso=test-only-cookie", providerSpecificData: {} }, + connectionId: "grok-stream-error-fallback", + log: { debug() {}, info() {}, warn() {}, error() {} }, + clientRawRequest: { + endpoint: "/v1/chat/completions", + body: structuredClone(requestBody), + headers: new Headers({ accept: "text/event-stream" }), + }, + userAgent: "grok-stream-error-boundary-test", + onRequestSuccess() { + requestSucceeded = true; + }, + onStreamFailure(failure: Record) { + streamFailures.push(failure); + }, + } as never); + + assert.equal(result.success, false); + assert.equal(result.status, 502); + assert.equal(requestSucceeded, false); + assert.deepEqual(streamFailures, []); + + const publicBody = await result.response.text(); + assert.match(publicBody, /Grok upstream stream failed/); + assert.doesNotMatch(publicBody, /FALLBACK_PRIVATE|never-public|\/srv\/grok/); + assert.doesNotMatch(publicBody, /"role":"assistant"|"finish_reason":"stop"/); +}); + +test("chatCore converts a Grok post-content failure into terminal wire error and failed persistence", async () => { + await settingsDb.updateSettings({ call_log_pipeline_enabled: true }); + const streamFailures: Array> = []; + const requestBody = { + model: "grok-4.1-fast", + messages: [{ role: "user", content: "pipeline proof" }], + stream: true, + }; + + __setTlsFetchOverrideForTesting(async () => ({ + status: 200, + headers: new Headers({ "Content-Type": "application/x-ndjson" }), + text: null, + body: grokEventStream([ + { result: { response: { token: "pipeline partial" } } }, + { + error: { + code: "PIPELINE_PRIVATE_CODE", + message: "PIPELINE_PRIVATE_DETAIL secret=never-public /srv/grok/pipeline.ts:7", + }, + }, + ]), + })); + + const result = await handleChatCore({ + body: structuredClone(requestBody), + modelInfo: { provider: "grok-web", model: "grok-4.1-fast", extendedContext: false }, + credentials: { apiKey: "sso=test-only-cookie", providerSpecificData: {} }, + connectionId: "grok-stream-error-boundary", + log: { debug() {}, info() {}, warn() {}, error() {} }, + clientRawRequest: { + endpoint: "/v1/chat/completions", + body: structuredClone(requestBody), + headers: new Headers({ accept: "text/event-stream" }), + }, + userAgent: "grok-stream-error-boundary-test", + onStreamFailure(failure: Record) { + streamFailures.push(failure); + }, + } as never); + + assert.equal(result.success, true); + const wire = await result.response.text(); + assert.match(wire, /"content":"pipeline partial"/); + assert.match(wire, /"finish_reason":"error"/); + assert.match(wire, /"message":"Grok upstream stream failed"/); + assert.match(wire, /"type":"server_error"/); + assert.match(wire, /"code":"server_error"/); + assert.match(wire, /data: \[DONE\]/); + assert.doesNotMatch(wire, /"finish_reason":"stop"/); + assert.doesNotMatch(wire, /PIPELINE_PRIVATE|never-public|\/srv\/grok/); + + assert.equal(streamFailures.length, 1); + assert.deepEqual(streamFailures[0], { + status: 502, + message: "Grok upstream stream failed", + code: "stream_pipeline_error", + type: "stream_error", + }); + + assert.equal(await callLogs.waitForCallLogSaves(3_000), true); + const persisted = await waitFor(async () => { + const rows = await callLogs.getCallLogs({ provider: "grok-web", status: "error", limit: 5 }); + return rows.find((row) => row.connectionId === "grok-stream-error-boundary") ?? null; + }); + assert.ok(persisted, "expected the pipeline failure to be persisted"); + assert.equal(persisted.status, 502); + assert.equal(persisted.error, "Grok upstream stream failed"); + + const detail = await callLogs.getCallLogById(persisted.id); + assert.ok(detail?.pipelinePayloads, "expected failed pipeline payloads in the call log"); + const persistedPayload = JSON.stringify(detail.pipelinePayloads); + assert.match(persistedPayload, /Grok upstream stream failed/); + assert.doesNotMatch(persistedPayload, /PIPELINE_PRIVATE|never-public|\/srv\/grok/); +}); + +test("Grok Web still emits streaming tool calls after delaying the assistant role", async () => { + let upstreamCancelCalls = 0; + const response = await executeStreamingBody( + stalledGrokEventStream( + [ + { + result: { + response: { + modelResponse: { + message: + '{"name":"memory_context_tool","arguments":{"query":"grok"}}', + }, + }, + }, + }, + ], + () => { + upstreamCancelCalls += 1; + } + ), + { + messages: [{ role: "user", content: "search memory" }], + stream: true, + tools: [ + { + type: "function", + function: { + name: "memory_context_tool", + parameters: { type: "object", properties: { query: { type: "string" } } }, + }, + }, + ], + } + ); + + assert.equal(response.status, 200); + const text = await response.text(); + assert.match(text, /"role":"assistant"/); + assert.match(text, /"tool_calls"/); + assert.match(text, /"name":"memory_context_tool"/); + assert.match(text, /"finish_reason":"tool_calls"/); + assert.doesNotMatch(text, /"error"/); + assert.equal(upstreamCancelCalls, 1); +}); diff --git a/tests/unit/grok-web-stream-error-boundary.test.ts b/tests/unit/grok-web-stream-error-boundary.test.ts new file mode 100644 index 0000000000..afc9eb5e37 --- /dev/null +++ b/tests/unit/grok-web-stream-error-boundary.test.ts @@ -0,0 +1,86 @@ +import assert from "node:assert/strict"; +import { spawn } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import test from "node:test"; + +const repoRoot = fileURLToPath(new URL("../../", import.meta.url)); +const fixturePath = fileURLToPath( + new URL("../fixtures/grok-web-stream-error-boundary-child.ts", import.meta.url) +); + +type FixtureResult = { + code: number | null; + signal: NodeJS.Signals | null; + stdout: string; + stderr: string; +}; + +function runFixture(): Promise { + // Keep the parent process pristine: test:unit:fast runs files with + // --test-isolation=none, so repository imports or env/DB mutations here can + // collide with unrelated tests. All stateful coverage lives in the child. + const childEnv: NodeJS.ProcessEnv = { + PATH: process.env.PATH, + NODE_PATH: process.env.NODE_PATH, + LANG: process.env.LANG, + LC_ALL: process.env.LC_ALL, + TZ: process.env.TZ, + TMPDIR: process.env.TMPDIR, + NODE_ENV: "test", + API_KEY_SECRET: "grok-boundary-test-only-secret-with-32-plus-characters", + DISABLE_SQLITE_AUTO_BACKUP: "true", + NO_COLOR: "1", + }; + // An inherited marker makes Node treat this nested --test run as recursive + // and silently skip the fixture instead of executing its seven regressions. + delete childEnv.NODE_TEST_CONTEXT; + + return new Promise((resolve, reject) => { + const child = spawn(process.execPath, ["--import", "tsx/esm", "--test", fixturePath], { + cwd: repoRoot, + env: childEnv, + stdio: ["ignore", "pipe", "pipe"], + }); + let stdout = ""; + let stderr = ""; + let timedOut = false; + + child.stdout.setEncoding("utf8"); + child.stderr.setEncoding("utf8"); + child.stdout.on("data", (chunk: string) => { + stdout += chunk; + }); + child.stderr.on("data", (chunk: string) => { + stderr += chunk; + }); + + const timeout = setTimeout(() => { + timedOut = true; + child.kill("SIGKILL"); + }, 120_000); + + child.once("error", (error) => { + clearTimeout(timeout); + reject(error); + }); + child.once("close", (code, signal) => { + clearTimeout(timeout); + if (timedOut) { + reject(new Error("Grok Web stream error boundary fixture timed out after 120 seconds")); + return; + } + resolve({ code, signal, stdout, stderr }); + }); + }); +} + +test("Grok Web stream error boundary passes in a process-isolated runtime", async () => { + const result = await runFixture(); + const output = `${result.stdout}\n${result.stderr}`; + + assert.equal(result.signal, null, output.slice(-12_000)); + assert.equal(result.code, 0, output.slice(-12_000)); + assert.match(output, /(?:^|\s)tests\s+7(?:\s|$)/m); + assert.match(output, /(?:^|\s)pass\s+7(?:\s|$)/m); + assert.match(output, /(?:^|\s)fail\s+0(?:\s|$)/m); +}); From 627fcba6050b8d6eae5f15790488ba2cbd83869d Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 21:00:14 -0300 Subject: [PATCH 068/143] fix(sse): preserve Perplexity stream failures (#12465) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 14 PRs desta campanha de error-boundary sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **120/120** nos 23 arquivos de teste que os PRs trazem. Um ponto que só apareceu no tree combinado: **#12465 e #12466 criam o mesmo arquivo novo** `open-sse/utils/streamReadiness.ts` (que não existe no tip) com desenhos divergentes de cancelamento — `cancelled` + `releaseLock` imediato num, `readInFlight`/`cancelRequested` com `cancelReader` fire-and-forget no outro. Adotei a versão do #12466, que difere e defere o release do lock para quando a leitura em voo termina, e validei a escolha rodando as suítes dos **dois** PRs contra ela: 21/21 no readiness compartilhado e 22/22 incluindo o boundary do Perplexity. --- ...ENDING-perplexity-stream-error-boundary.md | 1 + open-sse/executors/perplexity-web.ts | 353 ++++++---- open-sse/executors/perplexity-web/protocol.ts | 30 +- open-sse/utils/streamReadiness.ts | 62 +- ...exity-web-stream-error-boundary.fixture.ts | 648 ++++++++++++++++++ ...rplexity-web-stream-error-boundary.test.ts | 90 +++ tests/unit/stream-readiness.test.ts | 53 +- 7 files changed, 1064 insertions(+), 173 deletions(-) create mode 100644 changelog.d/fixes/PENDING-perplexity-stream-error-boundary.md create mode 100644 tests/fixtures/perplexity-web-stream-error-boundary.fixture.ts create mode 100644 tests/unit/perplexity-web-stream-error-boundary.test.ts diff --git a/changelog.d/fixes/PENDING-perplexity-stream-error-boundary.md b/changelog.d/fixes/PENDING-perplexity-stream-error-boundary.md new file mode 100644 index 0000000000..0d9a14a2db --- /dev/null +++ b/changelog.d/fixes/PENDING-perplexity-stream-error-boundary.md @@ -0,0 +1 @@ +- **fix(providers):** Perplexity Web no longer turns upstream stream failures into successful assistant text; pre-content failures remain eligible for fallback, partial output ends with a structured sanitized error, and failed sessions are not persisted diff --git a/open-sse/executors/perplexity-web.ts b/open-sse/executors/perplexity-web.ts index 6d0dc6d859..16ae613c8e 100644 --- a/open-sse/executors/perplexity-web.ts +++ b/open-sse/executors/perplexity-web.ts @@ -17,6 +17,7 @@ import { prepareToolMessages } from "../translator/webTools.ts"; import { buildToolModeResponse } from "./chatgptWebTools.ts"; import { sanitizeErrorMessage } from "../utils/error.ts"; import { buildSessionCookieHeader, mergeRefreshedCookie } from "../utils/nextAuthCookie.ts"; +import { formatTranslatedStreamError } from "../utils/streamErrorFormat.ts"; import { PPLX_SSE_ENDPOINT, PPLX_USER_AGENT, @@ -35,6 +36,8 @@ import { const SESSION_MAX_AGE_MS = 3600_000; const SESSION_MAX_ENTRIES = 200; +const PPLX_STREAM_ERROR_MESSAGE = "Perplexity upstream stream failed"; +const PPLX_STREAM_ERROR_CODE = "PPLX_STREAM_ERROR"; interface SessionEntry { backendUuid: string; @@ -102,155 +105,223 @@ function buildStreamingResponse( signal?: AbortSignal | null ): ReadableStream { const encoder = new TextEncoder(); + const streamAbortController = new AbortController(); + const forwardInputAbort = () => + streamAbortController.abort(signal?.reason ?? "perplexity_request_aborted"); + if (signal?.aborted) forwardInputAbort(); + else signal?.addEventListener("abort", forwardInputAbort, { once: true }); + let inputAbortListenerAttached = Boolean(signal && !signal.aborted); + const removeInputAbortListener = () => { + if (!inputAbortListenerAttached) return; + inputAbortListenerAttached = false; + signal?.removeEventListener("abort", forwardInputAbort); + }; + const abortEventStream = (reason: unknown) => { + removeInputAbortListener(); + if (!streamAbortController.signal.aborted) streamAbortController.abort(reason); + }; + const contentIterator = extractContent(eventStream, streamAbortController.signal)[ + Symbol.asyncIterator + ](); + let fullAnswer = ""; + let respBackendUuid: string | null = null; + let roleEmitted = false; + let finished = false; + let pendingFailure: (Error & { statusCode: number }) | null = null; - return new ReadableStream( - { - async start(controller) { - try { - // Initial role chunk - controller.enqueue( - encoder.encode( - sseChunk({ - id: cid, - object: "chat.completion.chunk", - created, - model, - system_fingerprint: null, - choices: [ - { index: 0, delta: { role: "assistant" }, finish_reason: null, logprobs: null }, - ], - }) - ) - ); + const enqueuePreContentFailure = (controller: ReadableStreamDefaultController) => { + controller.enqueue( + encoder.encode( + formatTranslatedStreamError({ + status: 502, + message: PPLX_STREAM_ERROR_MESSAGE, + type: "upstream_error", + code: PPLX_STREAM_ERROR_CODE, + }) + ) + ); + }; - let fullAnswer = ""; - let respBackendUuid: string | null = null; + const takeAssistantRoleChunk = (): string => { + if (roleEmitted) return ""; + roleEmitted = true; + return sseChunk({ + id: cid, + object: "chat.completion.chunk", + created, + model, + system_fingerprint: null, + choices: [ + { + index: 0, + delta: { role: "assistant" }, + finish_reason: null, + logprobs: null, + }, + ], + }); + }; - for await (const chunk of extractContent(eventStream, signal)) { - if (chunk.backendUuid) respBackendUuid = chunk.backendUuid; + const completeStream = (controller: ReadableStreamDefaultController) => { + if (finished) return; + finished = true; + controller.enqueue( + encoder.encode( + sseChunk({ + id: cid, + object: "chat.completion.chunk", + created, + model, + system_fingerprint: null, + choices: [{ index: 0, delta: {}, finish_reason: "stop", logprobs: null }], + }) + ) + ); + controller.enqueue(encoder.encode("data: [DONE]\n\n")); + sessionStore(history, currentMsg, cleanResponse(fullAnswer), respBackendUuid); + removeInputAbortListener(); + controller.close(); + }; - if (chunk.error) { - controller.enqueue( - encoder.encode( - sseChunk({ - id: cid, - object: "chat.completion.chunk", - created, - model, - system_fingerprint: null, - choices: [ - { - index: 0, - delta: { content: `[Error: ${chunk.error}]` }, - finish_reason: null, - logprobs: null, - }, - ], - }) - ) - ); - break; - } + const failStream = (controller: ReadableStreamDefaultController) => { + if (roleEmitted) { + pendingFailure = Object.assign(new Error(PPLX_STREAM_ERROR_MESSAGE), { + statusCode: 502, + }); + finished = true; + controller.close(); + return; + } + finished = true; + enqueuePreContentFailure(controller); + controller.close(); + }; - if (chunk.thinking) { - controller.enqueue( - encoder.encode( - sseChunk({ - id: cid, - object: "chat.completion.chunk", - created, - model, - system_fingerprint: null, - choices: [ - { - index: 0, - delta: { reasoning_content: chunk.thinking + "\n" }, - finish_reason: null, - logprobs: null, - }, - ], - }) - ) - ); - continue; - } + const providerStream = new ReadableStream({ + async pull(controller) { + if (finished) return; - if (chunk.done) { - fullAnswer = chunk.answer || fullAnswer; - break; - } - - let dt = chunk.delta || ""; - if (dt) { - dt = cleanResponse(dt, false); - if (dt) { - controller.enqueue( - encoder.encode( - sseChunk({ - id: cid, - object: "chat.completion.chunk", - created, - model, - system_fingerprint: null, - choices: [ - { index: 0, delta: { content: dt }, finish_reason: null, logprobs: null }, - ], - }) - ) - ); - } - } - if (chunk.answer) fullAnswer = chunk.answer; - } - - // Stop chunk - controller.enqueue( - encoder.encode( - sseChunk({ - id: cid, - object: "chat.completion.chunk", - created, - model, - system_fingerprint: null, - choices: [{ index: 0, delta: {}, finish_reason: "stop", logprobs: null }], - }) - ) - ); - controller.enqueue(encoder.encode("data: [DONE]\n\n")); - - sessionStore(history, currentMsg, cleanResponse(fullAnswer), respBackendUuid); - } catch (err) { - controller.enqueue( - encoder.encode( - sseChunk({ - id: cid, - object: "chat.completion.chunk", - created, - model, - system_fingerprint: null, - choices: [ - { - index: 0, - delta: { - content: `[Stream error: ${err instanceof Error ? err.message : String(err)}]`, - }, - finish_reason: "stop", - logprobs: null, - }, - ], - }) - ) - ); - controller.enqueue(encoder.encode("data: [DONE]\n\n")); - } finally { - try { - controller.close(); - } catch {} + try { + const next = await contentIterator.next(); + if (streamAbortController.signal.aborted) { + finished = true; + controller.close(); + return; } - }, + if (next.done === true) { + completeStream(controller); + return; + } + + const chunk = next.value; + if (chunk.backendUuid) respBackendUuid = chunk.backendUuid; + + if (chunk.error) { + failStream(controller); + removeInputAbortListener(); + void contentIterator.return?.(undefined).catch(() => undefined); + return; + } + + if (chunk.thinking) { + controller.enqueue( + encoder.encode( + takeAssistantRoleChunk() + + sseChunk({ + id: cid, + object: "chat.completion.chunk", + created, + model, + system_fingerprint: null, + choices: [ + { + index: 0, + delta: { reasoning_content: chunk.thinking + "\n" }, + finish_reason: null, + logprobs: null, + }, + ], + }) + ) + ); + return; + } + + if (chunk.done) { + fullAnswer = chunk.answer || fullAnswer; + completeStream(controller); + await contentIterator.return?.(undefined); + return; + } + + let dt = chunk.delta || ""; + if (dt) { + dt = cleanResponse(dt, false); + if (dt) { + controller.enqueue( + encoder.encode( + takeAssistantRoleChunk() + + sseChunk({ + id: cid, + object: "chat.completion.chunk", + created, + model, + system_fingerprint: null, + choices: [ + { index: 0, delta: { content: dt }, finish_reason: null, logprobs: null }, + ], + }) + ) + ); + } + } + if (chunk.answer) fullAnswer = chunk.answer; + } catch { + failStream(controller); + removeInputAbortListener(); + void contentIterator.return?.(undefined).catch(() => undefined); + } }, - { highWaterMark: 16384 } - ); + + cancel(reason) { + finished = true; + abortEventStream(reason); + void contentIterator.return?.(undefined).catch(() => undefined); + }, + }); + + // Erroring the provider stream immediately would discard output buffered by the readiness + // handoff. Drain each provider chunk through a backpressure-aware reader, then reject only the + // read after the last legitimate chunk. The outer pipeline converts that fixed public error to + // the client's canonical terminal frame and records the stream failure. + const providerReader = providerStream.getReader(); + let cancelled = false; + return new ReadableStream({ + async pull(controller) { + try { + const next = await providerReader.read(); + if (cancelled) return; + if (next.done === false) { + controller.enqueue(next.value); + return; + } + if (pendingFailure) { + controller.error(pendingFailure); + return; + } + controller.close(); + } catch (error) { + if (!cancelled) controller.error(error); + } + }, + + cancel(reason) { + if (cancelled) return; + cancelled = true; + abortEventStream(reason); + void providerReader.cancel(reason).catch(() => undefined); + }, + }); } async function buildNonStreamingResponse( diff --git a/open-sse/executors/perplexity-web/protocol.ts b/open-sse/executors/perplexity-web/protocol.ts index 12e98ccdc4..f3f8a318c5 100644 --- a/open-sse/executors/perplexity-web/protocol.ts +++ b/open-sse/executors/perplexity-web/protocol.ts @@ -213,6 +213,19 @@ export async function* readPplxSseEvents( const decoder = new TextDecoder(); let buffer = ""; let dataLines: string[] = []; + let readerFinished = false; + let readerCancelRequested = false; + + const cancelReader = (reason: unknown) => { + if (readerFinished || readerCancelRequested) return; + readerCancelRequested = true; + // Cancellation is a client-facing latency boundary. Request upstream cleanup once, but never + // await a hostile underlying source whose cancel hook does not settle. + void reader.cancel(reason).catch(() => undefined); + }; + const handleAbort = () => cancelReader(signal?.reason ?? "perplexity_stream_aborted"); + if (signal?.aborted) handleAbort(); + else signal?.addEventListener("abort", handleAbort, { once: true }); function flush(): PplxStreamEvent | null | "done" { if (dataLines.length === 0) return null; @@ -231,7 +244,10 @@ export async function* readPplxSseEvents( while (true) { if (signal?.aborted) return; const { value, done } = await reader.read(); - if (done) break; + if (done) { + readerFinished = true; + break; + } buffer += decoder.decode(value, { stream: true }); while (true) { @@ -263,7 +279,13 @@ export async function* readPplxSseEvents( const tail = flush(); if (tail && tail !== "done") yield tail; } finally { - reader.releaseLock(); + signal?.removeEventListener("abort", handleAbort); + cancelReader(signal?.reason ?? "perplexity_stream_reader_closed"); + try { + reader.releaseLock(); + } catch { + // A hostile source may keep its cancel promise pending; the lock can be released later by GC. + } } } @@ -915,6 +937,10 @@ export async function* extractContent( } } + // Cancellation is not a successful terminal event. In particular, do not synthesize the final + // `done` chunk: streaming callers use that signal to emit stop/[DONE] and persist the session. + if (signal?.aborted) return; + // End-of-stream without a COMPLETED frame still try the last text blob. if (!fullAnswer.trim() && lastEventText) { const fromText = extractAnswerFromFinalText(lastEventText); diff --git a/open-sse/utils/streamReadiness.ts b/open-sse/utils/streamReadiness.ts index 4bbeef1e0e..3774cc0260 100644 --- a/open-sse/utils/streamReadiness.ts +++ b/open-sse/utils/streamReadiness.ts @@ -421,29 +421,57 @@ function prependBufferedChunks( chunks: Uint8Array[], reader: ReadableStreamDefaultReader ): ReadableStream { + let bufferedIndex = 0; + let cancelled = false; + let readerReleased = false; + + const releaseReader = () => { + if (readerReleased) return; + readerReleased = true; + try { + reader.releaseLock(); + } catch { + // A hostile source can keep a read/cancel pending forever. The public stream must + // remain cancellable even when its abandoned source cannot release immediately. + } + }; + return new ReadableStream({ - async start(controller) { + async pull(controller) { + if (cancelled) return; + + // Replay exactly one readiness chunk per pull. Keeping the first buffered chunk at + // the stream's default high-water mark prevents an eager read of a later upstream + // failure from discarding that legitimate prefix before the caller attaches. + if (bufferedIndex < chunks.length) { + controller.enqueue(chunks[bufferedIndex]); + bufferedIndex += 1; + return; + } + try { - for (const chunk of chunks) { - controller.enqueue(chunk); + const { done, value } = await reader.read(); + if (cancelled) return; + if (done) { + releaseReader(); + controller.close(); + return; } - - while (true) { - const { done, value } = await reader.read(); - if (done) break; - if (value) controller.enqueue(value); - } - - controller.close(); + if (value) controller.enqueue(value); } catch (error) { - controller.error(error); - } finally { - reader.releaseLock(); + releaseReader(); + if (!cancelled) controller.error(error); } }, - async cancel(reason) { - await reader.cancel(reason).catch(() => {}); - reader.releaseLock(); + cancel(reason) { + if (cancelled) return; + cancelled = true; + // Do not await a provider's cancel hook: a hostile or stalled source must not make + // downstream cancellation hang. Release the lock once cancellation actually settles. + void reader + .cancel(reason) + .catch(() => {}) + .finally(releaseReader); }, }); } diff --git a/tests/fixtures/perplexity-web-stream-error-boundary.fixture.ts b/tests/fixtures/perplexity-web-stream-error-boundary.fixture.ts new file mode 100644 index 0000000000..8d02adbbad --- /dev/null +++ b/tests/fixtures/perplexity-web-stream-error-boundary.fixture.ts @@ -0,0 +1,648 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +assert.ok(process.env.DATA_DIR, "the subprocess fixture requires an isolated DATA_DIR"); +assert.ok( + process.env.OMNIROUTE_PLUGINS_DIR, + "the subprocess fixture requires an isolated OMNIROUTE_PLUGINS_DIR" +); + +const core = await import("../../src/lib/db/core.ts"); +const { getUsageHistory } = await import("../../src/lib/usage/usageHistory.ts"); +const { waitForCallLogSaves } = await import("../../src/lib/usage/callLogs.ts"); +const { closeCallLogArtifactWriter } = await import("../../src/lib/usage/callLogArtifactWriter.ts"); +const { PerplexityWebExecutor } = await import("../../open-sse/executors/perplexity-web.ts"); +const { __setTlsFetchOverrideForTesting } = + await import("../../open-sse/services/perplexityTlsClient.ts"); +const { handleChatCore } = await import("../../open-sse/handlers/chatCore.ts"); + +type StreamFailure = { status: number; message: string; code?: string; type?: string }; + +function createPerplexityStream( + events: Array> +): ReadableStream { + const encoder = new TextEncoder(); + const payload = + events.map((event) => `event: message\r\ndata: ${JSON.stringify(event)}\r\n\r\n`).join("") + + "event: end_of_stream\r\n\r\n"; + return new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(payload)); + controller.close(); + }, + }); +} + +async function executeWithUpstreamBody( + body: ReadableStream, + prompt = "hi" +): Promise { + __setTlsFetchOverrideForTesting(async () => ({ + status: 200, + headers: new Headers({ "Content-Type": "text/event-stream" }), + text: null, + body, + })); + + const executor = new PerplexityWebExecutor(); + const result = await executor.execute({ + model: "pplx-auto", + body: { messages: [{ role: "user", content: prompt }], stream: true }, + stream: true, + credentials: { apiKey: "test-cookie" }, + signal: AbortSignal.timeout(10_000), + log: null, + }); + return result.response; +} + +function executeStreaming( + events: Array>, + prompt = "hi" +): Promise { + return executeWithUpstreamBody(createPerplexityStream(events), prompt); +} + +async function executeThroughChatCore( + events: Array>, + prompt = "hi", + onStreamFailure?: (failure: StreamFailure) => void, + onRequestSuccess?: () => Promise, + model = "pplx-auto" +) { + return executeBodyThroughChatCore( + createPerplexityStream(events), + prompt, + onStreamFailure, + onRequestSuccess, + model + ); +} + +async function executeBodyThroughChatCore( + upstreamBody: ReadableStream, + prompt = "hi", + onStreamFailure?: (failure: StreamFailure) => void, + onRequestSuccess?: () => Promise, + model = "pplx-auto" +) { + __setTlsFetchOverrideForTesting(async () => ({ + status: 200, + headers: new Headers({ "Content-Type": "text/event-stream" }), + text: null, + body: upstreamBody, + })); + const body = { + model, + messages: [{ role: "user", content: prompt }], + stream: true, + }; + return handleChatCore({ + body: structuredClone(body), + modelInfo: { provider: "perplexity-web", model, extendedContext: false }, + credentials: { apiKey: "test-cookie", providerSpecificData: {} }, + log: { debug() {}, info() {}, warn() {}, error() {} }, + onRequestSuccess, + onStreamFailure, + clientRawRequest: { + endpoint: "/v1/chat/completions", + body: structuredClone(body), + headers: new Headers({ accept: "text/event-stream" }), + }, + userAgent: "perplexity-stream-error-boundary-test", + skipResourcePressureGuard: true, + }); +} + +function assertNoSensitiveDetail(value: string): void { + assert.doesNotMatch(value, /private-runtime\.ts/); + assert.doesNotMatch(value, /sk-pplx-secret/); + assert.doesNotMatch(value, /api_key/); +} + +function assertChatCompletionWire(value: string): void { + assert.doesNotMatch(value, /^event:/m, "Chat Completions must not receive Responses framing"); + const payloads = value + .split(/\r?\n/) + .filter((line) => line.startsWith("data:") && line.slice(5).trim() !== "[DONE]") + .map((line) => JSON.parse(line.slice(5).trim()) as Record); + assert.ok(payloads.length > 0); + for (const payload of payloads) { + assert.equal(payload.object, "chat.completion.chunk"); + assert.ok(Array.isArray(payload.choices)); + for (const choice of payload.choices as Array>) { + assert.equal(choice.index, 0); + assert.equal(typeof choice.delta, "object"); + assert.ok( + choice.finish_reason === null || typeof choice.finish_reason === "string", + "finish_reason must remain Chat Completions-compatible" + ); + } + } + assert.match(value, /data: \[DONE\]/); +} + +async function waitForPersistedStreamFailure(startedAt: Date, model = "pplx-auto") { + for (let attempt = 0; attempt < 80; attempt++) { + const rows = await getUsageHistory({ + provider: "perplexity-web", + model, + startDate: startedAt, + }); + const failure = rows.find( + (row) => + row.success === false && row.status === "502" && row.errorCode === "stream_pipeline_error" + ); + if (failure) return failure; + await new Promise((resolve) => setTimeout(resolve, 25)); + } + return null; +} + +async function waitForCondition(predicate: () => boolean, timeoutMs = 500): Promise { + const deadline = Date.now() + timeoutMs; + while (Date.now() < deadline) { + if (predicate()) return true; + await new Promise((resolve) => setTimeout(resolve, 10)); + } + return predicate(); +} + +test.afterEach(() => { + __setTlsFetchOverrideForTesting(null); +}); + +test.after(async () => { + __setTlsFetchOverrideForTesting(null); + assert.equal( + await waitForCallLogSaves(3_000), + true, + "call-log writes must drain before the isolated DATA_DIR is removed" + ); + await closeCallLogArtifactWriter(); + core.resetDbInstance(); +}); + +test("pre-content Perplexity failures remain unready and return a sanitized 502", async () => { + const result = await executeThroughChatCore([ + { + error_code: "PPLX_ERROR", + error_message: + "failed at /srv/omniroute/private-runtime.ts:42:7 token=sk-pplx-secret-123456 api_key=hidden", + }, + ]); + + assert.equal(result.success, false, "the handler must expose a fallback-eligible failure"); + assert.equal(result.status, 502); + assert.equal(result.response.status, 502); + assert.equal(result.errorCode, "STREAM_EARLY_EOF"); + const body = await result.response.text(); + assert.match(body, /"error"/); + assertNoSensitiveDetail(body); +}); + +test("thrown Perplexity stream failures remain unready and return a sanitized 502", async () => { + const result = await executeBodyThroughChatCore( + new ReadableStream({ + start(controller) { + controller.error( + new Error( + "socket failed at /srv/omniroute/private-runtime.ts:51:9 token=sk-pplx-secret-catch" + ) + ); + }, + }), + "throw before content" + ); + + assert.equal(result.success, false, "the handler must expose a fallback-eligible failure"); + assert.equal(result.status, 502); + assert.equal(result.response.status, 502); + const responseBody = await result.response.text(); + assert.match(responseBody, /"error"/); + assertNoSensitiveDetail(responseBody); +}); + +test("thrown failures after content preserve the prefix and terminate as a safe error", async () => { + const encoder = new TextEncoder(); + const partialAnswer = "partial before transport failure"; + let upstreamRead = false; + let finalizedFailure: StreamFailure | null = null; + const upstreamBody = new ReadableStream({ + pull(controller) { + if (!upstreamRead) { + upstreamRead = true; + controller.enqueue( + encoder.encode( + `event: message\r\ndata: ${JSON.stringify({ + backend_uuid: "uuid-thrown-must-not-store", + blocks: [ + { + intended_usage: "markdown", + markdown_block: { chunks: [partialAnswer], progress: "IN_PROGRESS" }, + }, + ], + status: "PENDING", + })}\r\n\r\n` + ) + ); + return; + } + controller.error( + new Error( + "transport failed at /srv/omniroute/private-runtime.ts:79 token=sk-pplx-secret-after" + ) + ); + }, + }); + + const result = await executeBodyThroughChatCore( + upstreamBody, + "post-content transport failure", + (failure) => { + finalizedFailure = failure; + } + ); + + assert.equal(result.success, true); + const output = await result.response.text(); + assert.match(output, new RegExp(partialAnswer)); + assert.match(output, /"finish_reason":"error"/); + assert.doesNotMatch(output, /"finish_reason":"stop"/); + assert.doesNotMatch(output, /response\.failed/); + assertNoSensitiveDetail(output); + assertChatCompletionWire(output); + assert.ok(finalizedFailure); + assert.equal(finalizedFailure.status, 502); + assert.equal(finalizedFailure.message, "Perplexity upstream stream failed"); +}); + +test("partial content is preserved before a terminal error and the failed session is not stored", async () => { + const firstPrompt = "partial-boundary-first-prompt"; + const partialAnswer = "safe partial answer"; + const requestStartedAt = new Date(Date.now() - 1_000); + let finalizedFailure: StreamFailure | null = null; + const firstResult = await executeThroughChatCore( + [ + { + backend_uuid: "uuid-must-not-be-stored", + blocks: [ + { + intended_usage: "markdown", + markdown_block: { chunks: [partialAnswer], progress: "IN_PROGRESS" }, + }, + ], + status: "PENDING", + }, + { + error_code: "PPLX_ERROR", + error_message: + "later failure at /srv/omniroute/private-runtime.ts:66:2 token=sk-pplx-secret-partial", + }, + ], + firstPrompt, + (failure) => { + finalizedFailure = failure; + } + ); + + assert.equal(firstResult.success, true, "legitimate partial output must satisfy readiness"); + const output = await firstResult.response.text(); + assert.match(output, /"role":"assistant"/); + assert.match(output, new RegExp(partialAnswer)); + assert.match(output, /"error":\{/); + assert.match(output, /"finish_reason":"error"/); + assert.doesNotMatch(output, /"finish_reason":"stop"/); + assert.doesNotMatch(output, /response\.failed/); + assert.doesNotMatch(output, /event:\s*response\.failed/); + assert.doesNotMatch(output, /"type":"response\.failed"/); + assert.doesNotMatch(output, /\[Error:/); + assertNoSensitiveDetail(output); + assertChatCompletionWire(output); + assert.ok(finalizedFailure, "the downstream pipeline must finalize the stream failure"); + assert.equal(finalizedFailure.status, 502); + assert.equal(finalizedFailure.message, "Perplexity upstream stream failed"); + const persistedFailure = await waitForPersistedStreamFailure(requestStartedAt); + assert.ok(persistedFailure, "the handler must persist the terminal stream failure"); + + let followUpRequestBody: string | undefined; + __setTlsFetchOverrideForTesting(async (_url, options) => { + followUpRequestBody = String(options.body ?? ""); + return { + status: 200, + headers: new Headers({ "Content-Type": "text/event-stream" }), + text: null, + body: createPerplexityStream([ + { + backend_uuid: "next-success-uuid", + blocks: [ + { + intended_usage: "markdown", + markdown_block: { chunks: ["next answer"], progress: "DONE" }, + }, + ], + status: "COMPLETED", + }, + ]), + }; + }); + + const executor = new PerplexityWebExecutor(); + const followUp = await executor.execute({ + model: "pplx-auto", + body: { + messages: [ + { role: "user", content: firstPrompt }, + { role: "assistant", content: partialAnswer }, + { role: "user", content: "continue" }, + ], + stream: false, + }, + stream: false, + credentials: { apiKey: "test-cookie" }, + signal: AbortSignal.timeout(10_000), + log: null, + }); + assert.equal(followUp.response.status, 200); + assert.ok(followUpRequestBody); + const sent = JSON.parse(followUpRequestBody) as { params?: Record }; + assert.equal( + sent.params?.last_backend_uuid, + undefined, + "a failed partial response must not create a reusable session" + ); +}); + +test("same-packet content and error preserve the prefix across repeated readiness handoffs", async () => { + for (let attempt = 0; attempt < 8; attempt++) { + const partialAnswer = `same-packet partial ${attempt}`; + let finalizedFailure: StreamFailure | null = null; + const result = await executeThroughChatCore( + [ + { + backend_uuid: `uuid-same-packet-${attempt}`, + blocks: [ + { + intended_usage: "markdown", + markdown_block: { chunks: [partialAnswer], progress: "IN_PROGRESS" }, + }, + ], + status: "PENDING", + }, + { + error_code: "PPLX_ERROR", + error_message: `same packet private failure ${attempt} token=sk-pplx-secret-repeat`, + }, + ], + `same-packet prompt ${attempt}`, + (failure) => { + finalizedFailure = failure; + } + ); + + assert.equal(result.success, true); + const output = await result.response.text(); + assert.match(output, new RegExp(partialAnswer)); + assert.match(output, /"finish_reason":"error"/); + assert.doesNotMatch(output, /"finish_reason":"stop"/); + assert.doesNotMatch(output, /response\.failed/); + assertNoSensitiveDetail(output); + assertChatCompletionWire(output); + assert.ok(finalizedFailure); + assert.equal(finalizedFailure.status, 502); + } +}); + +test("a delayed success hook cannot erase same-packet content before terminal failure", async () => { + const partialAnswer = "prefix must survive delayed success bookkeeping"; + const model = "pplx-auto-delayed-success-proof"; + const requestStartedAt = new Date(Date.now() - 1_000); + let finalizedFailure: StreamFailure | null = null; + const result = await executeThroughChatCore( + [ + { + backend_uuid: "uuid-delayed-success-hook-must-not-store", + blocks: [ + { + intended_usage: "markdown", + markdown_block: { chunks: [partialAnswer], progress: "IN_PROGRESS" }, + }, + ], + status: "PENDING", + }, + { + error_code: "PPLX_ERROR", + error_message: "private same-packet failure token=sk-pplx-secret-delayed-hook", + }, + ], + "delayed success hook prompt", + (failure) => { + finalizedFailure = failure; + }, + async () => { + await new Promise((resolve) => setTimeout(resolve, 25)); + }, + model + ); + + assert.equal(result.success, true, "the legitimate prefix must satisfy readiness"); + const output = await result.response.text(); + assert.match(output, new RegExp(partialAnswer)); + assert.match(output, /"finish_reason":"error"/); + assert.doesNotMatch(output, /"finish_reason":"stop"/); + assert.doesNotMatch(output, /response\.failed/); + assertNoSensitiveDetail(output); + assertChatCompletionWire(output); + assert.ok(finalizedFailure); + assert.equal(finalizedFailure.status, 502); + assert.equal(finalizedFailure.message, "Perplexity upstream stream failed"); + assert.ok( + await waitForPersistedStreamFailure(requestStartedAt, model), + "the delayed handoff must still persist the terminal stream failure" + ); +}); + +test("successful streamed completions still store their Perplexity session", async () => { + const firstPrompt = "successful-session-first-prompt"; + const firstAnswer = "successful session answer"; + const firstResponse = await executeStreaming( + [ + { + backend_uuid: "uuid-success-is-stored", + blocks: [ + { + intended_usage: "markdown", + markdown_block: { chunks: [firstAnswer], progress: "DONE" }, + }, + ], + status: "COMPLETED", + }, + ], + firstPrompt + ); + const firstOutput = await firstResponse.text(); + assert.match(firstOutput, new RegExp(firstAnswer)); + assert.match(firstOutput, /"finish_reason":"stop"/); + + let followUpRequestBody: string | undefined; + __setTlsFetchOverrideForTesting(async (_url, options) => { + followUpRequestBody = String(options.body ?? ""); + return { + status: 200, + headers: new Headers({ "Content-Type": "text/event-stream" }), + text: null, + body: createPerplexityStream([ + { + backend_uuid: "uuid-next-success", + blocks: [ + { + intended_usage: "markdown", + markdown_block: { chunks: ["continued"], progress: "DONE" }, + }, + ], + status: "COMPLETED", + }, + ]), + }; + }); + + const executor = new PerplexityWebExecutor(); + const followUp = await executor.execute({ + model: "pplx-auto", + body: { + messages: [ + { role: "user", content: firstPrompt }, + { role: "assistant", content: firstAnswer }, + { role: "user", content: "continue successful session" }, + ], + stream: false, + }, + stream: false, + credentials: { apiKey: "test-cookie" }, + signal: AbortSignal.timeout(10_000), + log: null, + }); + assert.equal(followUp.response.status, 200); + assert.ok(followUpRequestBody); + const sent = JSON.parse(followUpRequestBody) as { params?: Record }; + assert.equal(sent.params?.last_backend_uuid, "uuid-success-is-stored"); +}); + +test("downstream cancellation reaches a stalled Perplexity reader exactly once", async () => { + const encoder = new TextEncoder(); + const partialAnswer = "cancel after this prefix"; + let firstPull = true; + let upstreamCancelCount = 0; + let finalizedFailure: StreamFailure | null = null; + const upstreamBody = new ReadableStream({ + pull(controller) { + if (firstPull) { + firstPull = false; + controller.enqueue( + encoder.encode( + `event: message\r\ndata: ${JSON.stringify({ + backend_uuid: "uuid-cancel-must-not-store", + blocks: [ + { + intended_usage: "markdown", + markdown_block: { chunks: [partialAnswer], progress: "IN_PROGRESS" }, + }, + ], + status: "PENDING", + })}\r\n\r\n` + ) + ); + return; + } + return new Promise(() => {}); + }, + cancel() { + upstreamCancelCount += 1; + return new Promise(() => {}); + }, + }); + + const result = await executeBodyThroughChatCore( + upstreamBody, + "cancel stalled stream", + (failure) => { + finalizedFailure = failure; + } + ); + assert.equal(result.success, true); + assert.ok(result.response.body); + const reader = result.response.body.getReader(); + const decoder = new TextDecoder(); + let prefix = ""; + for (let readCount = 0; readCount < 4 && !prefix.includes(partialAnswer); readCount += 1) { + const next = await Promise.race([ + reader.read(), + new Promise<"timeout">((resolve) => setTimeout(() => resolve("timeout"), 500)), + ]); + if (next === "timeout" || next.done) break; + prefix += decoder.decode(next.value, { stream: true }); + } + + const cancelResult = await Promise.race([ + reader.cancel("client stopped reading").then(() => "settled"), + new Promise<"timeout">((resolve) => setTimeout(() => resolve("timeout"), 500)), + ]); + assert.match(prefix, new RegExp(partialAnswer)); + assert.doesNotMatch(prefix, /"finish_reason":"stop"/); + assert.doesNotMatch(prefix, /data: \[DONE\]/); + assert.equal(cancelResult, "settled", "client cancellation must not await a hostile upstream"); + assert.equal( + await waitForCondition(() => upstreamCancelCount === 1), + true, + "cancellation must reach the real upstream reader" + ); + assert.equal(upstreamCancelCount, 1); + await new Promise((resolve) => setTimeout(resolve, 25)); + assert.equal(finalizedFailure, null, "client cancellation must not finalize as provider failure"); + + let followUpRequestBody: string | undefined; + __setTlsFetchOverrideForTesting(async (_url, options) => { + followUpRequestBody = String(options.body ?? ""); + return { + status: 200, + headers: new Headers({ "Content-Type": "text/event-stream" }), + text: null, + body: createPerplexityStream([ + { + backend_uuid: "uuid-after-cancel", + blocks: [ + { + intended_usage: "markdown", + markdown_block: { chunks: ["answer after cancel"], progress: "DONE" }, + }, + ], + status: "COMPLETED", + }, + ]), + }; + }); + const executor = new PerplexityWebExecutor(); + const followUp = await executor.execute({ + model: "pplx-auto", + body: { + messages: [ + { role: "user", content: "cancel stalled stream" }, + { role: "assistant", content: partialAnswer }, + { role: "user", content: "continue after cancellation" }, + ], + stream: false, + }, + stream: false, + credentials: { apiKey: "test-cookie" }, + signal: AbortSignal.timeout(10_000), + log: null, + }); + assert.equal(followUp.response.status, 200); + assert.ok(followUpRequestBody); + const sent = JSON.parse(followUpRequestBody) as { params?: Record }; + assert.equal( + sent.params?.last_backend_uuid, + undefined, + "a cancelled response must not create a reusable session" + ); +}); diff --git a/tests/unit/perplexity-web-stream-error-boundary.test.ts b/tests/unit/perplexity-web-stream-error-boundary.test.ts new file mode 100644 index 0000000000..57daec0c15 --- /dev/null +++ b/tests/unit/perplexity-web-stream-error-boundary.test.ts @@ -0,0 +1,90 @@ +import assert from "node:assert/strict"; +import { spawn } from "node:child_process"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import test from "node:test"; + +const FIXTURE = fileURLToPath( + new URL("../fixtures/perplexity-web-stream-error-boundary.fixture.ts", import.meta.url) +); + +type FixtureResult = { + code: number | null; + signal: NodeJS.Signals | null; + stdout: string; + stderr: string; +}; + +function runIsolatedFixture(testRoot: string): Promise { + const dataDir = path.join(testRoot, "data"); + const pluginsDir = path.join(testRoot, "plugins"); + fs.mkdirSync(dataDir, { recursive: true }); + fs.mkdirSync(pluginsDir, { recursive: true }); + const childEnv: NodeJS.ProcessEnv = { + PATH: process.env.PATH, + NODE_PATH: process.env.NODE_PATH, + LANG: process.env.LANG ?? "C.UTF-8", + LC_ALL: process.env.LC_ALL, + TZ: process.env.TZ ?? "UTC", + TMPDIR: process.env.TMPDIR ?? os.tmpdir(), + NODE_ENV: "test", + APP_LOG_TO_FILE: "false", + API_KEY_SECRET: "perplexity-stream-boundary-test-secret-00000000000000000000000000000000", + DATA_DIR: dataDir, + OMNIROUTE_PLUGINS_DIR: pluginsDir, + }; + delete childEnv.NODE_TEST_CONTEXT; + + return new Promise((resolve, reject) => { + const child = spawn(process.execPath, ["--import", "tsx/esm", "--test", FIXTURE], { + cwd: process.cwd(), + env: childEnv, + stdio: ["ignore", "pipe", "pipe"], + timeout: 90_000, + }); + let stdout = ""; + let stderr = ""; + child.stdout.setEncoding("utf8").on("data", (chunk: string) => { + stdout += chunk; + }); + child.stderr.setEncoding("utf8").on("data", (chunk: string) => { + stderr += chunk; + }); + child.once("error", reject); + child.once("close", (code, signal) => resolve({ code, signal, stdout, stderr })); + }); +} + +test("Perplexity stream failures preserve protocol semantics in an isolated full pipeline", async () => { + const testRoot = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-pplx-boundary-parent-")); + const originalDataDir = process.env.DATA_DIR; + const originalPluginsDir = process.env.OMNIROUTE_PLUGINS_DIR; + const eventBusOwner = globalThis as { __omnirouteEventBus?: unknown }; + const originalEventBus = eventBusOwner.__omnirouteEventBus; + + try { + const result = await runIsolatedFixture(testRoot); + assert.equal( + result.code, + 0, + `isolated Perplexity fixture failed (signal=${result.signal ?? "none"})\n` + + `stdout:\n${result.stdout}\nstderr:\n${result.stderr}` + ); + assert.equal(result.signal, null); + assert.match(result.stdout, /ℹ tests 8/); + assert.match(result.stdout, /ℹ pass 8/); + assert.match(result.stdout, /ℹ fail 0/); + + assert.equal(process.env.DATA_DIR, originalDataDir); + assert.equal(process.env.OMNIROUTE_PLUGINS_DIR, originalPluginsDir); + assert.equal( + eventBusOwner.__omnirouteEventBus, + originalEventBus, + "the subprocess fixture must not replace the parent event bus singleton" + ); + } finally { + fs.rmSync(testRoot, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + } +}); diff --git a/tests/unit/stream-readiness.test.ts b/tests/unit/stream-readiness.test.ts index b2ea196818..7e2ea9c242 100644 --- a/tests/unit/stream-readiness.test.ts +++ b/tests/unit/stream-readiness.test.ts @@ -451,6 +451,43 @@ test("ensureStreamReadiness preserves buffered chunks when stream starts", async assert.match(text, / world/); }); +test("ensureStreamReadiness preserves its buffered prefix until a delayed consumer observes a later error", async () => { + const prefix = `data: ${JSON.stringify({ + object: "chat.completion.chunk", + choices: [ + { + index: 0, + delta: { role: "assistant", content: "prefix before failure" }, + finish_reason: null, + }, + ], + })}\n\n`; + let pullCount = 0; + const response = new Response( + new ReadableStream({ + pull(controller) { + pullCount += 1; + if (pullCount === 1) { + controller.enqueue(encoder.encode(prefix)); + return; + } + controller.error(new Error("later upstream failure")); + }, + }), + { status: 200, headers: { "Content-Type": "text/event-stream" } } + ); + + const result = await ensureStreamReadiness(response, { timeoutMs: 100 }); + assert.equal(result.ok, true); + await new Promise((resolve) => setTimeout(resolve, 25)); + + const reader = result.response.body!.getReader(); + const first = await reader.read(); + assert.equal(first.done, false); + assert.match(new TextDecoder().decode(first.value), /prefix before failure/); + await assert.rejects(() => reader.read(), /later upstream failure/); +}); + test("ensureStreamReadiness honors configured timeouts above 2000ms", async () => { const response = new Response( streamFromChunks( @@ -616,10 +653,7 @@ test("ensureStreamReadiness preserves sanitized error-only diagnostics on early assert.equal(result.response.status, 502); assert.equal(result.code, "STREAM_EARLY_EOF"); assert.equal(result.type, "stream_early_eof"); - assert.equal( - result.classificationReason, - "Stream ended before producing a non-ping SSE event" - ); + assert.equal(result.classificationReason, "Stream ended before producing a non-ping SSE event"); assert.equal( result.upstreamDiagnostic, "UPSTREAM_DETAIL quota exhausted; retry after 2s; empty content Bearer [REDACTED] " @@ -636,16 +670,9 @@ test("ensureStreamReadiness preserves sanitized error-only diagnostics on early assert.equal(body.upstream_details.error.message, result.upstreamDiagnostic); assert.equal(warnings.length, 1); - for (const surfaced of [ - result.reason, - body.upstream_details.error.message, - warnings[0], - ]) { + for (const surfaced of [result.reason, body.upstream_details.error.message, warnings[0]]) { assert.match(surfaced, /UPSTREAM_DETAIL/); - assert.doesNotMatch( - surfaced, - /SECOND_DETAIL|TOP_SECRET|\/srv\/omniroute\/handler\.ts/ - ); + assert.doesNotMatch(surfaced, /SECOND_DETAIL|TOP_SECRET|\/srv\/omniroute\/handler\.ts/); } }); From 7ae8bf4e052ba19d5a9cecd3298fdeca3610cb7b Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 21:01:09 -0300 Subject: [PATCH 069/143] fix(db): harden migration recovery snapshots (#12435) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 14 PRs desta campanha de error-boundary sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **120/120** nos 23 arquivos de teste que os PRs trazem. Um ponto que só apareceu no tree combinado: **#12465 e #12466 criam o mesmo arquivo novo** `open-sse/utils/streamReadiness.ts` (que não existe no tip) com desenhos divergentes de cancelamento — `cancelled` + `releaseLock` imediato num, `readInFlight`/`cancelRequested` com `cancelReader` fire-and-forget no outro. Adotei a versão do #12466, que difere e defere o release do lock para quando a leitura em voo termina, e validei a escolha rodando as suítes dos **dois** PRs contra ela: 21/21 no readiness compartilhado e 22/22 incluindo o boundary do Perplexity. --- .env.example | 17 +- changelog.d/fixes/migration-151-152-safety.md | 1 + docs/guides/DOCKER_GUIDE.md | 2 +- docs/reference/ENVIRONMENT.md | 18 +- src/lib/dataPaths.ts | 49 +- src/lib/db/backup.ts | 49 +- src/lib/db/backupRetention.ts | 17 +- src/lib/db/core.ts | 10 +- src/lib/db/migrationRunner.ts | 814 ++++++++--------- src/lib/db/migrationRunner/constants.ts | 8 + src/lib/db/migrationRunner/logger.ts | 13 + .../db/migrationRunner/preMigrationBackup.ts | 293 ++++++ src/lib/db/migrationRunner/schemaState.ts | 248 ++++++ .../datadir-test-context-guard-10428.test.ts | 96 +- tests/unit/db-backup-extended.test.ts | 26 + tests/unit/db-fresh-setup-9934.test.ts | 28 + ...-migration-missing-physical-schema.test.ts | 839 ++++++++++++++++++ ...db-migrationrunner-constants-split.test.ts | 15 +- ...e-migration-backup-retention-10421.test.ts | 385 ++++---- 19 files changed, 2303 insertions(+), 625 deletions(-) create mode 100644 changelog.d/fixes/migration-151-152-safety.md create mode 100644 src/lib/db/migrationRunner/logger.ts create mode 100644 src/lib/db/migrationRunner/preMigrationBackup.ts create mode 100644 src/lib/db/migrationRunner/schemaState.ts create mode 100644 tests/unit/db-migration-missing-physical-schema.test.ts diff --git a/.env.example b/.env.example index 9527957b3d..470a6107d6 100644 --- a/.env.example +++ b/.env.example @@ -55,10 +55,11 @@ INITIAL_PASSWORD=CHANGEME # loader (bin/cli/plugins.mjs) at a package tree — this one drives the server-side scanner. # OMNIROUTE_PLUGINS_DIR=/opt/omniroute/plugins -# Escape hatch for the test-context DATA_DIR guard (#10428). A test run that never -# chose a DATA_DIR is redirected to a throwaway temp dir so it cannot open the -# operator's real database. Set to 1 only for a deliberate run against the real -# DATA_DIR — never for CI. Used by: src/lib/dataPaths.ts +# Escape hatch for the test/eval DATA_DIR guard (#10428). A test or node eval/print +# probe (-e/--eval/-p/--print, including --eval=/--print=) that never chose a DATA_DIR +# is redirected to a throwaway temp dir so it cannot open the operator's real database. +# Set to 1 only for a deliberate run against the real DATA_DIR — never for CI. +# Used by: src/lib/dataPaths.ts # OMNIROUTE_ALLOW_DEFAULT_DATA_DIR=1 # Build provenance (#10427). OMNIROUTE_BUILD_SHA lets a container inject the artifact's git @@ -96,9 +97,11 @@ STORAGE_ENCRYPTION_KEY= # Default: v1 | Increment when rotating STORAGE_ENCRYPTION_KEY. STORAGE_ENCRYPTION_KEY_VERSION=v1 -# Automatic SQLite backup on startup. -# Used by: src/lib/db/backup.ts — creates a timestamped backup before migrations. -# Default: false (backups enabled) | Set true to skip backup on every restart. +# Routine/pre-write SQLite backups. +# Used by: src/lib/db/backup.ts. Set true only when those backups are managed externally. +# This never disables the migration runner's mandatory, content-addressed safety snapshot +# or its mass-migration guard for an existing persistent database. +# Default: false (routine backups enabled). DISABLE_SQLITE_AUTO_BACKUP=false # ── Redis (Rate Limiting) ── diff --git a/changelog.d/fixes/migration-151-152-safety.md b/changelog.d/fixes/migration-151-152-safety.md new file mode 100644 index 0000000000..752290573e --- /dev/null +++ b/changelog.d/fixes/migration-151-152-safety.md @@ -0,0 +1 @@ +- Harden SQLite upgrades around the historical migration-074 version collision: missing discovery and inspector tables are replayed atomically, pre-existing databases (including setup-created skeletons) receive reusable content-addressed safety snapshots, and Node test/eval probes without `DATA_DIR` are isolated from the operator database. diff --git a/docs/guides/DOCKER_GUIDE.md b/docs/guides/DOCKER_GUIDE.md index 9c2b536ab6..7d5137f6f9 100644 --- a/docs/guides/DOCKER_GUIDE.md +++ b/docs/guides/DOCKER_GUIDE.md @@ -613,7 +613,7 @@ In-process density (compression off the HTTP isolate) is [#11023](https://github ## Important Notes - **SQLite WAL Mode:** `docker stop` should be allowed to finish so OmniRoute can checkpoint the latest changes back into `storage.sqlite`. The bundled Compose files already set a 40s stop grace period. If you run the image directly, keep `--stop-timeout 40`. -- **`DISABLE_SQLITE_AUTO_BACKUP`:** Set to `true` if backups are managed externally. +- **`DISABLE_SQLITE_AUTO_BACKUP`:** Set to `true` if routine/pre-write backups are managed externally. Existing-database migrations still require their own durable safety snapshot and mass-migration guard. - **Data Persistence:** Always mount a volume to `/app/data` to persist your database, keys, and configurations across container restarts. - **Port Configuration:** Override `PORT` environment variable to change the default `20128` port. diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index 056a37beb9..27bf863d21 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -86,7 +86,7 @@ OmniRoute uses **SQLite** (via `better-sqlite3`) for all persistence. These vari | Variable | Default | Source File | Description | | -------------------------------------- | -------------------- | ----------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | | `DATA_DIR` | `~/.omniroute/` | `src/lib/db/core.ts` | Root directory for SQLite DB, backups, and data files. Override for Docker volumes or custom paths. | -| `OMNIROUTE_ALLOW_DEFAULT_DATA_DIR` | _(unset)_ | `src/lib/dataPaths.ts` | Escape hatch for the test-context DATA_DIR guard (#10428). Test runs with no `DATA_DIR` are redirected to a throwaway temp dir so they cannot open the operator's real database; set to `1` to opt back in to the real directory. | +| `OMNIROUTE_ALLOW_DEFAULT_DATA_DIR` | _(unset)_ | `src/lib/dataPaths.ts` | Escape hatch for the test/eval DATA_DIR guard (#10428). Tests and Node eval/print probes (`-e`/`--eval`/`-p`/`--print`, including `--eval=`/`--print=` forms) with no `DATA_DIR` are redirected to a throwaway temp dir so they cannot open the operator's real database; set to `1` to opt back in to the real directory. | | `OMNIROUTE_BUILD_SHA` | _(unset)_ | `src/lib/monitoring/buildSha.ts` | Git SHA of the running artifact. Stamped by `npm run build:release`; injectable in containers that ship without the `dist/BUILD_SHA` sentinel. Surfaced as `system.buildSha` on `/api/monitoring/health`. | | `OMNIROUTE_RELEASE_REF` | `origin/main` | `scripts/build/buildProvenance.ts` | Ref the pack-artifact provenance gate checks the build SHA against (#10427). | | `OMNIROUTE_ALLOW_CANARY_BUILD` | _(unset)_ | `scripts/build/buildProvenance.ts` | Set to `1` to allow packing a build whose SHA is not on the release line, recording it as a deliberate canary instead of failing the gate (#10427). | @@ -97,7 +97,7 @@ OmniRoute uses **SQLite** (via `better-sqlite3`) for all persistence. These vari | `OMNIROUTE_PLUGINS_DIR` | _(unset)_ | `src/lib/plugins/scanner.ts` | Directory the **runtime plugin scanner** reads — and the root the plugin manager installs into — overriding the home-derived default (#11827). Point it at the bind-mounted plugin tree in Docker/K8s instead of moving HOME just to relocate the scan path (HOME governs every other home-relative behaviour too). Unset = `~/.omniroute/plugins`, or `/tmp/.omniroute/plugins` when the process exports no home at all — the silent non-discovery this variable removes. The resolved directory is logged once at startup as `scanner.dir_resolved` with the input that won. Server-side only: CLI command plugins keep their own `OMNIROUTE_PLUGIN_PATH` (section 9). | | `STORAGE_ENCRYPTION_KEY` | _(empty = disabled)_ | `src/lib/db/encryption.ts` | AES key for full SQLite database encryption at rest. Generate with `openssl rand -hex 32`. | | `STORAGE_ENCRYPTION_KEY_VERSION` | `v1` | `scripts/build/bootstrap-env.mjs`, `electron/main.js` | Version label for the encryption key. Increment when performing key rotation to support decryption of old backups. | -| `DISABLE_SQLITE_AUTO_BACKUP` | `false` | `src/lib/db/backup.ts` | When `true`, skips automatic + pre-write SQLite file backups (startup, models.dev pricing save/clear, settings writes). Manual and pre-restore backups still run. Non-manual backups are also **throttled to at most once per 60 minutes** so hourly models.dev sync does not copy the whole DB on every pricing write. Dashboard **Settings → Storage** can disable auto-backup independently. | +| `DISABLE_SQLITE_AUTO_BACKUP` | `false` | `src/lib/db/backup.ts` | When `true`, skips routine/pre-write SQLite file backups (models.dev pricing save/clear, settings writes). Manual and pre-restore backups still run. It does **not** disable the migration runner's mandatory durable safety snapshot or mass-migration guard for an existing persistent DB. Non-manual backups are throttled to at most once per 60 minutes. Dashboard **Settings → Storage** can disable routine auto-backup independently. | | `OMNIROUTE_CRYPT_KEY` | _(unset)_ | `src/lib/db/encryption.ts` | **Legacy alias** for `STORAGE_ENCRYPTION_KEY`. Accepted as a fallback when the primary variable is absent. | | `OMNIROUTE_API_KEY_BASE64` | _(unset)_ | `src/lib/db/encryption.ts` | **Legacy alias** (Base64-encoded form) accepted as a fallback. Decoded automatically before use. | | `OMNIROUTE_DB_HEALTHCHECK_INTERVAL_MS` | _(unset)_ | `src/lib/db/core.ts` | Override the periodic SQLite healthcheck interval (ms). When unset, defaults are derived from `NODE_ENV`. | @@ -121,6 +121,16 @@ OmniRoute uses **SQLite** (via `better-sqlite3`) for all persistence. These vari | `BATCH_BACKOFF_MAX_MS` | `3600000` (1h) | `open-sse/services/batchProcessor.ts` | Cap (ms) for exponential backoff between batch item retries. | | `BATCH_MAX_CONCURRENT` | `1` | `open-sse/services/batchProcessor.ts` | Maximum number of batches processed concurrently. Raise to increase throughput; keep low to avoid rate-limit storms. | +> [!IMPORTANT] +> Before changing an existing persistent database, the migration runner publishes a complete, +> content-addressed snapshot under `DATA_DIR/db_backups/`. Publication requires a filesystem +> that supports same-filesystem, no-overwrite hard links plus durable file sync. POSIX hosts also +> require directory sync; on Windows, Node may reject directory handles, so OmniRoute flushes the +> published file and treats directory-entry sync as best effort. +> If the mounted `DATA_DIR` cannot provide those guarantees, startup fails closed before applying +> a migration. Move `DATA_DIR` to a volume with those primitives; do not use +> `DISABLE_SQLITE_AUTO_BACKUP` to bypass migration safety. + ### Scenarios | Scenario | Configuration | @@ -1321,8 +1331,8 @@ Provider quota endpoints, network tunnels (Tailscale, Ngrok, MITM debug proxy), | `TAILSCALED_BIN` | _(auto-detect)_ | `src/lib/tailscaleTunnel.ts` | Explicit path to the `tailscaled` daemon binary. | | `TAILSCALE_AUTHKEY` | _(unset)_ | `src/lib/tailscaleTunnel.ts` | Pre-shared Tailscale auth key for non-interactive / headless `tailscale up` (passed via `--auth-key=`). When unset, login falls back to the interactive browser auth URL. | | `NGROK_AUTHTOKEN` | _(unset)_ | `src/lib/ngrokTunnel.ts` | Authenticates outbound ngrok tunnels. | -| `DB_BACKUP_MAX_FILES` | `20` | `src/lib/db/backup.ts`, `src/lib/db/migrationRunner.ts` | Maximum SQLite backup files retained on disk. Applies to manual/scheduled backups and to pre-migration snapshots. Overrides the value saved from Settings → Database backup retention. | -| `DB_BACKUP_RETENTION_DAYS` | `0` | `src/lib/db/backup.ts`, `src/lib/db/migrationRunner.ts` | Maximum age (days) of retained backups. `0` disables age-based pruning. Applies to manual/scheduled backups and to pre-migration snapshots. Overrides the value saved from Settings → Database backup retention. | +| `DB_BACKUP_MAX_FILES` | `20` | `src/lib/db/backup.ts` | Maximum SQLite backup files retained by manual/scheduled backup cleanup. Migration snapshots are content-addressed and reused for an identical DB state; they are not pruned inside the concurrent migration window. Overrides the value saved from Settings → Database backup retention. | +| `DB_BACKUP_RETENTION_DAYS` | `0` | `src/lib/db/backup.ts` | Maximum age (days) retained by manual/scheduled backup cleanup. `0` disables age-based pruning. Migration snapshots are not pruned inside the concurrent migration window. Overrides the value saved from Settings → Database backup retention. | | `OMNIROUTE_BACKUP_SCHEDULE_JOB_INTERVAL_MS` | `30000` | `src/lib/jobs/backupScheduleJob.ts` | Tick interval (ms) of the server-side job that executes `backup-schedule.json`. Must stay well under the 1-minute cron granularity; values below `5000` or unparseable fall back to `30000`. | | `CONTAINER_HOST` | `docker` | `scripts/check-permissions.sh` | Container runtime hint for the entrypoint permission check. Set to `podman` for any Podman topology. Because the container cannot determine whether the engine is local or reached through Podman Machine, the warning stays topology-neutral and points to `contrib/podman/README.md`. | | `QUOTA_STORE_DRIVER` | `sqlite` | `src/lib/quota/storeFactory.ts` | Quota-share consumption store backend: `sqlite` (default) or `redis`. | diff --git a/src/lib/dataPaths.ts b/src/lib/dataPaths.ts index 01c93c2bad..b4b9801501 100644 --- a/src/lib/dataPaths.ts +++ b/src/lib/dataPaths.ts @@ -101,30 +101,70 @@ export function isTestContext(): boolean { ); } +/** + * `node --eval` / `node -e` (and their print variants) are common shapes used by + * one-off import probes. + * Such a process has no application entry point from which to establish storage intent, + * so defaulting it to the operator's durable database is unsafe. A deliberate production + * inspection can still opt in with an explicit DATA_DIR (preferred) or + * OMNIROUTE_ALLOW_DEFAULT_DATA_DIR=1. + */ +function isEvalProbeContext(): boolean { + return process.execArgv.some( + (arg) => + arg === "--eval" || + arg === "-e" || + arg === "-pe" || + arg === "-ep" || + arg.startsWith("--eval=") || + arg === "--print" || + arg === "-p" || + arg.startsWith("--print=") + ); +} + /** Process-wide redirect target, so repeated calls share one DB instead of one per call. */ let testContextDataDir: string | null = null; +let testContextCleanupRegistered = false; export function resolveWritableDataDir({ isCloud = false }: { isCloud?: boolean } = {}): string { const resolved = resolveDataDir({ isCloud }); + const configured = normalizeConfiguredPath(process.env.DATA_DIR); // Cloud/serverless never owns a writable home dir; leave its sentinel alone. if (isCloud) return resolved; - // #10428: a test/ad-hoc run that never chose a DATA_DIR would otherwise open the + // #10428: a test/eval-probe run that never chose a DATA_DIR would otherwise open the // OPERATOR'S REAL database (~/.omniroute/storage.sqlite — live provider credentials). // Redirect to a throwaway dir instead of throwing: the documented single-file command // (`node --import tsx/esm --test tests/unit/x.test.ts`) does not load the isolation // setup, and a hard failure there would only teach people to disable the guard. // `OMNIROUTE_ALLOW_DEFAULT_DATA_DIR=1` opts back in, so the intent is recorded. if ( - !process.env.DATA_DIR && - isTestContext() && + !configured && + (isTestContext() || isEvalProbeContext()) && process.env.OMNIROUTE_ALLOW_DEFAULT_DATA_DIR !== "1" ) { if (!testContextDataDir) { testContextDataDir = fs.mkdtempSync(path.join(os.tmpdir(), `${APP_NAME}-testctx-`)); + if (!testContextCleanupRegistered) { + testContextCleanupRegistered = true; + process.once("exit", () => { + if (!testContextDataDir) return; + try { + fs.rmSync(testContextDataDir, { + recursive: true, + force: true, + maxRetries: 5, + retryDelay: 25, + }); + } catch { + // An unclean exit is left to the operating system's temp-directory policy. + } + }); + } console.warn( - `[DATA_DIR] test context without DATA_DIR → using '${testContextDataDir}' instead of ` + + `[DATA_DIR] test/eval context without DATA_DIR → using '${testContextDataDir}' instead of ` + `'${resolved}'. Set DATA_DIR explicitly (or load tests/_setup/isolateDataDir.ts) to silence this.` ); } @@ -132,7 +172,6 @@ export function resolveWritableDataDir({ isCloud = false }: { isCloud?: boolean } // No explicit override → already the default user dir; nothing to fall back to. - const configured = normalizeConfiguredPath(process.env.DATA_DIR); if (!configured) return resolved; try { diff --git a/src/lib/db/backup.ts b/src/lib/db/backup.ts index effbaf09c0..d170a5b764 100644 --- a/src/lib/db/backup.ts +++ b/src/lib/db/backup.ts @@ -101,6 +101,24 @@ function getBackupDir() { return DB_BACKUPS_DIR || path.join(DATA_DIR, "db_backups"); } +function listBackupFilesNewestFirst(backupDir: string) { + return fs + .readdirSync(backupDir) + .filter((filename) => filename.startsWith("db_") && filename.endsWith(".sqlite")) + .flatMap((filename) => { + try { + return [{ filename, stat: fs.statSync(path.join(backupDir, filename)) }]; + } catch { + // A concurrent retention pass may remove an entry after readdir. + return []; + } + }) + .sort( + (left, right) => + right.stat.mtimeMs - left.stat.mtimeMs || right.filename.localeCompare(left.filename) + ); +} + export function cleanupDbBackups(options?: { maxFiles?: number; retentionDays?: number; @@ -272,16 +290,26 @@ export function backupDbFile(reason = "auto") { if (reason !== "manual" && reason !== "pre-restore") { // Shrink detection is useful for automatic safety backups, but it should // never block an explicit operator action like manual backup or pre-restore. + // Only timestamp-named automatic/manual backups are shrink baselines. The + // content-addressed migration snapshots are restore points, not periodic size + // samples; excluding them also keeps this lookup to names only with a single stat + // even in legacy directories containing tens of thousands of timestamp backups. const existingBackups = fs .readdirSync(backupDir) - .filter((f) => f.startsWith("db_") && f.endsWith(".sqlite")) + .filter((filename) => /^db_\d{4}-.*\.sqlite$/.test(filename)) .sort(); if (existingBackups.length > 0) { - const latestBackup = existingBackups[existingBackups.length - 1]; - const latestStat = fs.statSync(path.join(backupDir, latestBackup)); - if (latestStat.size > 4096 && stat.size < latestStat.size * 0.5) { - console.warn(`[DB] Backup SKIPPED — DB shrank from ${latestStat.size}B to ${stat.size}B`); - return null; + const latestBackup = existingBackups.at(-1)!; + try { + const latestStat = fs.statSync(path.join(backupDir, latestBackup)); + if (latestStat.size > 4096 && stat.size < latestStat.size * 0.5) { + console.warn( + `[DB] Backup SKIPPED — DB shrank from ${latestStat.size}B to ${stat.size}B` + ); + return null; + } + } catch (error: unknown) { + if ((error as NodeJS.ErrnoException | null)?.code !== "ENOENT") throw error; } } } @@ -316,16 +344,11 @@ export async function listDbBackups() { try { if (!fs.existsSync(backupDir)) return []; - const entries = fs - .readdirSync(backupDir) - .filter((f) => f.startsWith("db_") && f.endsWith(".sqlite")) - .sort() - .reverse(); + const entries = listBackupFilesNewestFirst(backupDir); const { tryOpenSync } = await import("@/lib/db/adapters/driverFactory"); - return entries.map((filename) => { + return entries.map(({ filename, stat }) => { const filePath = path.join(backupDir, filename); - const stat = fs.statSync(filePath); const match = filename.match(/^db_(.+?)_([^.]+)\.sqlite$/); const reason = match ? match[2] : "unknown"; diff --git a/src/lib/db/backupRetention.ts b/src/lib/db/backupRetention.ts index cbc9efeaa5..150bfe07d7 100644 --- a/src/lib/db/backupRetention.ts +++ b/src/lib/db/backupRetention.ts @@ -1,17 +1,12 @@ /** * Backup retention primitives — pure filesystem work, no `core.ts` dependency. * - * This module exists so BOTH backup call sites can share one retention policy: - * - * - `backup.ts` (manual/API/auto backups) — resolves the operator's settings from the - * database and delegates here. - * - `migrationRunner.ts` (pre-migration snapshots) — cannot import `backup.ts`, because - * `core.ts` already imports `migrationRunner.ts` and `backup.ts` imports `core.ts`; - * that edge would close a cycle. Keeping the policy here, free of `core`, lets the - * migration path prune without one. - * - * Before #10421 the migration path had no retention at all and `db_backups/` grew - * without bound (observed: 48.999 files / 204 GB against a 5,3 MB live database). + * `backup.ts` (manual/API/auto backups) resolves the operator's settings from the + * database and delegates pure family pruning here. The migration runner deliberately + * does not prune during its concurrent safety window: its snapshots are content-addressed + * and reused for an identical DB state, while manual/scheduled cleanup remains the single + * retention boundary. Before #10421, repeated failed starts created distinct timestamped + * snapshots and `db_backups/` grew without bound (observed: 48,999 files / 204 GB). */ import fs from "fs"; diff --git a/src/lib/db/core.ts b/src/lib/db/core.ts index d636899a32..c8ba9b1f18 100644 --- a/src/lib/db/core.ts +++ b/src/lib/db/core.ts @@ -1118,10 +1118,10 @@ export function getDbInstance(): SqliteDatabase { // This is needed so the migration runner skips the mass-migration safety abort // that would otherwise trigger because heuristic seeding marks some migrations // as applied, making the fresh DB look like a wiped existing DB (#1328). - // #9934: also classify as fresh a file that `omniroute setup` created with - // only the clipped skeleton schema (see the probe below) — even though the - // file exists, it has never had migrations run. - let isNewDb = !fs.existsSync(sqliteFile); + // #9934: also classify a setup-created skeleton as logically fresh for the mass guard, + // while tracking its pre-existing file independently for mandatory snapshot safety. + const databaseExistedBeforeInitialization = fs.existsSync(sqliteFile); + let isNewDb = !databaseExistedBeforeInitialization; // Detect and handle old schema format — preserve data when possible (#146) // Uses a single probe connection that becomes the real connection when possible. @@ -1310,7 +1310,7 @@ export function getDbInstance(): SqliteDatabase { VALUES ('001', 'initial_schema'); `); - runMigrations(db, { isNewDb }); + runMigrations(db, { isNewDb, databaseExistedBeforeInitialization }); // Fresh installs need the same post-migration index guarantee as upgraded // databases, including recovery from an interrupted migration 127 attempt. ensureUsageHistoryAccountIndex(db); diff --git a/src/lib/db/migrationRunner.ts b/src/lib/db/migrationRunner.ts index f53cd58070..58073518b8 100644 --- a/src/lib/db/migrationRunner.ts +++ b/src/lib/db/migrationRunner.ts @@ -21,37 +21,29 @@ import type { SqliteAdapter } from "./adapters/types"; import { DEFAULT_DATABASE_SETTINGS } from "@/types/databaseSettings"; import { isAutomatedTestProcess } from "@/shared/utils/testProcess"; import { - RENAMED_MIGRATION_COMPATIBILITY, LEGACY_VERSION_SLOT_MIGRATIONS, - SUPERSEDED_DUPLICATE_MIGRATIONS, - PHYSICAL_SCHEMA_SENTINELS, - INITIAL_SCHEMA_SENTINELS, OPTIONAL_FTS5_MIGRATION_VERSIONS, + RENAMED_MIGRATION_COMPATIBILITY, + SUPERSEDED_DUPLICATE_MIGRATIONS, } from "./migrationRunner/constants"; import { getExtraMigrationFiles } from "./migrationRunner/extraDirs"; -// Retention primitives live in their own `core`-free module: `core.ts` imports this file, -// so importing `backup.ts` (which imports `core.ts`) here would close a dependency cycle. +import { migrationConsole as console } from "./migrationRunner/logger"; import { - MAX_DB_BACKUPS, - DEFAULT_DB_BACKUP_RETENTION_DAYS, - parsePositiveInt, - parseNonNegativeInt, - pruneBackupDirectory, -} from "./backupRetention"; - -const isNodeTestRunnerChild = typeof process.env.NODE_TEST_CONTEXT === "string"; - -const console = { - log: (...args: unknown[]) => { - if (!isNodeTestRunnerChild) globalThis.console.log(...args); - }, - warn: (...args: unknown[]) => { - if (!isNodeTestRunnerChild) globalThis.console.warn(...args); - }, - error: (...args: unknown[]) => { - globalThis.console.error(...args); - }, -}; + createPreMigrationBackup, + hashFileSync, + type PreMigrationBackupReceipt, +} from "./migrationRunner/preMigrationBackup"; +import { + detectNameMismatches, + getPlausiblePendingCount, + hasColumn, + hasLedgerRepairCandidates, + hasPhysicalTable, + hasTable, + inferPhysicalSchemaBaseline, + reconcileRenumberedMigrations, + rehomeLegacyVersionSlotMigrations, +} from "./migrationRunner/schemaState"; /** * Resolve the migrations directory path safely across platforms. @@ -336,16 +328,96 @@ function getAppliedRecords(db: SqliteAdapter): Array<{ version: string; name: st }>; } -function hasTable(db: SqliteAdapter, tableName: string): boolean { - const row = db - .prepare("SELECT name FROM sqlite_master WHERE type IN ('table', 'view') AND name = ?") - .get(tableName) as { name?: string } | undefined; - return Boolean(row?.name); +/** + * Reopen a narrowly selected migration when the table it creates is physically absent. + * + * Historical databases can carry `074_discovery_results` or the rehomed + * `081_inspector_custom_hosts` in the ledger without the table itself (for example after a + * version-slot collision or an incomplete manual recovery). Treating either marker as + * authoritative leaves an incomplete schema. A same-named view does not count as the table; + * replaying the owning migration fails closed instead of silently advancing. + * + * This intentionally detects table absence only. It is not a general schema-healing layer: + * column/rebuild migrations continue to use targeted idempotency checks elsewhere. + */ +const REQUIRED_PHYSICAL_MIGRATIONS = [ + { version: "074", name: "discovery_results", tableName: "discovery_results" }, + { version: "081", name: "inspector_custom_hosts", tableName: "inspector_custom_hosts" }, +] as const; + +function validateRequiredPhysicalMigrationProvenance( + db: SqliteAdapter, + files: Array<{ version: string; name: string; path: string }> +): void { + for (const required of REQUIRED_PHYSICAL_MIGRATIONS) { + if (hasPhysicalTable(db, required.tableName)) continue; + + const migrationExists = files.some( + (file) => file.version === required.version && file.name === required.name + ); + if (!migrationExists) continue; + + const occupied = db + .prepare("SELECT version, name FROM _omniroute_migrations WHERE version = ?") + .get(required.version) as { version: string; name: string } | undefined; + if (!occupied || occupied.name === required.name) continue; + + const knownRenumberedCollision = RENAMED_MIGRATION_COMPATIBILITY.some( + (compatibility) => + compatibility.fromVersion === occupied.version && + compatibility.fromName === occupied.name && + files.some( + (file) => file.version === compatibility.toVersion && file.name === compatibility.toName + ) && + files.some( + (file) => + file.version === compatibility.fromVersion && file.name !== compatibility.fromName + ) + ); + const knownLegacySlotCollision = LEGACY_VERSION_SLOT_MIGRATIONS.some( + (legacy) => + legacy.version === occupied.version && + legacy.name === occupied.name && + files.some((file) => file.version === legacy.version && file.name !== legacy.name) + ); + const knownRepairableCollision = knownRenumberedCollision || knownLegacySlotCollision; + if (knownRepairableCollision) continue; + + throw new Error( + `[Migration] Required table "${required.tableName}" is missing, but version ` + + `${required.version} is recorded as unknown migration "${occupied.name}" instead of ` + + `"${required.name}". Refusing to treat this database as current.` + ); + } } -function hasColumn(db: SqliteAdapter, tableName: string, columnName: string): boolean { - const columns = db.prepare(`PRAGMA table_info(${tableName})`).all() as Array<{ name?: string }>; - return columns.some((column) => column.name === columnName); +function findAtomicPhysicalReplays( + db: SqliteAdapter, + files: Array<{ version: string; name: string; path: string }> +): Set { + const replayVersions = new Set(); + + for (const required of REQUIRED_PHYSICAL_MIGRATIONS) { + if (hasPhysicalTable(db, required.tableName)) continue; + + const migrationExists = files.some( + (file) => file.version === required.version && file.name === required.name + ); + if (!migrationExists) continue; + + const applied = db + .prepare("SELECT version, name FROM _omniroute_migrations WHERE version = ? AND name = ?") + .get(required.version, required.name) as { version: string; name: string } | undefined; + if (!applied) continue; + + replayVersions.add(required.version); + console.warn( + `[Migration] Will atomically replay ${required.version}_${required.name}: ledger recorded ` + + `"${applied.name}" but required table "${required.tableName}" is missing.` + ); + } + + return replayVersions; } function ensureColumn(db: SqliteAdapter, tableName: string, columnName: string, ddl: string): void { @@ -651,276 +723,31 @@ function applyCompressionCombosMigration(db: SqliteAdapter, migrationPath: strin `); } -function inferPhysicalSchemaBaseline(db: SqliteAdapter): { - version: string; - description: string; -} | null { - for (const sentinel of PHYSICAL_SCHEMA_SENTINELS) { - if (hasTable(db, sentinel.tableName)) { - return { - version: sentinel.version, - description: sentinel.description, - }; - } - } - - const hasInitialSchema = INITIAL_SCHEMA_SENTINELS.every((tableName) => hasTable(db, tableName)); - if (hasInitialSchema) { - return { - version: "001", - description: "initial schema tables", - }; - } - - return null; -} - -function getPlausiblePendingCount( - files: Array<{ version: string; name: string; path: string }>, - baselineVersion: string -): number { - const baseline = Number.parseInt(baselineVersion, 10); - return files.filter((file) => Number.parseInt(file.version, 10) > baseline).length; -} - /** - * Detect migration name mismatches — when a migration version number - * has been reused/renumbered with a different name. This is a strong signal - * that the migration tracking is corrupted or migrations were renumbered. - */ -function detectNameMismatches( - appliedRecords: Array<{ version: string; name: string }>, - files: Array<{ version: string; name: string; path: string }> -): Array<{ version: string; appliedName: string; diskName: string }> { - const appliedByName = new Map(appliedRecords.map((r) => [r.version, r.name])); - const mismatches: Array<{ version: string; appliedName: string; diskName: string }> = []; - - for (const file of files) { - const appliedName = appliedByName.get(file.version); - if (appliedName && appliedName !== file.name) { - mismatches.push({ - version: file.version, - appliedName, - diskName: file.name, - }); - } - } - - return mismatches; -} - -function reconcileRenumberedMigrations( - db: SqliteAdapter, - files: Array<{ version: string; name: string; path: string }> -): boolean { - let repaired = false; - - for (const compatibility of RENAMED_MIGRATION_COMPATIBILITY) { - const hasTargetFile = files.some( - (file) => file.version === compatibility.toVersion && file.name === compatibility.toName - ); - const hasSourceFile = files.some( - (file) => file.version === compatibility.fromVersion && file.name !== compatibility.fromName - ); - - if (!hasTargetFile || !hasSourceFile) { - continue; - } - - const legacyRow = db - .prepare("SELECT version, name FROM _omniroute_migrations WHERE version = ? AND name = ?") - .get(compatibility.fromVersion, compatibility.fromName) as - { version: string; name: string } | undefined; - if (!legacyRow) { - continue; - } - - const targetRow = db - .prepare("SELECT version FROM _omniroute_migrations WHERE version = ?") - .get(compatibility.toVersion) as { version: string } | undefined; - - const applyRepair = db.transaction(() => { - if (targetRow) { - db.prepare("DELETE FROM _omniroute_migrations WHERE version = ? AND name = ?").run( - compatibility.fromVersion, - compatibility.fromName - ); - } else { - db.prepare( - "UPDATE _omniroute_migrations SET version = ?, name = ? WHERE version = ? AND name = ?" - ).run( - compatibility.toVersion, - compatibility.toName, - compatibility.fromVersion, - compatibility.fromName - ); - } - }); - - applyRepair(); - repaired = true; - console.warn( - `[Migration] Reconciled renamed migration ${compatibility.fromVersion}_${compatibility.fromName} ` + - `to ${compatibility.toVersion}_${compatibility.toName} to preserve pending migrations.` - ); - - // After the compat rewrite, verify the old version slot is now free. - // A residual row (from a failed prior run, manual intervention, or edge-case - // UPDATE conflict) at the old version would shadow a NEW migration file - // placed at that version number — e.g. 028_create_files_and_batches.sql - // would be skipped because getAppliedVersions() still sees version "028". - const residualRow = db - .prepare("SELECT version, name FROM _omniroute_migrations WHERE version = ?") - .get(compatibility.fromVersion) as { version: string; name: string } | undefined; - if (residualRow) { - console.warn( - `[Migration] ⚠️ Residual row at version ${compatibility.fromVersion} ` + - `(name: "${residualRow.name}") still present after compat rewrite — ` + - `removing to unblock new migration at this version slot.` - ); - db.prepare("DELETE FROM _omniroute_migrations WHERE version = ?").run( - compatibility.fromVersion - ); - } - } - - return repaired; -} - -function rehomeLegacyVersionSlotMigrations( - db: SqliteAdapter, - files: Array<{ version: string; name: string; path: string }> -): boolean { - let repaired = false; - const diskNamesByVersion = new Map(files.map((file) => [file.version, file.name])); - - for (const legacy of LEGACY_VERSION_SLOT_MIGRATIONS) { - const diskName = diskNamesByVersion.get(legacy.version); - if (!diskName || diskName === legacy.name) { - continue; - } - - const legacyRow = db - .prepare("SELECT version, name FROM _omniroute_migrations WHERE version = ? AND name = ?") - .get(legacy.version, legacy.name) as { version: string; name: string } | undefined; - if (!legacyRow) { - continue; - } - - const legacyVersion = `legacy-${legacy.version}-${legacy.name}`; - const applyRepair = db.transaction(() => { - const existingLegacyRow = db - .prepare("SELECT version FROM _omniroute_migrations WHERE version = ?") - .get(legacyVersion) as { version: string } | undefined; - - if (existingLegacyRow) { - db.prepare("DELETE FROM _omniroute_migrations WHERE version = ? AND name = ?").run( - legacy.version, - legacy.name - ); - return; - } - - db.prepare("UPDATE _omniroute_migrations SET version = ? WHERE version = ? AND name = ?").run( - legacyVersion, - legacy.version, - legacy.name - ); - }); - - applyRepair(); - repaired = true; - console.warn( - `[Migration] Rehomed legacy migration ${legacy.version}_${legacy.name} ` + - `to ${legacyVersion} so current ${legacy.version}_${diskName} can apply.` - ); - } - - return repaired; -} - -/** - * Read a persisted `dbBackup` retention setting through the adapter that is ALREADY open - * for this migration run. + * Run a callback while holding SQLite's IMMEDIATE writer transaction. * - * `backup.ts`'s equivalent goes through `getDbInstance()`, which is unsafe here: this - * code runs from inside database initialization, so asking for the singleton would - * re-enter it. Reading off `db` keeps the same stored values without that risk. A DB too - * old to have `key_value` yet simply falls back to the default. + * Production adapters expose `immediate()` directly. A small number of long-standing + * migration tests and external callers still pass a raw better-sqlite3 Database, whose + * transaction wrapper exposes `.immediate()` instead. Supporting both shapes here keeps + * the safety transaction real: this must never degrade to a plain callback invocation. */ -function readStoredBackupSetting(db: SqliteAdapter, key: string, min: number): number | undefined { - try { - const row = db - .prepare("SELECT value FROM key_value WHERE namespace = ? AND key = ?") - .get("dbBackup", key) as { value?: string } | undefined; - if (!row?.value) return undefined; - const parsed = JSON.parse(row.value); - return Number.isInteger(parsed) && parsed >= min ? parsed : undefined; - } catch { - return undefined; +function runImmediateTransaction(db: SqliteAdapter, fn: () => T): T { + const adapterImmediate = (db as Partial).immediate; + if (typeof adapterImmediate === "function") { + let result!: T; + adapterImmediate.call(db, () => { + result = fn(); + }); + return result; } -} -/** - * Enforce the backup retention budget after a pre-migration snapshot (#10421). - * - * Precedence matches `backup.ts`: env override → persisted operator setting → default. - * Never throws: a migration must not fail because housekeeping did. - */ -function pruneMigrationBackups(db: SqliteAdapter, backupDir: string): void { - try { - const maxFiles = process.env.DB_BACKUP_MAX_FILES - ? parsePositiveInt(process.env.DB_BACKUP_MAX_FILES, MAX_DB_BACKUPS) - : (readStoredBackupSetting(db, "maxFiles", 1) ?? MAX_DB_BACKUPS); - const retentionDays = process.env.DB_BACKUP_RETENTION_DAYS - ? parseNonNegativeInt(process.env.DB_BACKUP_RETENTION_DAYS, DEFAULT_DB_BACKUP_RETENTION_DAYS) - : (readStoredBackupSetting(db, "retentionDays", 0) ?? DEFAULT_DB_BACKUP_RETENTION_DAYS); - - const result = pruneBackupDirectory({ backupDir, maxFiles, retentionDays }); - if (result.deletedFiles > 0) { - console.log( - `[Migration] Pruned ${result.deletedFiles} old backup file(s) ` + - `(${result.keptBackupFamilies} kept, maxFiles=${maxFiles}, retentionDays=${retentionDays}).` - ); - } - } catch (err: unknown) { - const message = err instanceof Error ? err.message : String(err); - console.warn(`[Migration] Failed to prune old backups: ${message}`); - } -} - -/** - * Create a pre-migration backup of the SQLite database using VACUUM INTO. - * Returns the backup path on success, null on failure. - */ -function createPreMigrationBackup(db: SqliteAdapter): string | null { - try { - const sqliteFile = db.name; - if (!sqliteFile || sqliteFile === ":memory:") return null; - - const backupDir = path.join(path.dirname(sqliteFile), "db_backups"); - if (!fs.existsSync(backupDir)) { - fs.mkdirSync(backupDir, { recursive: true }); - } - - const timestamp = new Date().toISOString().replace(/[:.]/g, "-"); - const backupPath = path.join(backupDir, `db_${timestamp}_pre-migration.sqlite`); - const escapedBackupPath = backupPath.replace(/'/g, "''"); - - db.exec(`VACUUM INTO '${escapedBackupPath}'`); - console.log(`[Migration] Pre-migration backup created: ${backupPath}`); - - // #10421: apply the operator's retention budget right here. Without this the - // migration path was the one backup producer that never pruned, so every process - // start with a pending migration added ~5 MB forever (observed: 49k files / 204 GB). - pruneMigrationBackups(db, backupDir); - - return backupPath; - } catch (err: unknown) { - const message = err instanceof Error ? err.message : String(err); - console.warn(`[Migration] Failed to create pre-migration backup: ${message}`); - return null; + const rawTransaction = db.transaction(fn) as ReturnType & { + immediate?: () => T; + }; + if (typeof rawTransaction.immediate !== "function") { + throw new Error("[Migration] Database adapter does not support IMMEDIATE transactions."); } + return rawTransaction.immediate(); } /** @@ -932,15 +759,243 @@ function createPreMigrationBackup(db: SqliteAdapter): string | null { * 2. Aborts if too many pending migrations on an existing DB (likely wipe) * 3. Creates automatic backup before running any migrations */ -export function runMigrations(db: SqliteAdapter, options?: { isNewDb?: boolean }): number { +export function runMigrations( + db: SqliteAdapter, + options?: { isNewDb?: boolean; databaseExistedBeforeInitialization?: boolean } +): number { const isNewDb = options?.isNewDb === true; + // `isNewDb` also covers a setup-created skeleton so it can bypass the mass-migration + // false positive. Snapshot eligibility must use the independent physical-file fact: + // that skeleton can already contain provider credentials and other operator state. + const databaseExistedBeforeInitialization = + options?.databaseExistedBeforeInitialization ?? !isNewDb; ensureMigrationsTable(db); const files = filterSupersededDuplicateMigrations(getMigrationFiles()); - rehomeLegacyVersionSlotMigrations(db, files); - reconcileRenumberedMigrations(db, files); - const applied = getAppliedVersions(db); - const appliedRecords = getAppliedRecords(db); + validateRequiredPhysicalMigrationProvenance(db, files); + let preMigrationBackup: PreMigrationBackupReceipt | null = null; + let plan!: { + atomicPhysicalReplays: Set; + appliedRecords: Array<{ version: string; name: string }>; + pending: typeof files; + deferredUnsupported: typeof files; + highestAppliedBeforeMigrations: number; + }; + let count = 0; + + const preliminaryApplied = getAppliedVersions(db); + const preliminaryAtomicReplays = findAtomicPhysicalReplays(db, files); + const preliminaryPending = files.filter( + (file) => !preliminaryApplied.has(file.version) || preliminaryAtomicReplays.has(file.version) + ); + const preliminaryDeferred = preliminaryPending.filter((migration) => + isDeferredUnsupportedMigration(db, migration) + ); + const preliminaryActionable = preliminaryPending.filter( + (migration) => !preliminaryDeferred.some((deferred) => deferred.version === migration.version) + ); + const preliminaryHasRepairCandidates = hasLedgerRepairCandidates(db, files); + + // Preserve the historical read-only/no-op path. Merely checking an already-current + // database must not acquire a writer lock (or fail SQLITE_BUSY because another supported + // host currently owns one). Safety state is recomputed under IMMEDIATE whenever work exists. + if (preliminaryActionable.length === 0 && !preliminaryHasRepairCandidates) { + const numericApplied = Array.from(preliminaryApplied) + .map((version) => Number.parseInt(version, 10)) + .filter((version) => !Number.isNaN(version)); + plan = { + atomicPhysicalReplays: preliminaryAtomicReplays, + appliedRecords: getAppliedRecords(db), + pending: preliminaryPending, + deferredUnsupported: preliminaryDeferred, + highestAppliedBeforeMigrations: numericApplied.length > 0 ? Math.max(...numericApplied) : 0, + }; + } + + // sql.js export() finalizes its active SAVEPOINT, so exporting from inside + // `db.immediate()` would make a later safety throw unable to roll repairs back. + // Its adapter is synchronous and in-memory, so no JavaScript writer can interleave + // between this preflight/export and the immediately following savepoint. + if ( + !plan && + db.driver === "sql.js" && + (preliminaryActionable.length > 0 || preliminaryHasRepairCandidates) + ) { + const needsSnapshot = + (preliminaryActionable.length > 0 || preliminaryHasRepairCandidates) && + db.name !== ":memory:" && + databaseExistedBeforeInitialization; + + if (needsSnapshot) { + preMigrationBackup = createPreMigrationBackup(db); + if (!preMigrationBackup) { + throw new Error( + "[Migration] Refusing to migrate an existing database without a durable snapshot. " + + "The DATA_DIR filesystem must support atomic hard-link publication." + ); + } + } + } + + // Hold SQLite's native writer lock through snapshot selection, compatibility repairs, + // and the mass-safety decision. Native adapters open a separate read-only connection + // for VACUUM INTO while competing writers remain blocked. The outer transaction then + // commits before migrations so the repository's one-transaction-per-file contract stays + // intact: an earlier successful migration remains committed if a later file fails. + if (!plan) + runImmediateTransaction(db, () => { + const appliedBeforeRepair = getAppliedVersions(db); + const hadAppliedBeforeRepair = appliedBeforeRepair.size > 0; + const preliminaryAtomicReplays = findAtomicPhysicalReplays(db, files); + const preliminaryPending = files.filter( + (file) => + !appliedBeforeRepair.has(file.version) || preliminaryAtomicReplays.has(file.version) + ); + const preliminaryActionable = preliminaryPending.filter( + (migration) => !isDeferredUnsupportedMigration(db, migration) + ); + const mayWriteExistingDatabase = + preliminaryActionable.length > 0 || hasLedgerRepairCandidates(db, files); + const needsSnapshot = + mayWriteExistingDatabase && db.name !== ":memory:" && databaseExistedBeforeInitialization; + + if (needsSnapshot && !preMigrationBackup) { + if (db.driver === "sql.js") { + throw new Error( + "[Migration] sql.js safety state changed after its pre-transaction snapshot preflight; " + + "refusing to export from inside the rollback savepoint." + ); + } + preMigrationBackup = createPreMigrationBackup(db); + if (!preMigrationBackup) { + throw new Error( + "[Migration] Refusing to migrate an existing database without a durable snapshot. " + + "The DATA_DIR filesystem must support atomic hard-link publication." + ); + } + } + + rehomeLegacyVersionSlotMigrations(db, files); + reconcileRenumberedMigrations(db, files); + + const atomicPhysicalReplays = findAtomicPhysicalReplays(db, files); + const applied = getAppliedVersions(db); + const appliedRecords = getAppliedRecords(db); + const pending = files.filter( + (file) => !applied.has(file.version) || atomicPhysicalReplays.has(file.version) + ); + const deferredUnsupported = pending.filter((migration) => + isDeferredUnsupportedMigration(db, migration) + ); + const actionablePending = pending.filter( + (migration) => + !deferredUnsupported.some((deferred) => deferred.version === migration.version) + ); + const isFreshSeedOnly = + applied.size === 1 && + applied.has("001") && + inferPhysicalSchemaBaseline(db) === null && + hasTable(db, "provider_connections"); + const requiresDurableBackup = + actionablePending.length > 0 && + db.name !== ":memory:" && + databaseExistedBeforeInitialization; + + // Recompute under the same writer transaction as repairs and fail before any + // ledger mutation can commit if the durable-snapshot requirement is not met. + if (requiresDurableBackup && !preMigrationBackup) { + throw new Error( + "[Migration] Refusing to migrate an existing database without a durable snapshot. " + + "The DATA_DIR filesystem must support atomic hard-link publication." + ); + } + + const isTestEnvironment = isAutomatedTestProcess(); + const maxPendingMigrations = resolveMaxPendingMigrations(); + if ( + actionablePending.length > 0 && + !isTestEnvironment && + !isNewDb && + !isFreshSeedOnly && + maxPendingMigrations > 0 && + (applied.size > 0 || hadAppliedBeforeRepair) && + actionablePending.length > maxPendingMigrations + ) { + const physicalBaseline = inferPhysicalSchemaBaseline(db); + const plausiblePendingCount = physicalBaseline + ? getPlausiblePendingCount(files, physicalBaseline.version) + : null; + + if (plausiblePendingCount !== null && actionablePending.length <= plausiblePendingCount) { + console.warn( + `[Migration] Allowing ${actionablePending.length} pending migrations on an existing database ` + + `because the physical schema only proves ${physicalBaseline?.version} ` + + `(${physicalBaseline?.description}).` + ); + } else { + const schemaHint = + physicalBaseline && plausiblePendingCount !== null + ? ` Physical schema already shows ${physicalBaseline.version} ` + + `(${physicalBaseline.description}), so at most ${plausiblePendingCount} pending ` + + `migration(s) are expected from a legitimate upgrade.` + : ""; + const bypassHint = + ` To bypass this check (e.g. after restoring a backup where the migration ` + + `tracking table was wiped), set OMNIROUTE_MAX_PENDING_MIGRATIONS=0 in your ` + + `server.env or DATA_DIR/.env and restart.`; + const msg = + `[Migration] 🛑 ABORT: Detected ${actionablePending.length} pending migrations on an existing database ` + + `(threshold is ${maxPendingMigrations}). ` + + `This usually means the migration tracking table was accidentally wiped. ` + + `Running all migrations from scratch will cause data loss or schema errors.` + + schemaHint + + bypassHint; + + if (memoizedSafetyAbort && memoizedSafetyAbort.message === msg) { + console.error( + `[Migration] 🛑 ABORT (repeat — see earlier detail): ` + + `${actionablePending.length} pending > threshold ${maxPendingMigrations}. ` + + `Set OMNIROUTE_MAX_PENDING_MIGRATIONS=0 to bypass.` + ); + throw memoizedSafetyAbort; + } + console.error(msg); + memoizedSafetyAbort = new MigrationSafetyAbortError(msg); + throw memoizedSafetyAbort; + } + } + + if ( + preMigrationBackup && + hashFileSync(preMigrationBackup.path) !== preMigrationBackup.sha256 + ) { + throw new Error( + "[Migration] Refusing to migrate because the pre-migration snapshot changed before use." + ); + } + + const numericApplied = Array.from(applied) + .map((version) => Number.parseInt(version, 10)) + .filter((version) => !Number.isNaN(version)); + const highestAppliedBeforeMigrations = + numericApplied.length > 0 ? Math.max(...numericApplied) : 0; + + plan = { + atomicPhysicalReplays, + appliedRecords, + pending, + deferredUnsupported, + highestAppliedBeforeMigrations, + }; + }); + + const { + atomicPhysicalReplays, + appliedRecords, + pending, + deferredUnsupported, + highestAppliedBeforeMigrations, + } = plan; // ── Safety Check 1: Detect migration name mismatches (renumbering) ── const mismatches = detectNameMismatches(appliedRecords, files); @@ -963,34 +1018,15 @@ export function runMigrations(db: SqliteAdapter, options?: { isNewDb?: boolean } ); } - // ── Gap Reconciliation: Identify non-contiguous missing migrations ── - // Do not rely on any highest-version-applied heuristic. We must explicitly - // iterate through all missing files on disk and apply them if they are missing - // from the _omniroute_migrations table. - const numericApplied = Array.from(applied) - .map((v) => Number.parseInt(v, 10)) - .filter((n) => !Number.isNaN(n)); - const highestApplied = numericApplied.length > 0 ? Math.max(...numericApplied) : 0; - const pending = files.filter((f) => { - const isMissing = !applied.has(f.version); - if (isMissing && Number(f.version) < highestApplied) { + for (const migration of pending) { + if (Number(migration.version) < highestAppliedBeforeMigrations) { console.warn( `[Migration] 🔄 RECONCILIATION: Found missing intermediate migration ` + - `${f.version}_${f.name} (highest applied is ${highestApplied}). ` + + `${migration.version}_${migration.name} ` + + `(highest applied is ${highestAppliedBeforeMigrations}). ` + `This gap will be back-filled to ensure schema integrity.` ); } - return isMissing; - }); - const deferredUnsupported = pending.filter((migration) => - isDeferredUnsupportedMigration(db, migration) - ); - const actionablePending = pending.filter( - (migration) => !deferredUnsupported.some((deferred) => deferred.version === migration.version) - ); - - if (pending.length === 0) { - return 0; // Nothing to do } if (deferredUnsupported.length > 0) { @@ -1003,101 +1039,28 @@ export function runMigrations(db: SqliteAdapter, options?: { isNewDb?: boolean } ); } - // ── Safety Check 2: Mass-migration detection (abort if existing DB + many migrations) ── - // Skip in test environments where fresh DBs legitimately have many pending migrations. - const isTestEnvironment = isAutomatedTestProcess(); - - // #3416: resolve the threshold at call time so OMNIROUTE_MAX_PENDING_MIGRATIONS - // can override the default (0 disables the check). The abort message below - // interpolates this resolved value, so it auto-reflects any override. - const maxPendingMigrations = resolveMaxPendingMigrations(); - - // #9934: `omniroute setup`'s openOmniRouteDb writes a partial skeleton file - // (provider_connections + key_value) that has never had migrations run. When - // the first `serve` opens it and auto-seeds only the 001 marker, the applied - // set is exactly {001} — which would otherwise look like a wiped existing DB - // and trip this abort on a brand-new install. This is distinct from a real - // wiped/backup-restored database: that case has a non-trivial physical schema - // (baseline inference is non-null) and full data tables, so it still aborts. - // The 001-marker-only state on a provider_connections skeleton is the fresh - // auto-seed — let it through. A genuinely empty table is already exempt via - // `applied.size > 0`, and an upgraded DB has a non-trivial applied set. - const isFreshSeedOnly = - applied.size === 1 && - applied.has("001") && - inferPhysicalSchemaBaseline(db) === null && - hasTable(db, "provider_connections"); - - if ( - !isTestEnvironment && - !isNewDb && - !isFreshSeedOnly && - process.env.DISABLE_SQLITE_AUTO_BACKUP !== "true" && - maxPendingMigrations > 0 && - applied.size > 0 && - actionablePending.length > maxPendingMigrations - ) { - const physicalBaseline = inferPhysicalSchemaBaseline(db); - const plausiblePendingCount = physicalBaseline - ? getPlausiblePendingCount(files, physicalBaseline.version) - : null; - - if (plausiblePendingCount !== null && actionablePending.length <= plausiblePendingCount) { - console.warn( - `[Migration] Allowing ${actionablePending.length} pending migrations on an existing database ` + - `because the physical schema only proves ${physicalBaseline?.version} ` + - `(${physicalBaseline?.description}).` - ); - } else { - const schemaHint = - physicalBaseline && plausiblePendingCount !== null - ? ` Physical schema already shows ${physicalBaseline.version} ` + - `(${physicalBaseline.description}), so at most ${plausiblePendingCount} pending ` + - `migration(s) are expected from a legitimate upgrade.` - : ""; - const bypassHint = - ` To bypass this check (e.g. after restoring a backup where the migration ` + - `tracking table was wiped), set OMNIROUTE_MAX_PENDING_MIGRATIONS=0 in your ` + - `server.env or DATA_DIR/.env and restart.`; - const msg = - `[Migration] 🛑 ABORT: Detected ${actionablePending.length} pending migrations on an existing database ` + - `(threshold is ${maxPendingMigrations}). ` + - `This usually means the migration tracking table was accidentally wiped. ` + - `Running all migrations from scratch will cause data loss or schema errors.` + - schemaHint + - bypassHint; - - // #6260: memoize so the cascade of downstream ensureDbInitialized() calls - // that re-open the DB throw the SAME instance and only log once. - if (memoizedSafetyAbort && memoizedSafetyAbort.message === msg) { - console.error( - `[Migration] 🛑 ABORT (repeat — see earlier detail): ` + - `${actionablePending.length} pending > threshold ${maxPendingMigrations}. ` + - `Set OMNIROUTE_MAX_PENDING_MIGRATIONS=0 to bypass.` - ); - throw memoizedSafetyAbort; - } - console.error(msg); - memoizedSafetyAbort = new MigrationSafetyAbortError(msg); - throw memoizedSafetyAbort; - } + if (preMigrationBackup && hashFileSync(preMigrationBackup.path) !== preMigrationBackup.sha256) { + throw new Error( + "[Migration] Refusing to migrate because the pre-migration snapshot changed before use." + ); } - // ── Safety Check 3: Pre-migration backup ── - // Skip backup if it's a completely fresh database (0 applied and all pending) - // or if running in tests (where AUTO_BACKUP might be disabled) - if (applied.size > 0 && process.env.DISABLE_SQLITE_AUTO_BACKUP !== "true") { - createPreMigrationBackup(db); - } - - let count = 0; - for (const migration of pending) { - if (isDeferredUnsupportedMigration(db, migration)) { - continue; - } + if (isDeferredUnsupportedMigration(db, migration)) continue; const applyMigration = db.transaction(() => { + if (atomicPhysicalReplays.has(migration.version)) { + const removed = db + .prepare("DELETE FROM _omniroute_migrations WHERE version = ? AND name = ?") + .run(migration.version, migration.name); + if (removed.changes !== 1) { + throw new Error( + `[Migration] Atomic replay lost its expected ledger marker for ` + + `${migration.version}_${migration.name}.` + ); + } + } + if (isSchemaAlreadyApplied(db, migration)) { console.warn( `[Migration] Skipped executing ${migration.version}_${migration.name} as schema changes are already present (Idempotency check).` @@ -1120,29 +1083,36 @@ export function runMigrations(db: SqliteAdapter, options?: { isNewDb?: boolean } try { applyMigration(); - count++; + count += 1; console.log(`[Migration] Applied: ${migration.version}_${migration.name}`); } catch (err: unknown) { const message = err instanceof Error ? err.message : String(err); - // "duplicate column name" means the column already exists — end state achieved, mark applied. - if (message.includes("duplicate column name")) { + if ( + message.includes("duplicate column name") && + !atomicPhysicalReplays.has(migration.version) + ) { const applyMarkerOnly = db.transaction(() => { db.prepare( "INSERT OR IGNORE INTO _omniroute_migrations (version, name) VALUES (?, ?)" ).run(migration.version, migration.name); }); applyMarkerOnly(); - count++; + count += 1; console.log( `[Migration] Applied (column pre-exists): ${migration.version}_${migration.name}` ); } else { console.error(`[Migration] FAILED: ${migration.version}_${migration.name} — ${message}`); - throw err; // Re-throw to prevent DB from starting in inconsistent state + throw err; } } } + // Retention intentionally does not run inside the migration window. Another process + // may still be using a different snapshot as its in-flight restore point. Manual and + // scheduled backup paths continue to enforce the operator's retention policy; retries + // here are bounded by the deterministic content address instead of destructive pruning. + if (count > 0) { console.log(`[Migration] ${count} migration(s) applied successfully.`); } @@ -1175,7 +1145,7 @@ function insertDefaultDatabaseSettings(db: SqliteAdapter) { // Run in an immediate transaction to avoid nested transactions try { - db.immediate(() => { + runImmediateTransaction(db, () => { tx(); }); } catch (error) { diff --git a/src/lib/db/migrationRunner/constants.ts b/src/lib/db/migrationRunner/constants.ts index 773f6e8c12..089837f93f 100644 --- a/src/lib/db/migrationRunner/constants.ts +++ b/src/lib/db/migrationRunner/constants.ts @@ -158,6 +158,14 @@ export const RENAMED_MIGRATION_COMPATIBILITY = [ toVersion: "151", toName: "windsurf_to_devin_desktop", }, + { + // inspector_custom_hosts was once published in slot 074, now occupied by + // discovery_results. Its canonical idempotent migration lives at 081. + fromVersion: "074", + fromName: "inspector_custom_hosts", + toVersion: "081", + toName: "inspector_custom_hosts", + }, { fromVersion: "134", fromName: "ccr_blocks", diff --git a/src/lib/db/migrationRunner/logger.ts b/src/lib/db/migrationRunner/logger.ts new file mode 100644 index 0000000000..c9f9b0d2e7 --- /dev/null +++ b/src/lib/db/migrationRunner/logger.ts @@ -0,0 +1,13 @@ +const isNodeTestRunnerChild = typeof process.env.NODE_TEST_CONTEXT === "string"; + +export const migrationConsole = { + log: (...args: unknown[]) => { + if (!isNodeTestRunnerChild) globalThis.console.log(...args); + }, + warn: (...args: unknown[]) => { + if (!isNodeTestRunnerChild) globalThis.console.warn(...args); + }, + error: (...args: unknown[]) => { + globalThis.console.error(...args); + }, +}; diff --git a/src/lib/db/migrationRunner/preMigrationBackup.ts b/src/lib/db/migrationRunner/preMigrationBackup.ts new file mode 100644 index 0000000000..3e2eea3caf --- /dev/null +++ b/src/lib/db/migrationRunner/preMigrationBackup.ts @@ -0,0 +1,293 @@ +import { createHash } from "crypto"; +import fs from "fs"; +import path from "path"; + +import type { SqliteAdapter } from "../adapters/types"; +import { tryOpenSync } from "../adapters/driverFactory"; +import { migrationConsole as console } from "./logger"; + +export type PreMigrationBackupReceipt = { + path: string; + sha256: string; +}; + +function fsyncDirectoryEntry(directory: string): void { + let fd: number | null = null; + try { + fd = fs.openSync(directory, "r"); + fs.fsyncSync(fd); + } catch (error: unknown) { + const code = (error as NodeJS.ErrnoException | null)?.code; + const windowsDirectoryHandleUnsupported = + process.platform === "win32" && + (code === "EACCES" || code === "EPERM" || code === "EISDIR" || code === "EINVAL"); + if (!windowsDirectoryHandleUnsupported) throw error; + } finally { + if (fd !== null) fs.closeSync(fd); + } +} + +export function hashFileSync(filePath: string): string { + const hash = createHash("sha256"); + const fd = fs.openSync(filePath, "r"); + const buffer = Buffer.allocUnsafe(1024 * 1024); + let position = 0; + + try { + while (true) { + const bytesRead = fs.readSync(fd, buffer, 0, buffer.length, position); + if (bytesRead === 0) break; + hash.update(buffer.subarray(0, bytesRead)); + position += bytesRead; + } + } finally { + fs.closeSync(fd); + } + + return hash.digest("hex"); +} + +function getReusablePreMigrationBackup( + candidatePath: string, + expectedSha256: string +): PreMigrationBackupReceipt | null { + if (!fs.existsSync(candidatePath)) return null; + + const before = fs.lstatSync(candidatePath); + if (!before.isFile() || hashFileSync(candidatePath) !== expectedSha256) { + throw new Error( + `[Migration] Content-addressed snapshot path exists with unexpected content: ${candidatePath}` + ); + } + const after = fs.lstatSync(candidatePath); + if ( + before.dev !== after.dev || + before.ino !== after.ino || + before.size !== after.size || + before.mtimeMs !== after.mtimeMs + ) { + throw new Error( + `[Migration] Content-addressed snapshot changed while it was being validated: ${candidatePath}` + ); + } + + return { path: candidatePath, sha256: expectedSha256 }; +} + +function publishSnapshotWithoutOverwrite(tempPath: string, destination: string): void { + // link() publishes a complete same-filesystem image atomically and, unlike rename(), + // fails with EEXIST instead of overwriting a path created by another process. There is + // deliberately no copy/rename fallback: filesystems without this primitive fail closed + // instead of exposing a partial canonical `.sqlite` file after a crash. + fs.linkSync(tempPath, destination); + const publishedFd = fs.openSync(destination, "r+"); + try { + // Flush through the published name as well as the already-fsynced temp handle. + // On Windows this maps to FlushFileBuffers and is the strongest file-level + // durability proof available when directory handles are unsupported by Node. + fs.fsyncSync(publishedFd); + } finally { + fs.closeSync(publishedFd); + } + fsyncDirectoryEntry(path.dirname(destination)); +} + +function fsyncReusableSnapshot(snapshotPath: string): void { + const fd = fs.openSync(snapshotPath, "r+"); + try { + fs.fsyncSync(fd); + } finally { + fs.closeSync(fd); + } +} + +type SqlJsSnapshotClone = { + run(sql: string): void; + export(): Uint8Array; + close(): void; +}; + +const SQLITE_HEADER_MIN_BYTES = 100; +const SQLITE_HEADER_MAGIC = "SQLite format 3\0"; +const SQLITE_CHANGE_COUNTER_OFFSET = 24; +const SQLITE_VERSION_VALID_FOR_OFFSET = 92; +const SQLITE_STANDALONE_CHANGE_COUNTER = 1; + +function exportCanonicalSqlJsSnapshot(raw: { export: () => Uint8Array }): Buffer { + const RawDatabase = ( + raw as unknown as { constructor: new (data: Uint8Array) => SqlJsSnapshotClone } + ).constructor; + let clone: SqlJsSnapshotClone | null = null; + + try { + // A rolled-back sql.js SAVEPOINT can leave SQLite's physical change counter advanced + // even though every logical row/schema change was undone. Canonicalize only a detached + // clone: VACUUM removes rollback-only page artifacts without touching the live database. + clone = new RawDatabase(raw.export()); + clone.run("VACUUM"); + const canonical = Buffer.from(clone.export()); + + if ( + canonical.length < SQLITE_HEADER_MIN_BYTES || + canonical.subarray(0, SQLITE_HEADER_MAGIC.length).toString("binary") !== SQLITE_HEADER_MAGIC + ) { + throw new Error("sql.js export did not produce a valid SQLite file header"); + } + + // SQLite file-header offsets 24 and 92 are the change counter and + // version-valid-for number. VACUUM keeps the two equal, but seeds them from the + // source image, so an otherwise identical rolled-back retry still gets a different + // byte hash. A standalone snapshot has no open readers to invalidate; assigning the + // same stable value to both fields preserves a valid/restorable header while making + // the complete canonical image deterministic. + canonical.writeUInt32BE(SQLITE_STANDALONE_CHANGE_COUNTER, SQLITE_CHANGE_COUNTER_OFFSET); + canonical.writeUInt32BE(SQLITE_STANDALONE_CHANGE_COUNTER, SQLITE_VERSION_VALID_FOR_OFFSET); + return canonical; + } finally { + clone?.close(); + } +} + +function writeSqlJsSnapshot(raw: { export: () => Uint8Array }, tempPath: string): void { + let fd: number | null = null; + + try { + fd = fs.openSync(tempPath, "wx"); + fs.writeFileSync(fd, exportCanonicalSqlJsSnapshot(raw)); + fs.fsyncSync(fd); + fs.closeSync(fd); + fd = null; + } catch (error: unknown) { + if (fd !== null) { + try { + fs.closeSync(fd); + } catch { + // The original snapshot error remains authoritative. + } + } + throw error; + } +} + +function cleanupOwnedSnapshotTemp(tempDir: string | null, tempPath: string | null): void { + if (!tempDir || !fs.existsSync(tempDir)) return; + + try { + // `tempDir` comes only from mkdtempSync below. Removing that exact owned directory + // lets Node retry Windows/AV EBUSY and EPERM failures without touching canonical backups. + fs.rmSync(tempDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 25 }); + } catch (error: unknown) { + const message = error instanceof Error ? error.message : String(error); + console.warn( + `[Migration] Failed to remove owned snapshot temp directory` + + `${tempPath ? ` (${tempPath})` : ""}: ${message}` + ); + } +} + +/** + * Create a synchronous pre-migration snapshot. + * + * Native SQLite drivers use VACUUM INTO. sql.js has an in-memory VFS, so a host + * path passed to VACUUM INTO is not writable; export its current database image + * directly instead. The SHA-256 content address lives in the first portion of the + * canonical `db__.sqlite` shape, preserving reason parsing while + * making unchanged retries an O(1) lookup even with tens of thousands of old backups. + * Work happens inside an exclusively-created + * temp directory, so failure cleanup has exact ownership. Publication uses an atomic, + * no-overwrite hard link. If the filesystem cannot provide that primitive, the caller + * fails closed instead of exposing a partial canonical `.sqlite` file. A content hash + * reuses an identical prior snapshot, so repeated zero-progress startups retain one + * restore point for that database state without ever deleting a published backup. + */ +export function createPreMigrationBackup(db: SqliteAdapter): PreMigrationBackupReceipt | null { + let backupPath: string | null = null; + let tempPath: string | null = null; + let tempDir: string | null = null; + + try { + const sqliteFile = db.name; + if (!sqliteFile || sqliteFile === ":memory:") return null; + + const backupDir = path.join(path.dirname(sqliteFile), "db_backups"); + if (!fs.existsSync(backupDir)) { + fs.mkdirSync(backupDir, { recursive: true }); + fsyncDirectoryEntry(path.dirname(backupDir)); + } + + tempDir = fs.mkdtempSync(path.join(backupDir, ".migration-snapshot-")); + tempPath = path.join(tempDir, "snapshot.sqlite"); + + if (db.driver === "sql.js") { + const raw = db.raw as { export?: () => Uint8Array } | null; + if (!raw || typeof raw.export !== "function") { + throw new Error("sql.js adapter does not expose database export()"); + } + writeSqlJsSnapshot(raw as { export: () => Uint8Array }, tempPath); + } else { + const escapedTempPath = tempPath.replace(/'/g, "''"); + const snapshotDb = tryOpenSync(sqliteFile, { readonly: true, fileMustExist: true }); + if (!snapshotDb) { + throw new Error("no synchronous read-only SQLite driver is available for snapshotting"); + } + try { + snapshotDb.exec(`VACUUM INTO '${escapedTempPath}'`); + } finally { + snapshotDb.close(); + } + const fd = fs.openSync(tempPath, "r+"); + try { + fs.fsyncSync(fd); + } finally { + fs.closeSync(fd); + } + } + + const sha256 = hashFileSync(tempPath); + backupPath = path.join(backupDir, `db_state-${sha256}_pre-migration.sqlite`); + const reusable = getReusablePreMigrationBackup(backupPath, sha256); + if (reusable) { + fsyncReusableSnapshot(reusable.path); + fsyncDirectoryEntry(backupDir); + cleanupOwnedSnapshotTemp(tempDir, tempPath); + tempDir = null; + tempPath = null; + console.log(`[Migration] Reusing identical pre-migration backup: ${reusable.path}`); + return reusable; + } + + try { + publishSnapshotWithoutOverwrite(tempPath, backupPath); + } catch (error: unknown) { + if ((error as NodeJS.ErrnoException | null)?.code !== "EEXIST") throw error; + const racedReusable = getReusablePreMigrationBackup(backupPath, sha256); + if (!racedReusable) throw error; + fsyncReusableSnapshot(racedReusable.path); + fsyncDirectoryEntry(backupDir); + cleanupOwnedSnapshotTemp(tempDir, tempPath); + tempDir = null; + tempPath = null; + console.log(`[Migration] Reusing concurrently published backup: ${racedReusable.path}`); + return racedReusable; + } + cleanupOwnedSnapshotTemp(tempDir, tempPath); + tempDir = null; + tempPath = null; + console.log(`[Migration] Pre-migration backup created: ${backupPath}`); + + return { path: backupPath, sha256 }; + } catch (error: unknown) { + // Never unlink a canonical backup here: publication may have failed because another + // actor created it first. The exclusive temp directory is the only cleanup authority. + cleanupOwnedSnapshotTemp(tempDir, tempPath); + const message = error instanceof Error ? error.message : String(error); + console.warn(`[Migration] Failed to create pre-migration backup: ${message}`); + throw new Error( + `[Migration] Refusing to migrate an existing database without a durable snapshot. ` + + `Snapshot creation failed: ${message}. The DATA_DIR filesystem must support atomic ` + + `no-overwrite hard links, durable file synchronization, and directory synchronization ` + + `where the platform exposes it.`, + { cause: error instanceof Error ? error : undefined } + ); + } +} diff --git a/src/lib/db/migrationRunner/schemaState.ts b/src/lib/db/migrationRunner/schemaState.ts new file mode 100644 index 0000000000..f38e28d0b1 --- /dev/null +++ b/src/lib/db/migrationRunner/schemaState.ts @@ -0,0 +1,248 @@ +import type { SqliteAdapter } from "../adapters/types"; +import { + INITIAL_SCHEMA_SENTINELS, + LEGACY_VERSION_SLOT_MIGRATIONS, + PHYSICAL_SCHEMA_SENTINELS, + RENAMED_MIGRATION_COMPATIBILITY, +} from "./constants"; +import { migrationConsole as console } from "./logger"; + +type MigrationFile = { version: string; name: string; path: string }; + +export function hasTable(db: SqliteAdapter, tableName: string): boolean { + const row = db + .prepare("SELECT name FROM sqlite_master WHERE type IN ('table', 'view') AND name = ?") + .get(tableName) as { name?: string } | undefined; + return Boolean(row?.name); +} + +export function hasPhysicalTable(db: SqliteAdapter, tableName: string): boolean { + const row = db + .prepare("SELECT name FROM sqlite_master WHERE type = 'table' AND name = ?") + .get(tableName) as { name?: string } | undefined; + return Boolean(row?.name); +} + +export function hasColumn(db: SqliteAdapter, tableName: string, columnName: string): boolean { + const columns = db.prepare(`PRAGMA table_info(${tableName})`).all() as Array<{ name?: string }>; + return columns.some((column) => column.name === columnName); +} + +export function inferPhysicalSchemaBaseline(db: SqliteAdapter): { + version: string; + description: string; +} | null { + for (const sentinel of PHYSICAL_SCHEMA_SENTINELS) { + if (hasTable(db, sentinel.tableName)) { + return { + version: sentinel.version, + description: sentinel.description, + }; + } + } + + const hasInitialSchema = INITIAL_SCHEMA_SENTINELS.every((tableName) => hasTable(db, tableName)); + if (hasInitialSchema) { + return { + version: "001", + description: "initial schema tables", + }; + } + + return null; +} + +export function getPlausiblePendingCount(files: MigrationFile[], baselineVersion: string): number { + const baseline = Number.parseInt(baselineVersion, 10); + return files.filter((file) => Number.parseInt(file.version, 10) > baseline).length; +} + +/** + * Detect migration name mismatches — when a migration version number + * has been reused/renumbered with a different name. This is a strong signal + * that the migration tracking is corrupted or migrations were renumbered. + */ +export function detectNameMismatches( + appliedRecords: Array<{ version: string; name: string }>, + files: MigrationFile[] +): Array<{ version: string; appliedName: string; diskName: string }> { + const appliedByName = new Map(appliedRecords.map((record) => [record.version, record.name])); + const mismatches: Array<{ version: string; appliedName: string; diskName: string }> = []; + + for (const file of files) { + const appliedName = appliedByName.get(file.version); + if (appliedName && appliedName !== file.name) { + mismatches.push({ + version: file.version, + appliedName, + diskName: file.name, + }); + } + } + + return mismatches; +} + +export function reconcileRenumberedMigrations(db: SqliteAdapter, files: MigrationFile[]): boolean { + let repaired = false; + + for (const compatibility of RENAMED_MIGRATION_COMPATIBILITY) { + const hasTargetFile = files.some( + (file) => file.version === compatibility.toVersion && file.name === compatibility.toName + ); + const hasSourceFile = files.some( + (file) => file.version === compatibility.fromVersion && file.name !== compatibility.fromName + ); + + if (!hasTargetFile || !hasSourceFile) { + continue; + } + + const legacyRow = db + .prepare("SELECT version, name FROM _omniroute_migrations WHERE version = ? AND name = ?") + .get(compatibility.fromVersion, compatibility.fromName) as + { version: string; name: string } | undefined; + if (!legacyRow) { + continue; + } + + const targetRow = db + .prepare("SELECT version, name FROM _omniroute_migrations WHERE version = ?") + .get(compatibility.toVersion) as { version: string; name: string } | undefined; + + const isSameSlotReplacement = compatibility.fromVersion === compatibility.toVersion; + if (targetRow && !isSameSlotReplacement && targetRow.name !== compatibility.toName) { + throw new Error( + `[Migration] Cannot reconcile ${compatibility.fromVersion}_${compatibility.fromName}: ` + + `target version ${compatibility.toVersion} is occupied by unknown migration ` + + `"${targetRow.name}" (expected "${compatibility.toName}").` + ); + } + + const applyRepair = db.transaction(() => { + if (targetRow) { + db.prepare("DELETE FROM _omniroute_migrations WHERE version = ? AND name = ?").run( + compatibility.fromVersion, + compatibility.fromName + ); + } else { + db.prepare( + "UPDATE _omniroute_migrations SET version = ?, name = ? WHERE version = ? AND name = ?" + ).run( + compatibility.toVersion, + compatibility.toName, + compatibility.fromVersion, + compatibility.fromName + ); + } + }); + + applyRepair(); + repaired = true; + console.warn( + `[Migration] Reconciled renamed migration ${compatibility.fromVersion}_${compatibility.fromName} ` + + `to ${compatibility.toVersion}_${compatibility.toName} to preserve pending migrations.` + ); + + // After the compat rewrite, verify the old version slot is now free. + // A residual row (from a failed prior run, manual intervention, or edge-case + // UPDATE conflict) at the old version would shadow a NEW migration file + // placed at that version number — e.g. 028_create_files_and_batches.sql + // would be skipped because getAppliedVersions() still sees version "028". + const residualRow = db + .prepare("SELECT version, name FROM _omniroute_migrations WHERE version = ?") + .get(compatibility.fromVersion) as { version: string; name: string } | undefined; + if (residualRow) { + console.warn( + `[Migration] ⚠️ Residual row at version ${compatibility.fromVersion} ` + + `(name: "${residualRow.name}") still present after compat rewrite — ` + + `removing to unblock new migration at this version slot.` + ); + db.prepare("DELETE FROM _omniroute_migrations WHERE version = ?").run( + compatibility.fromVersion + ); + } + } + + return repaired; +} + +export function rehomeLegacyVersionSlotMigrations( + db: SqliteAdapter, + files: MigrationFile[] +): boolean { + let repaired = false; + const diskNamesByVersion = new Map(files.map((file) => [file.version, file.name])); + + for (const legacy of LEGACY_VERSION_SLOT_MIGRATIONS) { + const diskName = diskNamesByVersion.get(legacy.version); + if (!diskName || diskName === legacy.name) { + continue; + } + + const legacyRow = db + .prepare("SELECT version, name FROM _omniroute_migrations WHERE version = ? AND name = ?") + .get(legacy.version, legacy.name) as { version: string; name: string } | undefined; + if (!legacyRow) { + continue; + } + + const legacyVersion = `legacy-${legacy.version}-${legacy.name}`; + const applyRepair = db.transaction(() => { + const existingLegacyRow = db + .prepare("SELECT version FROM _omniroute_migrations WHERE version = ?") + .get(legacyVersion) as { version: string } | undefined; + + if (existingLegacyRow) { + db.prepare("DELETE FROM _omniroute_migrations WHERE version = ? AND name = ?").run( + legacy.version, + legacy.name + ); + return; + } + + db.prepare("UPDATE _omniroute_migrations SET version = ? WHERE version = ? AND name = ?").run( + legacyVersion, + legacy.version, + legacy.name + ); + }); + + applyRepair(); + repaired = true; + console.warn( + `[Migration] Rehomed legacy migration ${legacy.version}_${legacy.name} ` + + `to ${legacyVersion} so current ${legacy.version}_${diskName} can apply.` + ); + } + + return repaired; +} + +export function hasLedgerRepairCandidates(db: SqliteAdapter, files: MigrationFile[]): boolean { + const diskNamesByVersion = new Map(files.map((file) => [file.version, file.name])); + for (const legacy of LEGACY_VERSION_SLOT_MIGRATIONS) { + const diskName = diskNamesByVersion.get(legacy.version); + if (!diskName || diskName === legacy.name) continue; + const row = db + .prepare("SELECT 1 FROM _omniroute_migrations WHERE version = ? AND name = ?") + .get(legacy.version, legacy.name); + if (row) return true; + } + + for (const compatibility of RENAMED_MIGRATION_COMPATIBILITY) { + const hasTargetFile = files.some( + (file) => file.version === compatibility.toVersion && file.name === compatibility.toName + ); + const hasSourceFile = files.some( + (file) => file.version === compatibility.fromVersion && file.name !== compatibility.fromName + ); + if (!hasTargetFile || !hasSourceFile) continue; + const row = db + .prepare("SELECT 1 FROM _omniroute_migrations WHERE version = ? AND name = ?") + .get(compatibility.fromVersion, compatibility.fromName); + if (row) return true; + } + + return false; +} diff --git a/tests/unit/datadir-test-context-guard-10428.test.ts b/tests/unit/datadir-test-context-guard-10428.test.ts index 93fad86c73..b7bfe3ccd9 100644 --- a/tests/unit/datadir-test-context-guard-10428.test.ts +++ b/tests/unit/datadir-test-context-guard-10428.test.ts @@ -1,5 +1,6 @@ import test from "node:test"; import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; import os from "node:os"; import path from "node:path"; import fs from "node:fs"; @@ -21,6 +22,30 @@ import fs from "node:fs"; */ const { resolveWritableDataDir, getDefaultDataDir } = await import("../../src/lib/dataPaths.ts"); +const redirectedDirs = new Set(); + +function assertOwnedRedirectDir(candidate: string): string { + const resolved = path.resolve(candidate); + const tempRoot = path.resolve(os.tmpdir()); + assert.ok( + resolved.startsWith(`${tempRoot}${path.sep}`) && + path.basename(resolved).startsWith("omniroute-testctx-"), + `refusing to treat a non-owned path as a test redirect: ${resolved}` + ); + return resolved; +} + +function rememberRedirectDir(candidate: string): string { + const resolved = assertOwnedRedirectDir(candidate); + redirectedDirs.add(resolved); + return resolved; +} + +test.after(() => { + for (const redirected of redirectedDirs) { + fs.rmSync(redirected, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + } +}); function withEnv(overrides: Record, run: () => void) { const saved: Record = {}; @@ -39,11 +64,58 @@ function withEnv(overrides: Record, run: () => void) } } +const EVAL_PROBE_SCRIPT = + "import('./src/lib/dataPaths.ts').then(({ resolveWritableDataDir }) => " + + "console.log('OMNIROUTE_TEST_DATA_DIR=' + resolveWritableDataDir()))"; + +function assertEvalProbeIsIsolated(evalArgs: string[], configuredDataDir = "") { + const result = spawnSync(process.execPath, ["--import", "tsx/esm", ...evalArgs], { + cwd: process.cwd(), + encoding: "utf8", + env: { + ...process.env, + DATA_DIR: configuredDataDir, + XDG_CONFIG_HOME: "", + NODE_ENV: "production", + NODE_TEST_CONTEXT: "", + VITEST: "", + OMNIROUTE_ALLOW_DEFAULT_DATA_DIR: "", + }, + }); + + assert.equal(result.status, 0, result.stderr); + const outputLine = result.stdout + .trim() + .split("\n") + .find((line) => line.startsWith("OMNIROUTE_TEST_DATA_DIR=")); + const resolved = outputLine?.slice("OMNIROUTE_TEST_DATA_DIR=".length) ?? ""; + const ownedRedirect = assertOwnedRedirectDir(resolved); + try { + assert.notEqual( + ownedRedirect, + path.join(os.homedir(), ".omniroute"), + "an eval/import probe must not inherit the normal server's default database" + ); + assert.equal( + fs.existsSync(ownedRedirect), + false, + "the child exit handler must remove its exact redirected DATA_DIR" + ); + } finally { + fs.rmSync(ownedRedirect, { + recursive: true, + force: true, + maxRetries: 5, + retryDelay: 100, + }); + } +} + test("G1: a test context with no DATA_DIR never resolves to the operator's real data dir", () => { withEnv( { DATA_DIR: undefined, NODE_ENV: "test", OMNIROUTE_ALLOW_DEFAULT_DATA_DIR: undefined }, () => { - const resolved = resolveWritableDataDir(); + const resolved = rememberRedirectDir(resolveWritableDataDir()); assert.notEqual( resolved, getDefaultDataDir(), @@ -101,7 +173,7 @@ test("G5: node:test subprocesses are detected through NODE_TEST_CONTEXT too", () OMNIROUTE_ALLOW_DEFAULT_DATA_DIR: undefined, }, () => { - const resolved = resolveWritableDataDir(); + const resolved = rememberRedirectDir(resolveWritableDataDir()); assert.notEqual(resolved, getDefaultDataDir()); assert.ok(resolved.startsWith(os.tmpdir())); } @@ -110,8 +182,24 @@ test("G5: node:test subprocesses are detected through NODE_TEST_CONTEXT too", () test("G6: the redirect is stable within a process (same dir on repeated calls)", () => { withEnv({ DATA_DIR: undefined, NODE_ENV: "test" }, () => { - const first = resolveWritableDataDir(); - const second = resolveWritableDataDir(); + const first = rememberRedirectDir(resolveWritableDataDir()); + const second = rememberRedirectDir(resolveWritableDataDir()); assert.equal(first, second, "a per-call temp dir would split the DB across handles"); }); }); + +test("G7: a node --eval probe without DATA_DIR is isolated from the operator home", () => { + assertEvalProbeIsIsolated(["--eval", EVAL_PROBE_SCRIPT]); +}); + +test("G8: the single-argument --eval= form is isolated too", () => { + assertEvalProbeIsIsolated([`--eval=${EVAL_PROBE_SCRIPT}`]); +}); + +test("G9: whitespace DATA_DIR is absent for a node -e probe", () => { + assertEvalProbeIsIsolated(["-e", EVAL_PROBE_SCRIPT], " "); +}); + +test("G10: a combined node -pe probe is isolated too", () => { + assertEvalProbeIsIsolated(["-pe", EVAL_PROBE_SCRIPT]); +}); diff --git a/tests/unit/db-backup-extended.test.ts b/tests/unit/db-backup-extended.test.ts index 4ed9a08064..b0ac9e4f6b 100644 --- a/tests/unit/db-backup-extended.test.ts +++ b/tests/unit/db-backup-extended.test.ts @@ -97,6 +97,32 @@ test("backupDbFile creates manual backups and listDbBackups returns metadata", a assert.equal(fs.existsSync(backupPath), true); }); +test("listDbBackups orders mixed timestamp and content-addressed names by mtime", async () => { + seedConnections(2); + fs.mkdirSync(core.DB_BACKUPS_DIR, { recursive: true }); + + const lexicallyFutureButOld = "db_2099-01-01T00-00-00-000Z_manual.sqlite"; + const timestampMiddle = "db_2026-09-02T00-00-00-000Z_manual.sqlite"; + const contentAddressedNewest = `db_state-${"a".repeat(64)}_pre-migration.sqlite`; + for (const filename of [lexicallyFutureButOld, timestampMiddle, contentAddressedNewest]) { + await core.getDbInstance().backup(path.join(core.DB_BACKUPS_DIR, filename)); + } + + const now = Date.now() / 1000; + fs.utimesSync(path.join(core.DB_BACKUPS_DIR, lexicallyFutureButOld), now - 120, now - 120); + fs.utimesSync(path.join(core.DB_BACKUPS_DIR, timestampMiddle), now - 60, now - 60); + fs.utimesSync(path.join(core.DB_BACKUPS_DIR, contentAddressedNewest), now, now); + + const backups = await backupDb.listDbBackups(); + assert.deepEqual( + backups.map((backup) => backup.id), + [contentAddressedNewest, timestampMiddle, lexicallyFutureButOld], + "content-addressed migration snapshots must not make filename order masquerade as recency" + ); + assert.equal(backups[0]?.reason, "pre-migration"); + assert.equal(backups[0]?.connectionCount, 2); +}); + test("listDbBackups returns an empty list when the backup directory is missing", async () => { fs.rmSync(core.DB_BACKUPS_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); const backups = await backupDb.listDbBackups(); diff --git a/tests/unit/db-fresh-setup-9934.test.ts b/tests/unit/db-fresh-setup-9934.test.ts index 2330eedb6e..984c3a9f10 100644 --- a/tests/unit/db-fresh-setup-9934.test.ts +++ b/tests/unit/db-fresh-setup-9934.test.ts @@ -101,6 +101,13 @@ test( const cli = await importFresh("bin/cli/sqlite.mjs"); const setup = await cli.openOmniRouteDb(); assert.ok(fs.existsSync(setup.dbPath), "setup created storage.sqlite"); + setup.db + .prepare( + `INSERT INTO provider_connections + (id, provider, created_at, updated_at) + VALUES (?, ?, ?, ?)` + ) + .run("setup-provider", "openai", "2026-09-02T00:00:00.000Z", "2026-09-02T00:00:00.000Z"); setup.db.close(); const onDisk = new Database(setup.dbPath, { readonly: true }); @@ -139,6 +146,27 @@ test( (maxRow?.maxV ?? 0) > 1, `expected migrations beyond 001 to run, got max=${maxRow?.maxV}` ); + + const backupDir = path.join(dataDir, "db_backups"); + const snapshots = fs + .readdirSync(backupDir) + .filter((name) => /^db_state-[a-f0-9]{64}_pre-migration\.sqlite$/.test(name)); + assert.equal( + snapshots.length, + 1, + "a setup-created file is logically fresh for the mass guard but physically existing for snapshot safety" + ); + + const snapshot = new Database(path.join(backupDir, snapshots[0]!), { readonly: true }); + try { + assert.deepEqual( + snapshot.prepare("SELECT id, provider FROM provider_connections").get(), + { id: "setup-provider", provider: "openai" }, + "the mandatory snapshot must preserve setup-created provider state" + ); + } finally { + snapshot.close(); + } } finally { if (originalDataDir === undefined) delete process.env.DATA_DIR; else process.env.DATA_DIR = originalDataDir; diff --git a/tests/unit/db-migration-missing-physical-schema.test.ts b/tests/unit/db-migration-missing-physical-schema.test.ts new file mode 100644 index 0000000000..7f3c86d6da --- /dev/null +++ b/tests/unit/db-migration-missing-physical-schema.test.ts @@ -0,0 +1,839 @@ +// ENVIRONMENT NOTE (sandbox better-sqlite3 / glibc limitation, not a code defect): +// This test constructs a real better-sqlite3 database. Production and CI load the +// native addon normally; see tests/unit/_helpers/betterSqlite3Availability.ts for +// the documented fallback context on older sandboxes. +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { spawnSync } from "node:child_process"; +import test from "node:test"; +import { fileURLToPath } from "node:url"; + +import Database from "better-sqlite3"; + +const isIsolatedChild = process.env.OMNIROUTE_DB_MIGRATION_SAFETY_CHILD === "1"; + +if (!isIsolatedChild) { + test("historical migration repair scenarios pass in an isolated process", () => { + const dataDir = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-schema-repair-data-")); + const migrationsDir = fs.mkdtempSync( + path.join(os.tmpdir(), "omniroute-schema-repair-migrations-") + ); + + try { + const childEnv = { + ...process.env, + DATA_DIR: dataDir, + OMNIROUTE_DB_MIGRATION_SAFETY_CHILD: "1", + OMNIROUTE_MAX_PENDING_MIGRATIONS: "", + OMNIROUTE_MIGRATIONS_DIR: migrationsDir, + }; + // Node's test runner exports this only to the current test worker. Passing it into + // another `node --test` process makes Node classify the nested file as recursive and + // skip every subtest while returning exit 0 — a dangerous false green. + delete childEnv.NODE_TEST_CONTEXT; + + const result = spawnSync( + process.execPath, + ["--import", "tsx/esm", "--test", fileURLToPath(import.meta.url)], + { + cwd: process.cwd(), + encoding: "utf8", + env: childEnv, + } + ); + + assert.equal( + result.status, + 0, + `isolated migration regressions failed\nstdout:\n${result.stdout}\nstderr:\n${result.stderr}` + ); + assert.match(result.stdout, /\btests 13\b/, "the isolated child must execute all subtests"); + assert.match(result.stdout, /\bpass 13\b/, "the isolated child must pass all subtests"); + } finally { + fs.rmSync(dataDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + fs.rmSync(migrationsDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + } + }); +} else { + const dataDir = process.env.DATA_DIR; + const migrationsDir = process.env.OMNIROUTE_MIGRATIONS_DIR; + assert.ok(dataDir, "isolated child requires an explicit DATA_DIR"); + assert.ok(migrationsDir, "isolated child requires an explicit migrations directory"); + const discoveryMigrationSql = fs.readFileSync( + path.resolve("src/lib/db/migrations/074_discovery_results.sql"), + "utf8" + ); + + fs.writeFileSync( + path.join(migrationsDir, "074_discovery_results.sql"), + discoveryMigrationSql, + "utf8" + ); + fs.writeFileSync( + path.join(migrationsDir, "081_inspector_custom_hosts.sql"), + ` + CREATE TABLE IF NOT EXISTS inspector_custom_hosts ( + host TEXT PRIMARY KEY, + enabled INTEGER NOT NULL DEFAULT 1 + ); + CREATE INDEX IF NOT EXISTS idx_inspector_custom_hosts_enabled + ON inspector_custom_hosts(enabled); + `, + "utf8" + ); + fs.writeFileSync( + path.join(migrationsDir, "151_windsurf_to_devin_desktop.sql"), + "UPDATE discovery_results SET provider_id = 'devin-desktop' WHERE provider_id = 'windsurf';", + "utf8" + ); + fs.writeFileSync( + path.join(migrationsDir, "152_remove_puter_provider.sql"), + "DELETE FROM discovery_results WHERE provider_id = 'puter';", + "utf8" + ); + + const { runMigrations } = await import("../../src/lib/db/migrationRunner.ts"); + + function listPreMigrationBackups(): string[] { + const backupDir = path.join(dataDir, "db_backups"); + if (!fs.existsSync(backupDir)) return []; + return fs + .readdirSync(backupDir) + .filter((name) => name.endsWith("_pre-migration.sqlite")) + .sort(); + } + + function withNonTestEnvironment(fn: () => T): T { + const previousNodeEnv = process.env.NODE_ENV; + const previousVitest = process.env.VITEST; + const previousArgv = [...process.argv]; + const previousExecArgv = [...process.execArgv]; + + delete process.env.NODE_ENV; + delete process.env.VITEST; + process.argv = process.argv.filter((arg) => !arg.includes("test")); + process.execArgv = process.execArgv.filter((arg) => !arg.includes("test")); + + try { + return fn(); + } finally { + process.argv = previousArgv; + process.execArgv = previousExecArgv; + if (previousNodeEnv === undefined) delete process.env.NODE_ENV; + else process.env.NODE_ENV = previousNodeEnv; + if (previousVitest === undefined) delete process.env.VITEST; + else process.env.VITEST = previousVitest; + } + } + + test.after(() => { + // The parent owns both explicit temp directories and removes them after this + // process exits. Keeping ownership there also covers child startup failures. + }); + + test("runner repairs the 074 inspector collision before migrations 151 and 152", () => { + const db = new Database(":memory:"); + + try { + db.exec(` + CREATE TABLE _omniroute_migrations ( + version TEXT PRIMARY KEY, + name TEXT NOT NULL, + applied_at TEXT NOT NULL DEFAULT (datetime('now')) + ); + CREATE TABLE inspector_custom_hosts ( + host TEXT PRIMARY KEY, + enabled INTEGER NOT NULL DEFAULT 1 + ); + INSERT INTO inspector_custom_hosts (host, enabled) + VALUES ('api.example.test', 1); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('074', 'inspector_custom_hosts'); + `); + + assert.equal( + db + .prepare( + "SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'discovery_results'" + ) + .get(), + undefined, + "precondition: the collided 074 marker hides the missing discovery_results table" + ); + + assert.equal(runMigrations(db as never), 3); + + assert.ok( + db + .prepare( + "SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'discovery_results'" + ) + .get(), + "074 must be replayed before migrations 151 and 152 reference discovery_results" + ); + assert.deepEqual( + db.prepare("SELECT version, name FROM _omniroute_migrations ORDER BY version").all(), + [ + { version: "074", name: "discovery_results" }, + { version: "081", name: "inspector_custom_hosts" }, + { version: "151", name: "windsurf_to_devin_desktop" }, + { version: "152", name: "remove_puter_provider" }, + ] + ); + assert.deepEqual( + db.prepare("SELECT host, enabled FROM inspector_custom_hosts").get(), + { host: "api.example.test", enabled: 1 }, + "re-homing the inspector marker to 081 must preserve the existing table data" + ); + assert.equal(runMigrations(db as never), 0, "the repaired state must be idempotent"); + } finally { + db.close(); + } + }); + + test("runner rehomes a collided 074 inspector marker even when both tables exist", () => { + const db = new Database(":memory:"); + + try { + db.exec(discoveryMigrationSql); + db.exec(` + CREATE TABLE _omniroute_migrations ( + version TEXT PRIMARY KEY, + name TEXT NOT NULL, + applied_at TEXT NOT NULL DEFAULT (datetime('now')) + ); + CREATE TABLE inspector_custom_hosts ( + host TEXT PRIMARY KEY, + enabled INTEGER NOT NULL DEFAULT 1 + ); + INSERT INTO inspector_custom_hosts (host, enabled) + VALUES ('api.example.test', 1); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('074', 'inspector_custom_hosts'); + `); + + assert.equal(runMigrations(db as never), 3); + assert.deepEqual( + db.prepare("SELECT version, name FROM _omniroute_migrations ORDER BY version").all(), + [ + { version: "074", name: "discovery_results" }, + { version: "081", name: "inspector_custom_hosts" }, + { version: "151", name: "windsurf_to_devin_desktop" }, + { version: "152", name: "remove_puter_provider" }, + ], + "the old 074 name must not remain as a permanent CRITICAL mismatch" + ); + assert.deepEqual(db.prepare("SELECT host FROM inspector_custom_hosts").get(), { + host: "api.example.test", + }); + } finally { + db.close(); + } + }); + + test("runner rebuilds both collided tables when neither physical table survived", () => { + const db = new Database(":memory:"); + + try { + db.exec(` + CREATE TABLE _omniroute_migrations ( + version TEXT PRIMARY KEY, + name TEXT NOT NULL, + applied_at TEXT NOT NULL DEFAULT (datetime('now')) + ); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('074', 'inspector_custom_hosts'); + `); + + assert.equal(runMigrations(db as never), 4); + assert.ok( + db + .prepare( + "SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'discovery_results'" + ) + .get(), + "the canonical 074 table must be restored" + ); + assert.ok( + db + .prepare( + "SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'inspector_custom_hosts'" + ) + .get(), + "the rehomed 081 marker must not hide a missing inspector table" + ); + assert.deepEqual( + db.prepare("SELECT version, name FROM _omniroute_migrations ORDER BY version").all(), + [ + { version: "074", name: "discovery_results" }, + { version: "081", name: "inspector_custom_hosts" }, + { version: "151", name: "windsurf_to_devin_desktop" }, + { version: "152", name: "remove_puter_provider" }, + ] + ); + } finally { + db.close(); + } + }); + + test("runner atomically replays 081 when its marker exists without the inspector table", () => { + const db = new Database(":memory:"); + + try { + db.exec(discoveryMigrationSql); + db.exec(` + CREATE TABLE _omniroute_migrations ( + version TEXT PRIMARY KEY, + name TEXT NOT NULL, + applied_at TEXT NOT NULL DEFAULT (datetime('now')) + ); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('074', 'discovery_results'); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('081', 'inspector_custom_hosts'); + `); + + assert.equal(runMigrations(db as never), 3); + assert.ok( + db + .prepare( + "SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'inspector_custom_hosts'" + ) + .get(), + "a valid 081 marker must be replayed when its physical table is absent" + ); + assert.deepEqual( + db.prepare("SELECT version, name FROM _omniroute_migrations ORDER BY version").all(), + [ + { version: "074", name: "discovery_results" }, + { version: "081", name: "inspector_custom_hosts" }, + { version: "151", name: "windsurf_to_devin_desktop" }, + { version: "152", name: "remove_puter_provider" }, + ] + ); + } finally { + db.close(); + } + }); + + test("runner fails closed when target 081 has unknown provenance", () => { + const db = new Database(":memory:"); + + try { + db.exec(` + CREATE TABLE inspector_custom_hosts ( + host TEXT PRIMARY KEY, + enabled INTEGER NOT NULL DEFAULT 1 + ); + CREATE TABLE _omniroute_migrations ( + version TEXT PRIMARY KEY, + name TEXT NOT NULL, + applied_at TEXT NOT NULL DEFAULT (datetime('now')) + ); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('074', 'inspector_custom_hosts'); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('081', 'unknown_historical_migration'); + `); + + assert.throws( + () => runMigrations(db as never), + /target version 081 is occupied by unknown migration/i + ); + assert.deepEqual( + db.prepare("SELECT version, name FROM _omniroute_migrations ORDER BY version").all(), + [ + { version: "074", name: "inspector_custom_hosts" }, + { version: "081", name: "unknown_historical_migration" }, + ], + "a target collision must preserve both provenance records" + ); + } finally { + db.close(); + } + }); + + test("runner rejects an unknown 074 marker even when all later migrations are marked", () => { + const db = new Database(":memory:"); + + try { + db.exec(` + CREATE TABLE _omniroute_migrations ( + version TEXT PRIMARY KEY, + name TEXT NOT NULL, + applied_at TEXT NOT NULL DEFAULT (datetime('now')) + ); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('074', 'unknown_historical_migration'); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('081', 'inspector_custom_hosts'); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('151', 'windsurf_to_devin_desktop'); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('152', 'remove_puter_provider'); + `); + + assert.throws( + () => runMigrations(db as never), + /required table "discovery_results" is missing.*unknown migration/i, + "unknown provenance must fail closed instead of being silently rewritten" + ); + assert.deepEqual( + db.prepare("SELECT version, name FROM _omniroute_migrations ORDER BY version").all(), + [ + { version: "074", name: "unknown_historical_migration" }, + { version: "081", name: "inspector_custom_hosts" }, + { version: "151", name: "windsurf_to_devin_desktop" }, + { version: "152", name: "remove_puter_provider" }, + ] + ); + } finally { + db.close(); + } + }); + + test("runner backs up an existing DB before reopening its only applied marker", () => { + const sqlitePath = path.join(dataDir, "only-marker.sqlite"); + const db = new Database(sqlitePath); + const previousDisableBackup = process.env.DISABLE_SQLITE_AUTO_BACKUP; + + try { + delete process.env.DISABLE_SQLITE_AUTO_BACKUP; + db.exec(` + CREATE TABLE provider_connections (id TEXT PRIMARY KEY); + CREATE TABLE _omniroute_migrations ( + version TEXT PRIMARY KEY, + name TEXT NOT NULL, + applied_at TEXT NOT NULL DEFAULT (datetime('now')) + ); + INSERT INTO provider_connections (id) VALUES ('existing-data'); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('074', 'discovery_results'); + `); + + assert.equal(runMigrations(db as never), 4); + + const backupDir = path.join(dataDir, "db_backups"); + const backups = fs + .readdirSync(backupDir) + .filter((name) => name.endsWith("_pre-migration.sqlite")); + assert.equal( + backups.length, + 1, + "removing the only marker must not make an existing DB look fresh and skip its snapshot" + ); + } finally { + db.close(); + if (previousDisableBackup === undefined) delete process.env.DISABLE_SQLITE_AUTO_BACKUP; + else process.env.DISABLE_SQLITE_AUTO_BACKUP = previousDisableBackup; + } + }); + + test("snapshot publication never deletes a raced final path", () => { + const sqlitePath = path.join(dataDir, "snapshot-publish-race.sqlite"); + const db = new Database(sqlitePath); + const previousDisableBackup = process.env.DISABLE_SQLITE_AUTO_BACKUP; + const originalLinkSync = fs.linkSync; + let racedFinalPath: string | null = null; + + try { + delete process.env.DISABLE_SQLITE_AUTO_BACKUP; + db.exec(` + CREATE TABLE _omniroute_migrations ( + version TEXT PRIMARY KEY, + name TEXT NOT NULL, + applied_at TEXT NOT NULL DEFAULT (datetime('now')) + ); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('074', 'discovery_results'); + `); + + fs.linkSync = ((_existingPath: fs.PathLike, newPath: fs.PathLike) => { + racedFinalPath = String(newPath); + fs.writeFileSync(racedFinalPath, "third-party-sentinel"); + throw Object.assign(new Error("destination already exists"), { code: "EEXIST" }); + }) as typeof fs.linkSync; + + assert.throws( + () => runMigrations(db as never), + /without a durable snapshot/, + "a raced final name must fail closed before atomic replay" + ); + assert.ok(racedFinalPath); + assert.equal( + fs.readFileSync(racedFinalPath, "utf8"), + "third-party-sentinel", + "snapshot failure cleanup must never unlink another actor's final path" + ); + assert.deepEqual(db.prepare("SELECT version, name FROM _omniroute_migrations").all(), [ + { version: "074", name: "discovery_results" }, + ]); + } finally { + fs.linkSync = originalLinkSync; + if (racedFinalPath && fs.existsSync(racedFinalPath)) fs.unlinkSync(racedFinalPath); + db.close(); + if (previousDisableBackup === undefined) delete process.env.DISABLE_SQLITE_AUTO_BACKUP; + else process.env.DISABLE_SQLITE_AUTO_BACKUP = previousDisableBackup; + } + }); + + test("snapshot publication fails closed when hard links are unsupported", () => { + const sqlitePath = path.join(dataDir, "snapshot-publish-fallback.sqlite"); + const db = new Database(sqlitePath); + const previousDisableBackup = process.env.DISABLE_SQLITE_AUTO_BACKUP; + const originalLinkSync = fs.linkSync; + const backupsBefore = listPreMigrationBackups(); + + try { + delete process.env.DISABLE_SQLITE_AUTO_BACKUP; + db.exec(` + CREATE TABLE _omniroute_migrations ( + version TEXT PRIMARY KEY, + name TEXT NOT NULL, + applied_at TEXT NOT NULL DEFAULT (datetime('now')) + ); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('074', 'discovery_results'); + `); + fs.linkSync = (() => { + throw Object.assign(new Error("hard links unsupported"), { code: "ENOTSUP" }); + }) as typeof fs.linkSync; + + assert.throws( + () => runMigrations(db as never), + /durable snapshot.*hard links unsupported.*hard links.*synchronization/is + ); + assert.deepEqual(db.prepare("SELECT version, name FROM _omniroute_migrations").all(), [ + { version: "074", name: "discovery_results" }, + ]); + assert.equal( + db + .prepare( + "SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'discovery_results'" + ) + .get(), + undefined + ); + assert.deepEqual(listPreMigrationBackups(), backupsBefore); + } finally { + fs.linkSync = originalLinkSync; + db.close(); + if (previousDisableBackup === undefined) delete process.env.DISABLE_SQLITE_AUTO_BACKUP; + else process.env.DISABLE_SQLITE_AUTO_BACKUP = previousDisableBackup; + } + }); + + test("an only-marker repair cannot disarm the mass-migration barrier on retry", () => { + const sqlitePath = path.join(dataDir, "only-marker-mass-safety.sqlite"); + const db = new Database(sqlitePath); + const previousDisableBackup = process.env.DISABLE_SQLITE_AUTO_BACKUP; + const previousMaxPending = process.env.OMNIROUTE_MAX_PENDING_MIGRATIONS; + + try { + process.env.DISABLE_SQLITE_AUTO_BACKUP = "true"; + process.env.OMNIROUTE_MAX_PENDING_MIGRATIONS = "1"; + db.exec(` + CREATE TABLE provider_connections (id TEXT PRIMARY KEY); + CREATE TABLE inspector_custom_hosts ( + host TEXT PRIMARY KEY, + enabled INTEGER NOT NULL DEFAULT 1 + ); + CREATE TABLE _omniroute_migrations ( + version TEXT PRIMARY KEY, + name TEXT NOT NULL, + applied_at TEXT NOT NULL DEFAULT (datetime('now')) + ); + INSERT INTO provider_connections (id) VALUES ('existing-data'); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('074', 'inspector_custom_hosts'); + `); + + const runOnce = () => withNonTestEnvironment(() => runMigrations(db as never)); + const backupsBefore = listPreMigrationBackups(); + + assert.throws(runOnce, /threshold is 1/i); + assert.deepEqual( + db.prepare("SELECT version, name FROM _omniroute_migrations").all(), + [{ version: "074", name: "inspector_custom_hosts" }], + "an abort must restore the marker that was rehomed to calculate the real pending set" + ); + const afterFirstAbort = listPreMigrationBackups(); + const created = afterFirstAbort.filter((name) => !backupsBefore.includes(name)); + assert.equal(created.length, 1, "the first abort must retain one restore point"); + assert.match(created[0]!, /^db_state-[a-f0-9]{64}_pre-migration\.sqlite$/); + + assert.throws(runOnce, /threshold is 1/i); + assert.deepEqual( + db.prepare("SELECT version, name FROM _omniroute_migrations").all(), + [{ version: "074", name: "inspector_custom_hosts" }], + "the second startup must hit the same barrier instead of treating the DB as fresh" + ); + assert.deepEqual( + listPreMigrationBackups(), + afterFirstAbort, + "the identical retry must reuse the first content-addressed snapshot" + ); + } finally { + db.close(); + if (previousDisableBackup === undefined) delete process.env.DISABLE_SQLITE_AUTO_BACKUP; + else process.env.DISABLE_SQLITE_AUTO_BACKUP = previousDisableBackup; + if (previousMaxPending === undefined) delete process.env.OMNIROUTE_MAX_PENDING_MIGRATIONS; + else process.env.OMNIROUTE_MAX_PENDING_MIGRATIONS = previousMaxPending; + } + }); + + test("a failed atomic 074 replay restores its marker and does not churn snapshots", () => { + const sqlitePath = path.join(dataDir, "failed-atomic-replay.sqlite"); + const db = new Database(sqlitePath); + const previousDisableBackup = process.env.DISABLE_SQLITE_AUTO_BACKUP; + + try { + delete process.env.DISABLE_SQLITE_AUTO_BACKUP; + db.exec(` + CREATE TABLE _omniroute_migrations ( + version TEXT PRIMARY KEY, + name TEXT NOT NULL, + applied_at TEXT NOT NULL DEFAULT (datetime('now')) + ); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('074', 'discovery_results'); + CREATE TRIGGER block_migration_ledger_replay + BEFORE INSERT ON _omniroute_migrations + WHEN NEW.version = '074' + BEGIN + SELECT RAISE(ABORT, 'ledger replay blocked'); + END; + `); + + const runOnce = () => runMigrations(db as never); + const backupsBefore = listPreMigrationBackups(); + + assert.throws(runOnce, /ledger replay blocked/); + assert.deepEqual(db.prepare("SELECT version, name FROM _omniroute_migrations").all(), [ + { version: "074", name: "discovery_results" }, + ]); + assert.equal( + db + .prepare( + "SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'discovery_results'" + ) + .get(), + undefined, + "the table creation and marker replacement must roll back together" + ); + const afterFirstFailure = listPreMigrationBackups(); + const created = afterFirstFailure.filter((name) => !backupsBefore.includes(name)); + assert.equal(created.length, 1, "the first failed replay must retain one restore point"); + assert.match(created[0]!, /^db_state-[a-f0-9]{64}_pre-migration\.sqlite$/); + + assert.throws(runOnce, /ledger replay blocked/); + assert.deepEqual( + listPreMigrationBackups(), + afterFirstFailure, + "the identical failed replay must reuse its content-addressed restore point" + ); + } finally { + db.close(); + if (previousDisableBackup === undefined) delete process.env.DISABLE_SQLITE_AUTO_BACKUP; + else process.env.DISABLE_SQLITE_AUTO_BACKUP = previousDisableBackup; + } + }); + + test("sql.js rolls ledger repairs back when the mass-migration barrier aborts", async () => { + const sqlitePath = path.join(dataDir, "sqljs-mass-safety.sqlite"); + const { createSqlJsAdapter } = await import("../../src/lib/db/adapters/sqljsAdapter.ts"); + const db = await createSqlJsAdapter(sqlitePath); + const previousMaxPending = process.env.OMNIROUTE_MAX_PENDING_MIGRATIONS; + const backupsBefore = listPreMigrationBackups(); + + try { + process.env.OMNIROUTE_MAX_PENDING_MIGRATIONS = "1"; + db.exec(` + CREATE TABLE provider_connections (id TEXT PRIMARY KEY); + CREATE TABLE inspector_custom_hosts ( + host TEXT PRIMARY KEY, + enabled INTEGER NOT NULL DEFAULT 1 + ); + CREATE TABLE _omniroute_migrations ( + version TEXT PRIMARY KEY, + name TEXT NOT NULL, + applied_at TEXT NOT NULL DEFAULT (datetime('now')) + ); + INSERT INTO provider_connections (id) VALUES ('existing-data'); + INSERT INTO inspector_custom_hosts (host) VALUES ('api.example.test'); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('074', 'inspector_custom_hosts'); + `); + + const runOnce = () => withNonTestEnvironment(() => runMigrations(db)); + const expectedLedger = [{ version: "074", name: "inspector_custom_hosts" }]; + + assert.throws(runOnce, /threshold is 1/i); + assert.deepEqual( + db.prepare("SELECT version, name FROM _omniroute_migrations ORDER BY version").all(), + expectedLedger, + "sql.js must roll the compatibility repair back with the safety savepoint" + ); + const afterFirstAbort = listPreMigrationBackups(); + const created = afterFirstAbort.filter((name) => !backupsBefore.includes(name)); + assert.equal(created.length, 1, "the first sql.js abort must retain one host snapshot"); + assert.match(created[0]!, /^db_state-[a-f0-9]{64}_pre-migration\.sqlite$/); + + assert.throws(runOnce, /threshold is 1/i); + assert.deepEqual( + db.prepare("SELECT version, name FROM _omniroute_migrations ORDER BY version").all(), + expectedLedger, + "a retry must see the same original ledger rather than committed repair residue" + ); + assert.deepEqual( + listPreMigrationBackups(), + afterFirstAbort, + "the identical sql.js abort must reuse its content-addressed snapshot" + ); + } finally { + db.close(); + if (previousMaxPending === undefined) delete process.env.OMNIROUTE_MAX_PENDING_MIGRATIONS; + else process.env.OMNIROUTE_MAX_PENDING_MIGRATIONS = previousMaxPending; + } + }); + + test("sql.js exports a real host snapshot before replaying 074", async () => { + const sqlitePath = path.join(dataDir, "sqljs-physical-replay.sqlite"); + const { createSqlJsAdapter } = await import("../../src/lib/db/adapters/sqljsAdapter.ts"); + const db = await createSqlJsAdapter(sqlitePath); + const previousDisableBackup = process.env.DISABLE_SQLITE_AUTO_BACKUP; + const backupsBefore = listPreMigrationBackups(); + + try { + delete process.env.DISABLE_SQLITE_AUTO_BACKUP; + db.exec(` + PRAGMA user_version = 42; + PRAGMA application_id = 1337; + CREATE TABLE _omniroute_migrations ( + version TEXT PRIMARY KEY, + name TEXT NOT NULL, + applied_at TEXT NOT NULL DEFAULT (datetime('now')) + ); + INSERT INTO _omniroute_migrations (version, name) + VALUES ('074', 'discovery_results'); + CREATE TRIGGER block_sqljs_ledger_replay + BEFORE INSERT ON _omniroute_migrations + WHEN NEW.version = '074' + BEGIN + SELECT RAISE(ABORT, 'sqljs ledger replay blocked'); + END; + `); + + assert.throws(() => runMigrations(db), /sqljs ledger replay blocked/); + assert.deepEqual(db.prepare("SELECT version, name FROM _omniroute_migrations").all(), [ + { version: "074", name: "discovery_results" }, + ]); + assert.equal( + db + .prepare( + "SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'discovery_results'" + ) + .get(), + undefined, + "sql.js must roll back the table and marker replacement together" + ); + const afterFirstFailure = listPreMigrationBackups(); + const firstCreated = afterFirstFailure.filter((name) => !backupsBefore.includes(name)); + assert.equal(firstCreated.length, 1, "sql.js must retain one host restore point"); + + assert.throws(() => runMigrations(db), /sqljs ledger replay blocked/); + assert.deepEqual( + listPreMigrationBackups(), + afterFirstFailure, + "the identical sql.js failure must reuse its content-addressed snapshot" + ); + + db.exec("DROP TRIGGER block_sqljs_ledger_replay"); + assert.equal(runMigrations(db), 4); + + const created = listPreMigrationBackups().filter((name) => !backupsBefore.includes(name)); + assert.equal( + created.length, + 2, + `dropping the trigger changes the DB state and must create a second snapshot: ${created}` + ); + + const snapshot = new Database(path.join(dataDir, "db_backups", created[0]!), { + readonly: true, + }); + try { + assert.equal(snapshot.pragma("integrity_check", { simple: true }), "ok"); + assert.equal(snapshot.pragma("user_version", { simple: true }), 42); + assert.equal(snapshot.pragma("application_id", { simple: true }), 1337); + const snapshotBytes = fs.readFileSync(path.join(dataDir, "db_backups", created[0]!)); + assert.equal(snapshotBytes.readUInt32BE(24), 1); + assert.equal( + snapshotBytes.readUInt32BE(92), + snapshotBytes.readUInt32BE(24), + "the normalized SQLite change counter and version-valid-for fields must agree" + ); + assert.deepEqual( + snapshot.prepare("SELECT version, name FROM _omniroute_migrations").all(), + [{ version: "074", name: "discovery_results" }] + ); + assert.equal( + snapshot + .prepare( + "SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'discovery_results'" + ) + .get(), + undefined, + "the snapshot must contain the complete pre-replay image" + ); + } finally { + snapshot.close(); + } + + const { listDbBackups } = await import("../../src/lib/db/backup.ts"); + const listed = await listDbBackups(); + assert.equal( + listed.find((backup) => backup.id === created[0])?.reason, + "pre-migration", + "the content address must not change the public backup reason" + ); + + db.close(); + const reopened = await createSqlJsAdapter(sqlitePath); + try { + assert.deepEqual( + reopened + .prepare("SELECT version, name FROM _omniroute_migrations ORDER BY version") + .all(), + [ + { version: "074", name: "discovery_results" }, + { version: "081", name: "inspector_custom_hosts" }, + { version: "151", name: "windsurf_to_devin_desktop" }, + { version: "152", name: "remove_puter_provider" }, + ] + ); + assert.ok( + reopened + .prepare( + "SELECT name FROM sqlite_master WHERE type = 'index' AND name = 'idx_discovery_results_provider'" + ) + .get() + ); + assert.ok( + reopened + .prepare( + "SELECT name FROM sqlite_master WHERE type = 'index' AND name = 'idx_discovery_results_status'" + ) + .get() + ); + } finally { + reopened.close(); + } + } finally { + if (db.open) db.close(); + if (previousDisableBackup === undefined) delete process.env.DISABLE_SQLITE_AUTO_BACKUP; + else process.env.DISABLE_SQLITE_AUTO_BACKUP = previousDisableBackup; + } + }); +} diff --git a/tests/unit/db-migrationrunner-constants-split.test.ts b/tests/unit/db-migrationrunner-constants-split.test.ts index d8cebe6f1f..eb932cbb13 100644 --- a/tests/unit/db-migrationrunner-constants-split.test.ts +++ b/tests/unit/db-migrationrunner-constants-split.test.ts @@ -70,8 +70,8 @@ describe("migrationRunner/constants — exact small-table snapshots", () => { // ── large tables — count + shape + spot-checks (corruption guard) ───────────── describe("migrationRunner/constants — large-table integrity", () => { - it("RENAMED_MIGRATION_COMPATIBILITY has 31 well-formed entries", () => { - assert.equal(RENAMED_MIGRATION_COMPATIBILITY.length, 31); + it("RENAMED_MIGRATION_COMPATIBILITY has 32 well-formed entries", () => { + assert.equal(RENAMED_MIGRATION_COMPATIBILITY.length, 32); for (const e of RENAMED_MIGRATION_COMPATIBILITY) { assert.equal(typeof e.fromVersion, "string"); assert.equal(typeof e.fromName, "string"); @@ -113,6 +113,17 @@ describe("migrationRunner/constants — large-table integrity", () => { "144", ] ); + assert.deepEqual( + RENAMED_MIGRATION_COMPATIBILITY.find( + (e) => e.fromVersion === "074" && e.fromName === "inspector_custom_hosts" + ), + { + fromVersion: "074", + fromName: "inspector_custom_hosts", + toVersion: "081", + toName: "inspector_custom_hosts", + } + ); assert.deepEqual(RENAMED_MIGRATION_COMPATIBILITY.at(-7), { fromVersion: "134", fromName: "ccr_blocks", diff --git a/tests/unit/db-pre-migration-backup-retention-10421.test.ts b/tests/unit/db-pre-migration-backup-retention-10421.test.ts index 798bb3cb66..836043d036 100644 --- a/tests/unit/db-pre-migration-backup-retention-10421.test.ts +++ b/tests/unit/db-pre-migration-backup-retention-10421.test.ts @@ -1,27 +1,24 @@ // ENVIRONMENT NOTE (sandbox better-sqlite3 / glibc limitation, not a code defect): -// This test constructs or exercises a real better-sqlite3-backed SQLite database. -// better-sqlite3 is a native addon; production and CI load it normally, but some -// sandboxes/dev boxes ship a system glibc older than the prebuilt binary requires -// ("GLIBC_2.29 not found"), so the native module fails to dlopen and any test that -// reaches better-sqlite3 directly (or asserts stdout that the load-failure warning -// would pollute) fails HERE while passing in CI. This is a known environment -// limitation, not a defect in the code under test: the OmniRoute runtime itself -// cascades to node:sqlite/sql.js when better-sqlite3 is unavailable. See -// tests/unit/_helpers/betterSqlite3Availability.ts for a guard helper. -// #10421 — pre-migration backups were created on every migration run and never pruned, -// so `db_backups/` grew without bound (observed: 48.999 files / 204 GB against a 5,3 MB -// live database). The pruning logic already existed in `cleanupDbBackups()` but nothing -// on the migration path ever reached it. These tests pin the retention step to the -// backup call site so the operator's maxFiles/retentionDays budget is honored there too. +// This suite uses a real on-disk better-sqlite3 database because migration snapshots +// must exercise SQLite's native read-only VACUUM path. Production and CI load the native +// addon normally; see tests/unit/_helpers/betterSqlite3Availability.ts for older sandboxes. +// +// #10421 — repeated failed startups once created a fresh timestamped snapshot every time +// and pruned unrelated restore points. Migration safety now publishes a content-addressed +// snapshot once per database state, never deletes a published snapshot, and leaves retention +// to the manual/scheduled backup paths outside the migration window. -import test from "node:test"; import assert from "node:assert/strict"; import fs from "node:fs"; import os from "node:os"; import path from "node:path"; +import test from "node:test"; import { pathToFileURL } from "node:url"; + import Database from "better-sqlite3"; +import { createBetterSqliteAdapter } from "../../src/lib/db/adapters/betterSqliteAdapter.ts"; + const serial = { concurrency: false }; async function importFresh(modulePath: string) { @@ -29,27 +26,23 @@ async function importFresh(modulePath: string) { return import(`${url}?test=${Date.now()}-${Math.random().toString(16).slice(2)}`); } -function withMockedMigrationFs(files: Record, fn: () => void) { +function withMockedMigrationFs(files: Record, fn: () => T): T { const originalExistsSync = fs.existsSync; const originalReaddirSync = fs.readdirSync; const originalReadFileSync = fs.readFileSync; - const isMigrationDir = (target: unknown) => String(target).replaceAll("\\", "/").endsWith("/src/lib/db/migrations") || String(target).replaceAll("\\", "/").endsWith("/migrations"); fs.existsSync = ((target: unknown) => { if (isMigrationDir(target)) return true; - const fileName = path.basename(String(target)); - if (Object.hasOwn(files, fileName)) return true; + if (Object.hasOwn(files, path.basename(String(target)))) return true; return originalExistsSync(target as string); }) as typeof fs.existsSync; - fs.readdirSync = ((target: string, options?: unknown) => { if (isMigrationDir(target)) return Object.keys(files); return originalReaddirSync(target, options as never); }) as typeof fs.readdirSync; - fs.readFileSync = ((target: unknown, options?: unknown) => { const fileName = path.basename(String(target)); if (Object.hasOwn(files, fileName)) return files[fileName]; @@ -65,149 +58,264 @@ function withMockedMigrationFs(files: Record, fn: () => void) { } } -/** Minimal SqliteAdapter over a real on-disk file (VACUUM INTO needs a file, not :memory:). */ function createFileDb(sqlitePath: string) { - const db = new Database(sqlitePath); - - return { - driver: "better-sqlite3", - get open() { - return db.open; - }, - get name() { - return db.name; - }, - prepare: (sql: string) => db.prepare(sql), - exec: (sql: string) => db.exec(sql), - pragma: (str: string, options?: unknown) => db.pragma(str, options as never), - transaction: (fn: (...args: unknown[]) => unknown) => { - const tx = db.transaction((...args: unknown[]) => fn(...args)); - return (...args: unknown[]) => tx(...args); - }, - immediate: (fn: () => void) => fn(), - async backup() {}, - checkpoint() {}, - close: () => db.close(), - get raw() { - return db; - }, - }; + return createBetterSqliteAdapter(new Database(sqlitePath)); } -/** - * Build a DB that already has migrations applied (so the pre-migration backup path is - * reached: it requires `applied.size > 0`) plus one pending migration to trigger a run. - */ -function seedAppliedDb(db: ReturnType) { +function seedExistingDb(db: ReturnType): void { db.exec(` CREATE TABLE provider_connections (id TEXT PRIMARY KEY); CREATE TABLE combos (id TEXT PRIMARY KEY); CREATE TABLE call_logs (id TEXT PRIMARY KEY); - `); -} - -/** - * Record 001 as applied in the runner's own ledger table. `runMigrations` only takes a - * pre-migration backup when `applied.size > 0`, so this is what puts the test on the - * code path under exercise. - */ -function seedAppliedMigration(db: ReturnType) { - db.exec(` - CREATE TABLE IF NOT EXISTS _omniroute_migrations ( + CREATE TABLE _omniroute_migrations ( version TEXT PRIMARY KEY, name TEXT NOT NULL, applied_at TEXT NOT NULL DEFAULT (datetime('now')) ); + INSERT INTO provider_connections (id) VALUES ('existing-data'); + INSERT INTO _omniroute_migrations (version, name) VALUES ('001', 'initial_schema'); `); - db.prepare( - "INSERT OR REPLACE INTO _omniroute_migrations (version, name, applied_at) VALUES (?, ?, ?)" - ).run("001", "initial_schema", new Date().toISOString()); } -function makeTempDataDir() { - const dir = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-backup-retention-")); +function seedSetupSkeleton(db: ReturnType): void { + db.exec(` + CREATE TABLE provider_connections (id TEXT PRIMARY KEY); + CREATE TABLE _omniroute_migrations ( + version TEXT PRIMARY KEY, + name TEXT NOT NULL, + applied_at TEXT NOT NULL DEFAULT (datetime('now')) + ); + INSERT INTO provider_connections (id) VALUES ('setup-preserved-data'); + INSERT INTO _omniroute_migrations (version, name) VALUES ('001', 'initial_schema'); + `); +} + +function makeTempDataDir(): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-migration-snapshot-")); fs.mkdirSync(path.join(dir, "db_backups"), { recursive: true }); return dir; } -/** Pre-existing backups, oldest first, with distinct mtimes so retention ordering is stable. */ -function seedBackups(backupDir: string, count: number) { +function seedTraditionalBackups(backupDir: string, count: number): string[] { const names: string[] = []; - for (let i = 0; i < count; i++) { - const name = `db_2026-08-${String(i + 1).padStart(2, "0")}T00-00-00-000Z_pre-migration.sqlite`; - const filePath = path.join(backupDir, name); - fs.writeFileSync(filePath, "x"); - const t = new Date(2026, 7, i + 1).getTime() / 1000; - fs.utimesSync(filePath, t, t); + for (let index = 0; index < count; index += 1) { + const name = + `db_2026-08-${String(index + 1).padStart(2, "0")}` + "T00-00-00-000Z_pre-migration.sqlite"; + fs.writeFileSync(path.join(backupDir, name), `seed-${index}`); names.push(name); } return names; } -function countBackups(backupDir: string) { - return fs.readdirSync(backupDir).filter((n) => n.startsWith("db_")).length; +function listCanonicalBackups(backupDir: string): string[] { + if (!fs.existsSync(backupDir)) return []; + return fs + .readdirSync(backupDir) + .filter((name) => name.startsWith("db_") && name.endsWith(".sqlite")) + .sort(); } -function withEnv(vars: Record, fn: () => void) { - const saved: Record = {}; - for (const [k, v] of Object.entries(vars)) { - saved[k] = process.env[k]; - if (v === undefined) delete process.env[k]; - else process.env[k] = v; +function listOwnedTempDirs(backupDir: string): string[] { + if (!fs.existsSync(backupDir)) return []; + return fs.readdirSync(backupDir).filter((name) => name.startsWith(".migration-snapshot-")); +} + +function withEnv(vars: Record, fn: () => T): T { + const saved = new Map(); + for (const [key, value] of Object.entries(vars)) { + saved.set(key, process.env[key]); + if (value === undefined) delete process.env[key]; + else process.env[key] = value; } try { return fn(); } finally { - for (const [k, v] of Object.entries(saved)) { - if (v === undefined) delete process.env[k]; - else process.env[k] = v; + for (const [key, value] of saved) { + if (value === undefined) delete process.env[key]; + else process.env[key] = value; } } } test( - "#10421 runMigrations prunes pre-migration backups to the configured maxFiles", + "repeated zero-progress failures reuse one content-addressed snapshot without pruning", serial, async () => { const dataDir = makeTempDataDir(); const backupDir = path.join(dataDir, "db_backups"); - const sqlitePath = path.join(dataDir, "storage.sqlite"); - const db = createFileDb(sqlitePath); + const db = createFileDb(path.join(dataDir, "storage.sqlite")); try { - seedAppliedDb(db); - seedBackups(backupDir, 30); - assert.equal(countBackups(backupDir), 30, "precondition: 30 stale backups on disk"); - + seedExistingDb(db); + const seeded = seedTraditionalBackups(backupDir, 6); const { runMigrations } = await importFresh("src/lib/db/migrationRunner.ts"); + const files = { + "001_initial_schema.sql": "SELECT 1;", + "002_broken_probe.sql": "INSERT INTO table_that_does_not_exist VALUES (1);", + }; + const fail = () => withMockedMigrationFs(files, () => runMigrations(db)); - withEnv( - { - DB_BACKUP_MAX_FILES: "5", - DB_BACKUP_RETENTION_DAYS: "0", - DISABLE_SQLITE_AUTO_BACKUP: undefined, - }, - () => { + assert.throws(fail, /table_that_does_not_exist/); + const afterFirst = listCanonicalBackups(backupDir); + const contentAddressed = afterFirst.filter((name) => name.startsWith("db_state-")); + assert.equal(contentAddressed.length, 1); + assert.match(contentAddressed[0]!, /^db_state-[a-f0-9]{64}_pre-migration\.sqlite$/); + assert.equal( + seeded.every((name) => afterFirst.includes(name)), + true, + "migration failure must not prune pre-existing restore points" + ); + + assert.throws(fail, /table_that_does_not_exist/); + assert.deepEqual( + listCanonicalBackups(backupDir), + afterFirst, + "an unchanged failed startup must reuse the exact content-addressed snapshot" + ); + assert.deepEqual(listOwnedTempDirs(backupDir), []); + } finally { + db.close(); + fs.rmSync(dataDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + } + } +); + +test( + "an existing DB fails closed when hard-link publication is unavailable even with auto backup disabled", + serial, + async () => { + const dataDir = makeTempDataDir(); + const backupDir = path.join(dataDir, "db_backups"); + const db = createFileDb(path.join(dataDir, "storage.sqlite")); + const originalLinkSync = fs.linkSync; + + try { + seedExistingDb(db); + const { runMigrations } = await importFresh("src/lib/db/migrationRunner.ts"); + fs.linkSync = (() => { + throw Object.assign(new Error("hard links unsupported by this filesystem"), { + code: "ENOTSUP", + }); + }) as typeof fs.linkSync; + + assert.throws( + () => + withEnv({ DISABLE_SQLITE_AUTO_BACKUP: "true" }, () => + withMockedMigrationFs( + { + "001_initial_schema.sql": "SELECT 1;", + "002_ordinary_pending.sql": "CREATE TABLE must_not_apply (id INTEGER);", + }, + () => runMigrations(db) + ) + ), + /durable snapshot.*hard links unsupported.*hard links.*synchronization/is + ); + assert.equal( + db.prepare("SELECT name FROM sqlite_master WHERE name = 'must_not_apply'").get(), + undefined, + "an ordinary pending migration must not run without its mandatory snapshot" + ); + assert.deepEqual( + db.prepare("SELECT version, name FROM _omniroute_migrations ORDER BY version").all(), + [{ version: "001", name: "initial_schema" }] + ); + assert.deepEqual(listCanonicalBackups(backupDir), []); + assert.deepEqual(listOwnedTempDirs(backupDir), []); + } finally { + fs.linkSync = originalLinkSync; + db.close(); + fs.rmSync(dataDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + } + } +); + +test( + "a pre-existing setup skeleton requires a snapshot even when mass-migration safety treats it as fresh", + serial, + async () => { + const dataDir = makeTempDataDir(); + const backupDir = path.join(dataDir, "db_backups"); + const db = createFileDb(path.join(dataDir, "storage.sqlite")); + const originalLinkSync = fs.linkSync; + + try { + seedSetupSkeleton(db); + const { runMigrations } = await importFresh("src/lib/db/migrationRunner.ts"); + fs.linkSync = (() => { + throw Object.assign(new Error("hard links unsupported by this filesystem"), { + code: "ENOTSUP", + }); + }) as typeof fs.linkSync; + + assert.throws( + () => withMockedMigrationFs( { "001_initial_schema.sql": "SELECT 1;", - "002_retention_probe.sql": "CREATE TABLE retention_probe_10421 (id INTEGER);", + "002_ordinary_pending.sql": "CREATE TABLE must_not_apply (id INTEGER);", }, - () => { - // Mark 001 as applied so `applied.size > 0` and the backup path is reached. - seedAppliedMigration(db); + () => + runMigrations(db, { + isNewDb: true, + databaseExistedBeforeInitialization: true, + }) + ), + /durable snapshot.*hard links unsupported.*hard links.*synchronization/is + ); + assert.equal( + db.prepare("SELECT name FROM sqlite_master WHERE name = 'must_not_apply'").get(), + undefined, + "a setup-created persistent DB must not change when its safety snapshot cannot publish" + ); + assert.deepEqual( + db.prepare("SELECT id FROM provider_connections").all(), + [{ id: "setup-preserved-data" }], + "the setup-created provider state must remain untouched" + ); + assert.deepEqual(listCanonicalBackups(backupDir), []); + assert.deepEqual(listOwnedTempDirs(backupDir), []); + } finally { + fs.linkSync = originalLinkSync; + db.close(); + fs.rmSync(dataDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + } + } +); - runMigrations(db); - } - ); - } +test( + "successful migrations retain existing backups and do not prune inside the migration window", + serial, + async () => { + const dataDir = makeTempDataDir(); + const backupDir = path.join(dataDir, "db_backups"); + const db = createFileDb(path.join(dataDir, "storage.sqlite")); + + try { + seedExistingDb(db); + const seeded = seedTraditionalBackups(backupDir, 6); + const { runMigrations } = await importFresh("src/lib/db/migrationRunner.ts"); + + const count = withEnv({ DB_BACKUP_MAX_FILES: "1", DB_BACKUP_RETENTION_DAYS: "0" }, () => + withMockedMigrationFs( + { + "001_initial_schema.sql": "SELECT 1;", + "002_success.sql": "CREATE TABLE migration_success (id INTEGER);", + }, + () => runMigrations(db) + ) ); - const remaining = countBackups(backupDir); + assert.equal(count, 1); assert.ok( - remaining <= 5, - `expected retention to cap db_backups at 5 files, found ${remaining} — ` + - `pre-migration backups are accumulating unbounded (#10421)` + db.prepare("SELECT name FROM sqlite_master WHERE name = 'migration_success'").get() + ); + const after = listCanonicalBackups(backupDir); + assert.equal(after.filter((name) => name.startsWith("db_state-")).length, 1); + assert.equal( + seeded.every((name) => after.includes(name)), + true, + "retention must remain outside the concurrent migration window" ); } finally { db.close(); @@ -216,51 +324,26 @@ test( } ); -test("#10421 the newest pre-migration backup survives pruning", serial, async () => { +test("an already-current DB does not acquire an IMMEDIATE writer lock", serial, async () => { const dataDir = makeTempDataDir(); - const backupDir = path.join(dataDir, "db_backups"); - const sqlitePath = path.join(dataDir, "storage.sqlite"); - const db = createFileDb(sqlitePath); + const db = createFileDb(path.join(dataDir, "storage.sqlite")); try { - seedAppliedDb(db); - seedBackups(backupDir, 10); - + seedExistingDb(db); const { runMigrations } = await importFresh("src/lib/db/migrationRunner.ts"); - - withEnv( - { - DB_BACKUP_MAX_FILES: "3", - DB_BACKUP_RETENTION_DAYS: "0", - DISABLE_SQLITE_AUTO_BACKUP: undefined, + const noWriterAdapter = { + ...db, + immediate: () => { + throw new Error("unexpected IMMEDIATE writer lock"); }, - () => { - withMockedMigrationFs( - { - "001_initial_schema.sql": "SELECT 1;", - "002_retention_probe.sql": "CREATE TABLE retention_probe_10421b (id INTEGER);", - }, - () => { - seedAppliedMigration(db); + }; - runMigrations(db); - } - ); - } + assert.equal( + withMockedMigrationFs({ "001_initial_schema.sql": "SELECT 1;" }, () => + runMigrations(noWriterAdapter) + ), + 0 ); - - const remaining = fs.readdirSync(backupDir).filter((n) => n.startsWith("db_")); - assert.ok(remaining.length <= 3, `expected <=3 backups, found ${remaining.length}`); - - // The backup written by THIS run must be among the survivors — pruning must never - // discard the snapshot that protects the migration it was taken for. - const seededNames = new Set( - Array.from({ length: 10 }, (_, i) => { - return `db_2026-08-${String(i + 1).padStart(2, "0")}T00-00-00-000Z_pre-migration.sqlite`; - }) - ); - const fresh = remaining.filter((n) => !seededNames.has(n)); - assert.equal(fresh.length, 1, `expected the run's own backup to survive, got ${fresh.length}`); } finally { db.close(); fs.rmSync(dataDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); From 350ac8c12ddccbd7bd2c28df234f98d4de43e694 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 21:06:28 -0300 Subject: [PATCH 070/143] fix(sse): preserve 1min.ai partial output before stream errors (#12466) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado após reconciliar com o #12465, que entrou primeiro e criou o mesmo arquivo novo `open-sse/utils/streamReadiness.ts` com desenho divergente de cancelamento. Mantive a versão desta branch, que defere o release do lock para quando a leitura em voo termina e faz `reader.cancel()` fire-and-forget — assim uma promise de provider que nunca resolve não torna o cancelamento ilimitado. A escolha não foi por preferência: rodei as suítes dos **dois** PRs contra ela, 21/21 no readiness compartilhado e **22/22** incluindo o boundary do Perplexity do próprio #12465. typecheck:core limpo. --- .../pending-onemin-stream-error-boundary.md | 1 + open-sse/executors/oneminai.ts | 341 +++++++--- open-sse/utils/streamReadiness.ts | 50 +- .../oneminai-stream-error-boundary.fixture.ts | 620 ++++++++++++++++++ .../oneminai-stream-error-boundary.test.ts | 91 +++ tests/unit/stream-readiness.test.ts | 120 +++- 6 files changed, 1083 insertions(+), 140 deletions(-) create mode 100644 changelog.d/fixes/pending-onemin-stream-error-boundary.md create mode 100644 tests/fixtures/oneminai-stream-error-boundary.fixture.ts create mode 100644 tests/unit/oneminai-stream-error-boundary.test.ts diff --git a/changelog.d/fixes/pending-onemin-stream-error-boundary.md b/changelog.d/fixes/pending-onemin-stream-error-boundary.md new file mode 100644 index 0000000000..818b124a2f --- /dev/null +++ b/changelog.d/fixes/pending-onemin-stream-error-boundary.md @@ -0,0 +1 @@ +- **fix(providers):** keep 1min.ai HTTP 200 stream errors out of assistant content, preserve partial output, and expose sanitized terminal errors so pre-content failures can fall back. diff --git a/open-sse/executors/oneminai.ts b/open-sse/executors/oneminai.ts index 023f6ce425..5b3e2f4dda 100644 --- a/open-sse/executors/oneminai.ts +++ b/open-sse/executors/oneminai.ts @@ -16,6 +16,8 @@ type OpenAIMessage = { }; const CHAT_URL = "https://api.1min.ai/api/chat-with-ai"; +const MAX_STREAM_ERROR_DATA_CHARS = 64 * 1024; +const STREAM_ERROR_FALLBACK = "1min.ai upstream stream failed"; const ROLE_LABELS: Record = { system: "System", developer: "System", @@ -69,7 +71,35 @@ function buildSseChunk(data: unknown): string { return `data: ${JSON.stringify(data)}\n\n`; } -function buildOpenAiJsonCompletion(content: string, model: string, id: string, created: number): Response { +function parseStreamErrorMessage(data: string): string { + if (!data || data.length > MAX_STREAM_ERROR_DATA_CHARS) return STREAM_ERROR_FALLBACK; + + try { + const parsed = asRecord(JSON.parse(data)); + const directMessage = typeof parsed.message === "string" ? parsed.message.trim() : ""; + if (directMessage) return directMessage; + + if (typeof parsed.error === "string") { + const errorMessage = parsed.error.trim(); + if (errorMessage) return errorMessage; + } + + const nestedError = asRecord(parsed.error); + const nestedMessage = typeof nestedError.message === "string" ? nestedError.message.trim() : ""; + if (nestedMessage) return nestedMessage; + } catch { + // Malformed and over-complex payloads use the fixed public fallback below. + } + + return STREAM_ERROR_FALLBACK; +} + +function buildOpenAiJsonCompletion( + content: string, + model: string, + id: string, + created: number +): Response { return new Response( JSON.stringify({ id, @@ -84,7 +114,11 @@ function buildOpenAiJsonCompletion(content: string, model: string, id: string, c ); } -function toOpenAiErrorResponse(status: number, message: string, upstreamDetails?: unknown): Response { +function toOpenAiErrorResponse( + status: number, + message: string, + upstreamDetails?: unknown +): Response { return new Response(JSON.stringify(buildErrorBody(status, message, upstreamDetails)), { status, headers: { "Content-Type": "application/json" }, @@ -96,109 +130,214 @@ function toOpenAiErrorResponse(status: number, message: string, upstreamDetails? * data: {...}) from the upstream Response body and re-emit them as standard * OpenAI chat.completion.chunk SSE. */ -function translateSseStream(upstreamBody: ReadableStream, model: string, id: string, created: number): ReadableStream { +function translateSseStream( + upstreamBody: ReadableStream, + model: string, + id: string, + created: number +): ReadableStream { const decoder = new TextDecoder(); const encoder = new TextEncoder(); + const reader = upstreamBody.getReader(); + const pendingChunks: Uint8Array[] = []; + let buffer = ""; + let finished = false; + let roleEmitted = false; + let terminalError: Error | null = null; + let upstreamCancelRequested = false; + let downstreamCancelled = false; + let readInFlight = false; + let readerReleased = false; + + const releaseReader = () => { + if (readerReleased) return; + readerReleased = true; + reader.releaseLock(); + }; + + const cancelUpstream = (reason: unknown) => { + if (upstreamCancelRequested) return; + upstreamCancelRequested = true; + try { + // Upstream cleanup is provider-controlled and may never settle. The + // translated stream owns the reader lock and releases it independently. + void reader.cancel(reason).catch(() => {}); + } catch { + // Cancellation is cleanup-only; the terminal state is already fixed. + } + }; + + const queueChunk = (text: string) => { + pendingChunks.push(encoder.encode(text)); + }; + + const emitRole = () => { + if (roleEmitted) return; + roleEmitted = true; + queueChunk( + buildSseChunk({ + id, + object: "chat.completion.chunk", + created, + model, + choices: [{ index: 0, delta: { role: "assistant" }, finish_reason: null }], + }) + ); + }; + + const finish = () => { + if (finished) return; + finished = true; + queueChunk( + buildSseChunk({ + id, + object: "chat.completion.chunk", + created, + model, + choices: [{ index: 0, delta: {}, finish_reason: "stop" }], + }) + ); + queueChunk("data: [DONE]\n\n"); + }; + + const emitContent = (text: string) => { + if (!text) return; + emitRole(); + queueChunk( + buildSseChunk({ + id, + object: "chat.completion.chunk", + created, + model, + choices: [{ index: 0, delta: { content: text }, finish_reason: null }], + }) + ); + }; + + const emitError = (data: string) => { + if (finished) return; + finished = true; + cancelUpstream("1min.ai upstream stream error"); + + if (!roleEmitted) { + const message = parseStreamErrorMessage(data); + queueChunk(buildSseChunk(buildErrorBody(502, message))); + queueChunk("data: [DONE]\n\n"); + return; + } + + // A bare `{ error }` frame is dropped by the OpenAI passthrough sanitizer. + // Preserve every content delta already queued, then error the source with + // a fixed public message. pipeWithDisconnect() converts it into a native + // terminal error frame and drives usage, call-log, and fallback finalizers. + terminalError = Object.assign(new Error(STREAM_ERROR_FALLBACK), { + statusCode: 502, + }); + }; + + // SSE event framing: "event:"/"data:" lines, blank-line separated records. + const processEvent = (eventText: string) => { + let eventType = "message"; + const dataLines: string[] = []; + for (const rawLine of eventText.split("\n")) { + if (rawLine.startsWith("event:")) { + eventType = rawLine.slice(6).trim(); + } else if (rawLine.startsWith("data:")) { + dataLines.push(rawLine.slice(5).trim()); + } + } + const data = dataLines.join("\n"); + if (eventType === "content") { + try { + const parsed = asRecord(JSON.parse(data)); + if (typeof parsed.content === "string") emitContent(parsed.content); + } catch { + // Ignore malformed content events rather than surfacing partial JSON. + } + } else if (eventType === "error") { + emitError(data); + } else if (eventType === "done") { + finish(); + } + // "result" carries the final full aiRecord, redundant with the content + // events already streamed — intentionally ignored. + }; + + const processBufferedEvents = () => { + let separatorIndex = buffer.indexOf("\n\n"); + while (separatorIndex !== -1 && !finished) { + processEvent(buffer.slice(0, separatorIndex)); + buffer = buffer.slice(separatorIndex + 2); + separatorIndex = buffer.indexOf("\n\n"); + } + }; return new ReadableStream({ - async start(controller) { - controller.enqueue( - encoder.encode( - buildSseChunk({ - id, - object: "chat.completion.chunk", - created, - model, - choices: [{ index: 0, delta: { role: "assistant" }, finish_reason: null }], - }) - ) - ); + async pull(controller) { + if (downstreamCancelled) return; - const reader = upstreamBody.getReader(); - let buffer = ""; - let finished = false; - - const finish = () => { - if (finished) return; - finished = true; - controller.enqueue( - encoder.encode( - buildSseChunk({ - id, - object: "chat.completion.chunk", - created, - model, - choices: [{ index: 0, delta: {}, finish_reason: "stop" }], - }) - ) - ); - controller.enqueue(encoder.encode("data: [DONE]\n\n")); - controller.close(); - }; - - const emitContent = (text: string) => { - if (!text) return; - controller.enqueue( - encoder.encode( - buildSseChunk({ - id, - object: "chat.completion.chunk", - created, - model, - choices: [{ index: 0, delta: { content: text }, finish_reason: null }], - }) - ) - ); - }; - - // SSE event framing: "event:"/"data:" lines, blank-line separated records. - const processEvent = (eventText: string) => { - let eventType = "message"; - const dataLines: string[] = []; - for (const rawLine of eventText.split("\n")) { - if (rawLine.startsWith("event:")) { - eventType = rawLine.slice(6).trim(); - } else if (rawLine.startsWith("data:")) { - dataLines.push(rawLine.slice(5).trim()); - } - } - const data = dataLines.join("\n"); - if (eventType === "content") { - try { - const parsed = asRecord(JSON.parse(data)); - if (typeof parsed.content === "string") emitContent(parsed.content); - } catch { - // Ignore malformed content events rather than surfacing partial JSON. - } - } else if (eventType === "error") { - emitContent(`\n[1min.ai error: ${data}]`); - finish(); - } else if (eventType === "done") { - finish(); - } - // "result" carries the final full aiRecord, redundant with the content - // events already streamed — intentionally ignored. - }; - - try { - while (!finished) { - const { done, value } = await reader.read(); - if (done) break; - buffer += decoder.decode(value, { stream: true }); - let separatorIndex = buffer.indexOf("\n\n"); - while (separatorIndex !== -1) { - processEvent(buffer.slice(0, separatorIndex)); - buffer = buffer.slice(separatorIndex + 2); - separatorIndex = buffer.indexOf("\n\n"); - } - } - if (!finished && buffer.trim()) processEvent(buffer); - finish(); - } catch (error) { - controller.error(error); - } finally { - reader.releaseLock(); + if (pendingChunks.length > 0) { + controller.enqueue(pendingChunks.shift()!); + return; } + + if (terminalError) { + releaseReader(); + controller.error(terminalError); + return; + } + + if (finished) { + releaseReader(); + controller.close(); + return; + } + + readInFlight = true; + try { + while (pendingChunks.length === 0 && !finished && !downstreamCancelled) { + const { done, value } = await reader.read(); + if (downstreamCancelled) return; + if (done) { + buffer += decoder.decode(); + if (buffer.trim()) processEvent(buffer); + finish(); + break; + } + + buffer += decoder.decode(value, { stream: true }); + // Process the complete upstream chunk, even after it queues output. + // One network read may contain multiple content events followed by + // an error; the internal queue preserves all of them in order. + processBufferedEvents(); + } + + if (downstreamCancelled) return; + if (pendingChunks.length > 0) { + controller.enqueue(pendingChunks.shift()!); + } else if (terminalError) { + releaseReader(); + controller.error(terminalError); + } else if (finished) { + releaseReader(); + controller.close(); + } + } catch (error) { + releaseReader(); + if (!downstreamCancelled) controller.error(error); + } finally { + readInFlight = false; + if (downstreamCancelled) releaseReader(); + } + }, + cancel(reason) { + downstreamCancelled = true; + pendingChunks.length = 0; + // A client disconnect must release the upstream reader even when its + // next pull never settles. Do not await provider cleanup here: the + // downstream cancellation contract must remain bounded. + cancelUpstream(reason ?? "1min.ai downstream cancelled"); + if (!readInFlight) releaseReader(); }, }); } @@ -290,7 +429,9 @@ export class OneMinAiExecutor extends BaseExecutor { const aiRecord = asRecord(json.aiRecord); const detail = asRecord(aiRecord.aiRecordDetail); const resultObject = Array.isArray(detail.resultObject) ? detail.resultObject : []; - const content = resultObject.filter((part): part is string => typeof part === "string").join(""); + const content = resultObject + .filter((part): part is string => typeof part === "string") + .join(""); return { response: buildOpenAiJsonCompletion(content, model, id, created), diff --git a/open-sse/utils/streamReadiness.ts b/open-sse/utils/streamReadiness.ts index 3774cc0260..908c18725c 100644 --- a/open-sse/utils/streamReadiness.ts +++ b/open-sse/utils/streamReadiness.ts @@ -422,56 +422,66 @@ function prependBufferedChunks( reader: ReadableStreamDefaultReader ): ReadableStream { let bufferedIndex = 0; - let cancelled = false; + let readInFlight = false; + let cancelRequested = false; let readerReleased = false; const releaseReader = () => { if (readerReleased) return; readerReleased = true; + reader.releaseLock(); + }; + + const cancelReader = (reason: unknown) => { + if (cancelRequested) return; + cancelRequested = true; + try { - reader.releaseLock(); + // The provider controls this promise and may never settle. Cancellation + // of the replay stream must remain bounded, so cleanup is deliberately + // fire-and-forget while the in-flight read releases the lock in `pull`. + void reader.cancel(reason).catch(() => {}); } catch { - // A hostile source can keep a read/cancel pending forever. The public stream must - // remain cancellable even when its abandoned source cannot release immediately. + // A synchronous cancellation failure is cleanup-only; the downstream + // stream has already been cancelled by its consumer. } + + if (!readInFlight) releaseReader(); }; return new ReadableStream({ async pull(controller) { - if (cancelled) return; + if (cancelRequested) return; - // Replay exactly one readiness chunk per pull. Keeping the first buffered chunk at - // the stream's default high-water mark prevents an eager read of a later upstream - // failure from discarding that legitimate prefix before the caller attaches. + // Replay exactly one readiness chunk per demand. Reading the source + // eagerly here would let a subsequent source error clear this queue + // before the consumer has observed the buffered prefix. if (bufferedIndex < chunks.length) { controller.enqueue(chunks[bufferedIndex]); bufferedIndex += 1; return; } + readInFlight = true; try { const { done, value } = await reader.read(); - if (cancelled) return; + if (cancelRequested) return; if (done) { releaseReader(); controller.close(); - return; + } else if (value) { + controller.enqueue(value); } - if (value) controller.enqueue(value); } catch (error) { releaseReader(); - if (!cancelled) controller.error(error); + if (!cancelRequested) controller.error(error); + } finally { + readInFlight = false; + if (cancelRequested) releaseReader(); } }, cancel(reason) { - if (cancelled) return; - cancelled = true; - // Do not await a provider's cancel hook: a hostile or stalled source must not make - // downstream cancellation hang. Release the lock once cancellation actually settles. - void reader - .cancel(reason) - .catch(() => {}) - .finally(releaseReader); + cancelReader(reason); }, }); } diff --git a/tests/fixtures/oneminai-stream-error-boundary.fixture.ts b/tests/fixtures/oneminai-stream-error-boundary.fixture.ts new file mode 100644 index 0000000000..5af1b6be9c --- /dev/null +++ b/tests/fixtures/oneminai-stream-error-boundary.fixture.ts @@ -0,0 +1,620 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +assert.ok(process.env.DATA_DIR, "the parent harness must provide an isolated DATA_DIR"); +assert.ok( + process.env.OMNIROUTE_PLUGINS_DIR, + "the parent harness must provide an isolated OMNIROUTE_PLUGINS_DIR" +); + +const [ + { OneMinAiExecutor }, + { ensureStreamReadiness }, + dbCore, + settingsDb, + callLogs, + usageHistory, + accountSemaphore, + readCache, + { handleChatCore }, +] = await Promise.all([ + import("../../open-sse/executors/oneminai.ts"), + import("../../open-sse/utils/streamReadiness.ts"), + import("../../src/lib/db/core.ts"), + import("../../src/lib/db/settings.ts"), + import("../../src/lib/usage/callLogs.ts"), + import("../../src/lib/usage/usageHistory.ts"), + import("../../open-sse/services/accountSemaphore.ts"), + import("../../src/lib/db/readCache.ts"), + import("../../open-sse/handlers/chatCore.ts"), +]); + +const originalFetch = globalThis.fetch; +const encoder = new TextEncoder(); +const STREAM_URL = "https://api.1min.ai/api/chat-with-ai?isStreaming=true"; + +type PersistenceIdentity = { + model: string; + connectionId: string; +}; + +const PRE_CONTENT_IDENTITY: PersistenceIdentity = { + model: "gpt-4o-mini-onemin-pre-content-boundary", + connectionId: "onemin-stream-pre-content-boundary", +}; +const BATCHED_IDENTITY: PersistenceIdentity = { + model: "gpt-4o-mini-onemin-batched-boundary", + connectionId: "onemin-stream-batched-boundary", +}; +const PARTIAL_IDENTITY: PersistenceIdentity = { + model: "gpt-4o-mini-onemin-partial-boundary", + connectionId: "onemin-stream-partial-boundary", +}; + +function installFetchFactory(responseFactory: () => Response): () => number { + let calls = 0; + globalThis.fetch = async (input, init = {}) => { + calls += 1; + assert.equal(String(input), STREAM_URL, "the test must never permit another network target"); + assert.equal(init.method, "POST"); + assert.equal((init.headers as Record)["API-KEY"], "unit-test-key"); + + return responseFactory(); + }; + return () => calls; +} + +function createStreamingResponse(events: string[]): Response { + return new Response( + new ReadableStream({ + start(controller) { + for (const event of events) controller.enqueue(encoder.encode(event)); + controller.close(); + }, + }), + { status: 200, headers: { "Content-Type": "text/event-stream" } } + ); +} + +function installStreamingFetch(events: string[]): () => number { + return installFetchFactory(() => createStreamingResponse(events)); +} + +async function executeStreaming(events: string[]): Promise { + const getCalls = installStreamingFetch(events); + const result = await new OneMinAiExecutor().execute({ + model: "gpt-4o-mini", + body: { messages: [{ role: "user", content: "hello" }] }, + stream: true, + credentials: { apiKey: "unit-test-key" }, + signal: AbortSignal.timeout(10_000), + log: null, + }); + assert.equal(getCalls(), 1); + return result.response; +} + +function noopLog() { + return { debug() {}, info() {}, warn() {}, error() {} }; +} + +async function invokeStreamingChatCore( + identity: PersistenceIdentity, + onStreamFailure?: (failure: { + status: number; + message: string; + code?: string; + type?: string; + }) => void, + onRequestSuccess?: () => Promise | void +) { + await settingsDb.updateSettings({ call_log_pipeline_enabled: true }); + readCache.invalidateDbCache("settings"); + const body = { + model: identity.model, + stream: true, + messages: [{ role: "user", content: "hello" }], + }; + + return handleChatCore({ + body: structuredClone(body), + modelInfo: { provider: "oneminai", model: identity.model, extendedContext: false }, + credentials: { + apiKey: "unit-test-key", + connectionId: identity.connectionId, + providerSpecificData: {}, + }, + connectionId: identity.connectionId, + log: noopLog(), + clientRawRequest: { + endpoint: "/v1/chat/completions", + body: structuredClone(body), + headers: new Headers({ + accept: "text/event-stream", + "x-omniroute-session-id": identity.connectionId, + }), + }, + userAgent: identity.connectionId, + onRequestSuccess, + onStreamFailure, + } as never); +} + +async function waitFor(read: () => Promise, timeoutMs = 5_000): Promise { + const deadline = Date.now() + timeoutMs; + while (Date.now() < deadline) { + const value = await read(); + if (value) return value; + await new Promise((resolve) => setTimeout(resolve, 25)); + } + return null; +} + +async function getOneMinCallLog(identity: PersistenceIdentity) { + assert.equal( + await callLogs.waitForCallLogSaves(5_000), + true, + "call-log persistence must drain before inspection" + ); + const rows = await callLogs.getCallLogs({ + provider: "oneminai", + model: identity.model, + limit: 20, + }); + const row = Array.isArray(rows) + ? rows.find( + (candidate) => + candidate.connectionId === identity.connectionId && + (candidate.model === identity.model || candidate.requestedModel === identity.model) + ) + : null; + return row ? callLogs.getCallLogById(row.id) : null; +} + +async function getOneMinUsage(identity: PersistenceIdentity) { + const rows = await usageHistory.getUsageHistory({ + provider: "oneminai", + model: identity.model, + }); + return rows.find((row) => row.connectionId === identity.connectionId) ?? null; +} + +async function assertUnusedPersistenceIdentity(identity: PersistenceIdentity) { + assert.equal( + await getOneMinCallLog(identity), + null, + `call-log identity must be unused before scenario: ${identity.connectionId}` + ); + assert.equal( + await getOneMinUsage(identity), + null, + `usage identity must be unused before scenario: ${identity.connectionId}` + ); +} + +async function readUntil( + reader: ReadableStreamDefaultReader, + marker: string +): Promise { + const decoder = new TextDecoder(); + let text = ""; + while (!text.includes(marker)) { + const { done, value } = await reader.read(); + assert.equal(done, false, `stream ended before ${marker}`); + if (value) text += decoder.decode(value, { stream: true }); + } + return text; +} + +async function readRemaining(reader: ReadableStreamDefaultReader): Promise { + const decoder = new TextDecoder(); + let text = ""; + for (;;) { + const { done, value } = await reader.read(); + if (done) return text + decoder.decode(); + if (value) text += decoder.decode(value, { stream: true }); + } +} + +test.afterEach(async () => { + const drained = await callLogs.waitForCallLogSaves(5_000); + globalThis.fetch = originalFetch; + usageHistory.clearPendingRequests(); + accountSemaphore.resetAll(); + assert.equal(drained, true, "all call-log saves must drain before the next test"); +}); + +test.after(async () => { + const drained = await callLogs.waitForCallLogSaves(5_000); + try { + await callLogs.closeCallLogSaves(5_000); + } finally { + globalThis.fetch = originalFetch; + usageHistory.clearPendingRequests(); + accountSemaphore.resetAll(); + dbCore.resetDbInstance(); + } + assert.equal(drained, true, "all call-log saves must drain before teardown"); +}); + +test("1min.ai pre-content stream errors stay errors and permit readiness fallback", async () => { + const rawMessage = + "quota lookup failed at /srv/omniroute/open-sse/executors/oneminai.ts:170\n" + + " at translateSseStream (/srv/omniroute/open-sse/executors/oneminai.ts:99:5)"; + const response = await executeStreaming([ + `event: error\ndata: ${JSON.stringify({ error: { message: rawMessage } })}\n\n`, + ]); + const clientCopy = response.clone(); + + const readiness = await ensureStreamReadiness(response, { + timeoutMs: 2_000, + provider: "oneminai", + model: "gpt-4o-mini", + }); + assert.equal(readiness.ok, false); + if (readiness.ok) assert.fail("an error-only stream must not become ready"); + assert.equal(readiness.response.status, 502); + const fallbackBody = await readiness.response.text(); + assert.match(fallbackBody, /STREAM_EARLY_EOF/); + assert.doesNotMatch(fallbackBody, /\/srv\/omniroute/); + assert.doesNotMatch(fallbackBody, /translateSseStream/); + + const clientText = await clientCopy.text(); + assert.match(clientText, /^data: \{"error":/); + assert.match(clientText, /quota lookup failed at /); + assert.match(clientText, /data: \[DONE\]/); + assert.doesNotMatch(clientText, /"role":"assistant"/); + assert.doesNotMatch(clientText, /"finish_reason":"stop"/); + assert.doesNotMatch(clientText, /\/srv\/omniroute/); + assert.doesNotMatch(clientText, /translateSseStream/); +}); + +test("chatCore turns a pre-content 1min.ai stream error into persisted HTTP 502", async () => { + await assertUnusedPersistenceIdentity(PRE_CONTENT_IDENTITY); + installStreamingFetch([ + `event: error\ndata: ${JSON.stringify({ + error: { + message: + "quota lookup failed at /srv/omniroute/open-sse/executors/oneminai.ts:230 api_key=pre-content-secret\nstack tail", + }, + })}\n\n`, + ]); + + const result = await invokeStreamingChatCore(PRE_CONTENT_IDENTITY); + assert.equal(result.success, false); + if (result.success) assert.fail("a pre-content error must not commit HTTP 200"); + assert.equal(result.status, 502); + assert.equal(result.response.status, 502); + const clientBody = await result.response.text(); + assert.match(clientBody, /STREAM_EARLY_EOF/); + assert.doesNotMatch(clientBody, /pre-content-secret/); + assert.doesNotMatch(clientBody, /\/srv\/omniroute/); + assert.doesNotMatch(clientBody, /stack tail/); + + const detail = await waitFor(() => getOneMinCallLog(PRE_CONTENT_IDENTITY)); + assert.ok(detail, "the failed pre-content attempt must be persisted"); + assert.equal(detail.status, 502); + const persisted = JSON.stringify(detail); + assert.doesNotMatch(persisted, /pre-content-secret/); + assert.doesNotMatch(persisted, /\/srv\/omniroute/); + assert.doesNotMatch(persisted, /stack tail/); + + const usage = await waitFor(() => getOneMinUsage(PRE_CONTENT_IDENTITY)); + assert.ok(usage, "the failed pre-content usage record must be persisted"); + assert.equal(usage.success, false); + assert.equal(usage.status, "502"); + assert.equal(usage.errorCode, "STREAM_EARLY_EOF"); +}); + +test("chatCore preserves batched 1min.ai content before its terminal stream error", async () => { + await assertUnusedPersistenceIdentity(BATCHED_IDENTITY); + installStreamingFetch([ + 'event: content\ndata: {"content":"batched partial one"}\n\n' + + 'event: content\ndata: {"content":"batched partial two"}\n\n' + + `event: error\ndata: ${JSON.stringify({ + message: + "provider failed at /srv/omniroute/open-sse/executors/oneminai.ts:230 api_key=batched-secret", + })}\n\n`, + ]); + const failures: Array<{ + status: number; + message: string; + code?: string; + type?: string; + }> = []; + const requestSuccessPhases: string[] = []; + + const result = await invokeStreamingChatCore( + BATCHED_IDENTITY, + (failure) => failures.push(failure), + async () => { + requestSuccessPhases.push("started"); + await new Promise((resolve) => setTimeout(resolve, 30)); + requestSuccessPhases.push("finished"); + } + ); + assert.equal(result.success, true, "batched real content must cross the readiness boundary"); + assert.deepEqual(requestSuccessPhases, ["started", "finished"]); + assert.ok(result.response.body); + const clientText = await result.response.text(); + const firstContentIndex = clientText.indexOf("batched partial one"); + const secondContentIndex = clientText.indexOf("batched partial two"); + const errorIndex = clientText.indexOf('"error":'); + const doneIndex = clientText.indexOf("data: [DONE]"); + + assert.ok(firstContentIndex >= 0, "the first queued content delta must not be discarded"); + assert.ok(secondContentIndex >= 0, "the second queued content delta must not be discarded"); + assert.ok(firstContentIndex < secondContentIndex, "batched content must retain upstream order"); + assert.ok(secondContentIndex < errorIndex, "all batched content must precede its terminal error"); + assert.ok( + errorIndex < doneIndex, + `the terminal error must precede [DONE]: ${JSON.stringify(clientText)}` + ); + assert.match(clientText, /"finish_reason":"error"/); + assert.doesNotMatch(clientText, /"finish_reason":"stop"/); + assert.doesNotMatch(clientText, /response\.failed/); + assert.doesNotMatch(clientText, /batched-secret/); + assert.doesNotMatch(clientText, /\/srv\/omniroute/); + assert.deepEqual(failures, [ + { + status: 502, + message: "1min.ai upstream stream failed", + code: "stream_pipeline_error", + type: "stream_error", + }, + ]); + + const pending = usageHistory.getPendingRequests(); + assert.deepEqual(Object.keys(pending.byModel), []); + assert.deepEqual(Object.keys(pending.byAccount), []); + + const completed = [...usageHistory.getCompletedDetails().values()]; + assert.equal(completed.length, 1); + assert.equal(completed[0].status, 502); + assert.equal(completed[0].error, "1min.ai upstream stream failed"); + assert.equal(completed[0].errorCode, "stream_pipeline_error"); + + const detail = await waitFor(() => getOneMinCallLog(BATCHED_IDENTITY)); + assert.ok(detail, "the batched terminal stream failure must be persisted"); + assert.equal(detail.status, 502); + assert.equal(detail.error, "1min.ai upstream stream failed"); + const persisted = JSON.stringify(detail); + assert.doesNotMatch(persisted, /batched-secret/); + assert.doesNotMatch(persisted, /\/srv\/omniroute/); + + const usage = await waitFor(() => getOneMinUsage(BATCHED_IDENTITY)); + assert.ok(usage, "the batched terminal failure usage record must be persisted"); + assert.equal(usage.success, false); + assert.equal(usage.status, "502"); + assert.equal(usage.errorCode, "stream_pipeline_error"); +}); + +test("chatCore preserves partial 1min.ai content then finalizes and persists a stream failure", async () => { + await assertUnusedPersistenceIdentity(PARTIAL_IDENTITY); + let upstreamController: ReadableStreamDefaultController | null = null; + let cancelCalls = 0; + const getCalls = installFetchFactory( + () => + new Response( + new ReadableStream({ + start(controller) { + upstreamController = controller; + controller.enqueue( + encoder.encode('event: content\ndata: {"content":"partial answer"}\n\n') + ); + }, + cancel() { + cancelCalls += 1; + }, + }), + { status: 200, headers: { "Content-Type": "text/event-stream" } } + ) + ); + const failures: Array<{ + status: number; + message: string; + code?: string; + type?: string; + }> = []; + + const result = await invokeStreamingChatCore(PARTIAL_IDENTITY, (failure) => + failures.push(failure) + ); + assert.equal(getCalls(), 1); + assert.equal(result.success, true, "real content must cross the readiness boundary"); + assert.ok(result.response.body); + const reader = result.response.body.getReader(); + let clientText = await readUntil(reader, "partial answer"); + + assert.ok(upstreamController); + upstreamController.enqueue( + encoder.encode( + `event: error\ndata: ${JSON.stringify({ + message: + "provider failed at /srv/omniroute/open-sse/executors/oneminai.ts:230 api_key=post-content-secret\nstack tail", + })}\n\n` + ) + ); + clientText += await readRemaining(reader); + + const roleIndex = clientText.indexOf('"role":"assistant"'); + const contentIndex = clientText.indexOf("partial answer"); + const errorIndex = clientText.indexOf('"error":'); + const doneIndex = clientText.indexOf("data: [DONE]"); + + assert.ok(roleIndex >= 0 && roleIndex < contentIndex, "the role must precede real content"); + assert.ok(contentIndex < errorIndex, "partial content must remain before the terminal error"); + assert.ok(errorIndex < doneIndex, "the pipeline error must precede [DONE]"); + assert.equal(clientText.match(/"role":"assistant"/g)?.length, 1); + assert.match(clientText, /"finish_reason":"error"/); + assert.match(clientText, /1min\.ai upstream stream failed/); + assert.doesNotMatch(clientText, /"finish_reason":"stop"/); + assert.doesNotMatch(clientText, /response\.failed/); + assert.doesNotMatch(clientText, /post-content-secret/); + assert.doesNotMatch(clientText, /\/srv\/omniroute/); + assert.doesNotMatch(clientText, /stack tail/); + + assert.equal(cancelCalls, 1, "the upstream source must be cancelled after its terminal error"); + assert.equal(failures.length, 1); + assert.deepEqual(failures[0], { + status: 502, + message: "1min.ai upstream stream failed", + code: "stream_pipeline_error", + type: "stream_error", + }); + const pending = usageHistory.getPendingRequests(); + assert.deepEqual(Object.keys(pending.byModel), []); + assert.deepEqual(Object.keys(pending.byAccount), []); + + const completed = [...usageHistory.getCompletedDetails().values()]; + assert.equal(completed.length, 1); + assert.equal(completed[0].status, 502); + assert.equal(completed[0].error, "1min.ai upstream stream failed"); + assert.equal(completed[0].errorCode, "stream_pipeline_error"); + + const detail = await waitFor(() => getOneMinCallLog(PARTIAL_IDENTITY)); + assert.ok(detail, "the post-content stream failure must be persisted"); + assert.equal(detail.status, 502); + assert.equal(detail.error, "1min.ai upstream stream failed"); + const persisted = JSON.stringify(detail); + assert.match(persisted, /1min\.ai upstream stream failed/); + assert.doesNotMatch(persisted, /post-content-secret/); + assert.doesNotMatch(persisted, /\/srv\/omniroute/); + assert.doesNotMatch(persisted, /stack tail/); + + const usage = await waitFor(() => getOneMinUsage(PARTIAL_IDENTITY)); + assert.ok(usage, "the post-content failure usage record must be persisted"); + assert.equal(usage.success, false); + assert.equal(usage.status, "502"); + assert.equal(usage.errorCode, "stream_pipeline_error"); +}); + +test("1min.ai error completion does not wait for an upstream cancel promise", async () => { + let cancelCalls = 0; + const getCalls = installFetchFactory( + () => + new Response( + new ReadableStream({ + start(controller) { + controller.enqueue( + encoder.encode('event: error\ndata: {"message":"capacity unavailable"}\n\n') + ); + }, + cancel() { + cancelCalls += 1; + return new Promise(() => {}); + }, + }), + { status: 200, headers: { "Content-Type": "text/event-stream" } } + ) + ); + + const result = await new OneMinAiExecutor().execute({ + model: "gpt-4o-mini", + body: { messages: [{ role: "user", content: "hello" }] }, + stream: true, + credentials: { apiKey: "unit-test-key" }, + signal: AbortSignal.timeout(10_000), + log: null, + }); + const clientText = await Promise.race([ + result.response.text(), + new Promise((_resolve, reject) => + setTimeout(() => reject(new Error("translated stream stayed pending on cancel")), 500) + ), + ]); + + assert.equal(getCalls(), 1); + assert.equal(cancelCalls, 1); + assert.match(clientText, /capacity unavailable/); + assert.match(clientText, /data: \[DONE\]/); +}); + +test("1min.ai propagates downstream cancellation without awaiting upstream cleanup", async () => { + let upstreamController: ReadableStreamDefaultController | null = null; + let cancelCalls = 0; + let markPullStarted: (() => void) | null = null; + const pullStarted = new Promise((resolve) => { + markPullStarted = resolve; + }); + const getCalls = installFetchFactory( + () => + new Response( + new ReadableStream({ + start(controller) { + upstreamController = controller; + controller.enqueue( + encoder.encode('event: content\ndata: {"content":"partial answer"}\n\n') + ); + }, + pull() { + markPullStarted?.(); + return new Promise(() => {}); + }, + cancel() { + cancelCalls += 1; + return new Promise(() => {}); + }, + }), + { status: 200, headers: { "Content-Type": "text/event-stream" } } + ) + ); + + const result = await new OneMinAiExecutor().execute({ + model: "gpt-4o-mini", + body: { messages: [{ role: "user", content: "hello" }] }, + stream: true, + credentials: { apiKey: "unit-test-key" }, + signal: AbortSignal.timeout(10_000), + log: null, + }); + assert.ok(result.response.body); + const reader = result.response.body.getReader(); + + try { + const clientText = await readUntil(reader, "partial answer"); + assert.match(clientText, /"role":"assistant"/); + await pullStarted; + await Promise.race([ + reader.cancel("client disconnected"), + new Promise((_resolve, reject) => + setTimeout(() => reject(new Error("downstream cancellation stayed pending")), 500) + ), + ]); + + assert.equal(getCalls(), 1); + assert.equal(cancelCalls, 1, "downstream cancellation must reach the upstream reader once"); + assert.deepEqual(await reader.read(), { value: undefined, done: true }); + } finally { + try { + upstreamController?.close(); + } catch { + // The fixed path has already cancelled and closed the upstream stream. + } + } +}); + +test("1min.ai accepts the bounded error-string shape without exposing a success chunk", async () => { + const response = await executeStreaming([ + 'event: error\ndata: {"error":"billing temporarily unavailable"}\n\n', + ]); + const clientText = await response.text(); + + assert.match(clientText, /"error":\{"message":"billing temporarily unavailable"/); + assert.doesNotMatch(clientText, /"role":"assistant"/); + assert.doesNotMatch(clientText, /"finish_reason":"stop"/); +}); + +test("1min.ai replaces oversized stream-error payloads with a fixed public fallback", async () => { + const oversizedMessage = `private-prefix-${"x".repeat(70 * 1024)}`; + const response = await executeStreaming([ + `event: error\ndata: ${JSON.stringify({ message: oversizedMessage })}\n\n`, + ]); + const clientText = await response.text(); + + assert.match(clientText, /1min\.ai upstream stream failed/); + assert.ok(clientText.length < 1_024, "the oversized upstream payload must not be reflected"); + assert.doesNotMatch(clientText, /private-prefix/); + assert.doesNotMatch(clientText, /"role":"assistant"/); + assert.doesNotMatch(clientText, /"finish_reason":"stop"/); +}); diff --git a/tests/unit/oneminai-stream-error-boundary.test.ts b/tests/unit/oneminai-stream-error-boundary.test.ts new file mode 100644 index 0000000000..88cc402e6e --- /dev/null +++ b/tests/unit/oneminai-stream-error-boundary.test.ts @@ -0,0 +1,91 @@ +import assert from "node:assert/strict"; +import { execFile } from "node:child_process"; +import { mkdirSync, mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import test from "node:test"; +import { fileURLToPath } from "node:url"; +import { promisify } from "node:util"; + +const execFileAsync = promisify(execFile); +const fixturePath = fileURLToPath( + new URL("../fixtures/oneminai-stream-error-boundary.fixture.ts", import.meta.url) +); + +type ChildFailure = Error & { + stdout?: string | Buffer; + stderr?: string | Buffer; +}; + +test( + "1min.ai stream-error boundary passes in an isolated persistence subprocess", + { timeout: 180_000 }, + async () => { + const originalDataDir = process.env.DATA_DIR; + const originalPluginsDir = process.env.OMNIROUTE_PLUGINS_DIR; + const originalFetch = globalThis.fetch; + const testRoot = mkdtempSync(join(tmpdir(), "omniroute-onemin-stream-error-child-")); + const testDataDir = join(testRoot, "data"); + const testPluginsDir = join(testRoot, "plugins"); + + mkdirSync(testDataDir, { recursive: true }); + mkdirSync(testPluginsDir, { recursive: true }); + const childEnv: NodeJS.ProcessEnv = { + APP_LOG_TO_FILE: "false", + DATA_DIR: testDataDir, + DISABLE_SQLITE_AUTO_BACKUP: "true", + NODE_ENV: "test", + OMNIROUTE_PLUGINS_DIR: testPluginsDir, + }; + for (const name of ["PATH", "NODE_PATH", "LANG", "LC_ALL", "TZ", "TMPDIR"] as const) { + const value = process.env[name]; + if (value !== undefined) childEnv[name] = value; + } + // A nested `node --test` must create its own runner context instead of + // inheriting the parent's private reporter channel. + delete childEnv.NODE_TEST_CONTEXT; + + try { + let stdout = ""; + let stderr = ""; + try { + const child = await execFileAsync( + process.execPath, + ["--import", "tsx/esm", "--test", "--test-concurrency=1", fixturePath], + { + cwd: process.cwd(), + encoding: "utf8", + env: childEnv, + maxBuffer: 2 * 1024 * 1024, + timeout: 170_000, + } + ); + stdout = child.stdout; + stderr = child.stderr; + } catch (error) { + const failure = error as ChildFailure; + assert.fail( + [ + `isolated 1min.ai fixture failed: ${failure.message}`, + failure.stdout ? String(failure.stdout) : "", + failure.stderr ? String(failure.stderr) : "", + ] + .filter(Boolean) + .join("\n") + ); + } + + const childOutput = `${stdout}\n${stderr}`; + assert.match(childOutput, /tests 8/); + assert.match(childOutput, /pass 8/); + assert.match(childOutput, /fail 0/); + assert.doesNotMatch(childOutput, /not ok|failed to drain|stayed pending/i); + } finally { + rmSync(testRoot, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + } + + assert.equal(process.env.DATA_DIR, originalDataDir); + assert.equal(process.env.OMNIROUTE_PLUGINS_DIR, originalPluginsDir); + assert.equal(globalThis.fetch, originalFetch); + } +); diff --git a/tests/unit/stream-readiness.test.ts b/tests/unit/stream-readiness.test.ts index 7e2ea9c242..04432fc62c 100644 --- a/tests/unit/stream-readiness.test.ts +++ b/tests/unit/stream-readiness.test.ts @@ -451,27 +451,24 @@ test("ensureStreamReadiness preserves buffered chunks when stream starts", async assert.match(text, / world/); }); -test("ensureStreamReadiness preserves its buffered prefix until a delayed consumer observes a later error", async () => { - const prefix = `data: ${JSON.stringify({ - object: "chat.completion.chunk", - choices: [ - { - index: 0, - delta: { role: "assistant", content: "prefix before failure" }, - finish_reason: null, - }, - ], - })}\n\n`; - let pullCount = 0; +test("ensureStreamReadiness replays buffered chunks before a subsequent source error", async () => { + let reads = 0; const response = new Response( new ReadableStream({ pull(controller) { - pullCount += 1; - if (pullCount === 1) { - controller.enqueue(encoder.encode(prefix)); + reads += 1; + if (reads === 1) { + controller.enqueue( + encoder.encode( + `data: ${JSON.stringify({ + object: "chat.completion.chunk", + choices: [{ index: 0, delta: { role: "assistant", content: "prefix" } }], + })}\n\n` + ) + ); return; } - controller.error(new Error("later upstream failure")); + controller.error(Object.assign(new Error("terminal source failure"), { statusCode: 502 })); }, }), { status: 200, headers: { "Content-Type": "text/event-stream" } } @@ -479,13 +476,96 @@ test("ensureStreamReadiness preserves its buffered prefix until a delayed consum const result = await ensureStreamReadiness(response, { timeoutMs: 100 }); assert.equal(result.ok, true); - await new Promise((resolve) => setTimeout(resolve, 25)); + assert.ok(result.response.body); + const reader = result.response.body.getReader(); + const first = await reader.read(); - const reader = result.response.body!.getReader(); + assert.equal(first.done, false); + assert.match(new TextDecoder().decode(first.value), /prefix/); + await assert.rejects(reader.read(), /terminal source failure/); +}); + +test("ensureStreamReadiness replays multiple buffered chunks in order before an error", async () => { + let reads = 0; + const response = new Response( + new ReadableStream({ + pull(controller) { + reads += 1; + if (reads === 1) { + controller.enqueue(encoder.encode(": keepalive\n\n")); + return; + } + if (reads === 2) { + controller.enqueue( + encoder.encode( + `data: ${JSON.stringify({ + object: "chat.completion.chunk", + choices: [{ index: 0, delta: { role: "assistant", content: "ready" } }], + })}\n\n` + ) + ); + return; + } + controller.error(new Error("failure after buffered prefix")); + }, + }), + { status: 200, headers: { "Content-Type": "text/event-stream" } } + ); + + const result = await ensureStreamReadiness(response, { timeoutMs: 100 }); + assert.equal(result.ok, true); + assert.ok(result.response.body); + const reader = result.response.body.getReader(); + const first = await reader.read(); + const second = await reader.read(); + + assert.equal(first.done, false); + assert.equal(second.done, false); + assert.match(new TextDecoder().decode(first.value), /keepalive/); + assert.match(new TextDecoder().decode(second.value), /ready/); + await assert.rejects(reader.read(), /failure after buffered prefix/); +}); + +test("ensureStreamReadiness cancellation is bounded when upstream cancel never settles", async () => { + let cancelCalls = 0; + const response = new Response( + new ReadableStream({ + start(controller) { + controller.enqueue( + encoder.encode( + `data: ${JSON.stringify({ + object: "chat.completion.chunk", + choices: [{ index: 0, delta: { role: "assistant", content: "prefix" } }], + })}\n\n` + ) + ); + }, + pull() { + return new Promise(() => {}); + }, + cancel() { + cancelCalls += 1; + return new Promise(() => {}); + }, + }), + { status: 200, headers: { "Content-Type": "text/event-stream" } } + ); + + const result = await ensureStreamReadiness(response, { timeoutMs: 100 }); + assert.equal(result.ok, true); + assert.ok(result.response.body); + const reader = result.response.body.getReader(); const first = await reader.read(); assert.equal(first.done, false); - assert.match(new TextDecoder().decode(first.value), /prefix before failure/); - await assert.rejects(() => reader.read(), /later upstream failure/); + + await Promise.race([ + reader.cancel("client disconnected"), + new Promise((_resolve, reject) => + setTimeout(() => reject(new Error("readiness cancellation stayed pending")), 500) + ), + ]); + await reader.cancel("duplicate cancellation"); + assert.equal(cancelCalls, 1); }); test("ensureStreamReadiness honors configured timeouts above 2000ms", async () => { From 5ba42476700b9953e048f544ce3048394499f2c6 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 21:17:17 -0300 Subject: [PATCH 071/143] chore(quality): rebaseline file-size caps the error-boundary campaign grew past (#12654) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rebaseline medido no tip com os 14 PRs da campanha mergeados. A anotação registra que 4 das 6 linhas do codex.ts são drift anterior à campanha, não crescimento dela. Não toca stream.ts. --- .../maintenance/error-boundary-campaign-filesize.md | 1 + config/quality/file-size-baseline.json | 7 ++++--- 2 files changed, 5 insertions(+), 3 deletions(-) create mode 100644 changelog.d/maintenance/error-boundary-campaign-filesize.md diff --git a/changelog.d/maintenance/error-boundary-campaign-filesize.md b/changelog.d/maintenance/error-boundary-campaign-filesize.md new file mode 100644 index 0000000000..f57eef1c49 --- /dev/null +++ b/changelog.d/maintenance/error-boundary-campaign-filesize.md @@ -0,0 +1 @@ +- **chore(quality):** rebaseline the file-size caps the error-boundary campaign grew past (`open-sse/executors/codex.ts`, `open-sse/vendor/codex-chatgpt-web/bridge.ts`, both via [#12444](https://github.com/diegosouzapw/OmniRoute/pull/12444)) diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index c39d3d5d41..97b881c510 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -412,7 +412,7 @@ "open-sse/executors/antigravity.ts": 1665, "open-sse/executors/base.ts": 1751, "open-sse/executors/chatgpt-web.ts": 5056, - "open-sse/executors/codex.ts": 1499, + "open-sse/executors/codex.ts": 1505, "open-sse/executors/cursor.ts": 1759, "open-sse/executors/muse-spark-web.ts": 1405, "open-sse/handlers/chatCore.ts": 5984, @@ -428,7 +428,7 @@ "open-sse/utils/proxyFetch.ts": 1271, "open-sse/utils/stream.ts": 3072, "open-sse/vendor/codex-chatgpt-web/adapters/chatgpt-web/browser-worker.ts": 4398, - "open-sse/vendor/codex-chatgpt-web/bridge.ts": 1322, + "open-sse/vendor/codex-chatgpt-web/bridge.ts": 1335, "src/app/(dashboard)/dashboard/HomePageClient.tsx": 1344, "src/app/(dashboard)/dashboard/api-manager/ApiManagerPageClient.tsx": 3186, "src/app/(dashboard)/dashboard/combos/page.tsx": 5018, @@ -636,5 +636,6 @@ "_rebaseline_2026_09_02_12325_merge_v3851": "Merge of release/v3.8.51 into #12325. Both sides grew chatCore.ts at the same chokepoint: #12239 took it 5946->5976 upstream, and this PR adds its +9 non-Codex 429 branch on top. check-file-size.mjs counts split(\"\\\\n\").length (trailing-newline empty element), so the merged file is 5981. The cap is the merged LOC, not either side alone; no other entry moves.", "_rebaseline_2026_09_03_houminxi_batch_stacked": "Crescimento medido DEPOIS que os 9 PRs da leva HouMinXi entraram, quando cada um empilhou sobre o rebaseline do anterior: providers/page.tsx 2007->2025 (+18 = feedback de erro por linha do import CSV do #12504 somado a busca por nome/baseUrl do #12495, ambos no mesmo painel de conexoes); chatCore.ts 5981->5984 (+3 = o #12325 invalida o cache generico de quota no 429 upstream, ao lado do ramo Codex ja existente); accountFallback.ts 2461->2467 (+6 = o #12566 empilha a carve-out de familia Antigravity sobre o rebaseline 2422->2461 que o #12590 registrou para o carve-out credits_exhausted da Moonshot; os dois tocam checkFallbackError). Cada PR mediu certo isoladamente, mas nenhum enxergava o empilhamento. Fiacao em chokepoints existentes. NAO cobre codex.ts nem stream.ts, que ja violavam no tip antes desta leva (drift da base).", "_rebaseline_2026_09_03_12604_claude_code_2_1_258": "PR #12604 (bump da wire identity do Claude Code 2.1.220->2.1.258, commits do @ggiak vindos do #12402) crescimento proprio: src/app/(dashboard)/dashboard/settings/components/RoutingTab.tsx 1606->1607 (+1, a linha do seletor que acompanha a nova versao de identidade). Uma linha num painel de settings ja existente; nao ha o que extrair. Coberto por client-identity-profiles e claude-codex-identity-version-sync (138/138 focados).", - "_rebaseline_2026_09_03_hartmark_batch": "Leva hartmark (#12293 #12355 #12447 #12445 #12446 #12460 #12461 #12338 #12448) crescimento proprio, medido no tip com os nove mergeados: src/app/(dashboard)/dashboard/combos/page.tsx 5012->5018 (+6, #12355 impede que a falha de bundling do tiktoken de um provider sem relacao derrube /api/providers, e o painel passa a lidar com o estado degradado); open-sse/services/combo.ts 4023->4036 (+13, #12338 nos fixes do universal-handoff: nota de bare-fallback, escopo por mesma requisicao e log da falha silenciosa). Fiacao em chokepoints existentes do roteamento de combo. NAO cobre codex.ts nem stream.ts, ja violando no tip antes desta leva (drift da base)." + "_rebaseline_2026_09_03_hartmark_batch": "Leva hartmark (#12293 #12355 #12447 #12445 #12446 #12460 #12461 #12338 #12448) crescimento proprio, medido no tip com os nove mergeados: src/app/(dashboard)/dashboard/combos/page.tsx 5012->5018 (+6, #12355 impede que a falha de bundling do tiktoken de um provider sem relacao derrube /api/providers, e o painel passa a lidar com o estado degradado); open-sse/services/combo.ts 4023->4036 (+13, #12338 nos fixes do universal-handoff: nota de bare-fallback, escopo por mesma requisicao e log da falha silenciosa). Fiacao em chokepoints existentes do roteamento de combo. NAO cobre codex.ts nem stream.ts, ja violando no tip antes desta leva (drift da base).", + "_rebaseline_2026_09_03_error_boundary_campaign": "Campanha de error-boundary (#12431 #12438 #12444 #12454 #12455 #12456 #12457 #12458 #12459 #12465 #12466 #12467 #12469 #12435), medido no tip com os 14 mergeados. open-sse/executors/codex.ts 1499->1505: os primeiros 4 (1499->1503) sao DRIFT ANTERIOR a esta campanha, ja presente no tip antes dela; os 2 ultimos (1503->1505) sao do #12444, que fecha o boundary de falha da resposta do Codex. Absorver o drift junto foi inevitavel porque o cap e um numero so, mas fica registrado aqui que 4 das 6 linhas nao sao desta leva. open-sse/vendor/codex-chatgpt-web/bridge.ts 1322->1335 (+13): tambem do #12444, no mesmo caminho de falha. NAO cobre open-sse/utils/stream.ts, que segue violando por drift anterior e independente." } From 109cf0f26c67a479079eeef0917c295ba80c8b46 Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Thu, 3 Sep 2026 20:30:32 -0400 Subject: [PATCH 072/143] fix(providers): reclassify Cerebras as a one-time $5 signup credit (#12591) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado sobre o tip de release/v3.8.51 — e o PR ficou completo depois que o autor adicionou a tabela de preços. O que fazia o teste falhar antes não era a lista free (que o PR já tinha corrigido em `LEGACY_FREE_PROVIDERS` e `tierDefaults.json`), e sim que `classifyTier` cai no ramo cost-based e todos os modelos Cerebras estavam declarados com `input: 0, output: 0` — $0/M ≤ threshold devolve 'free' de qualquer jeito. A tabela de preços resolve isso na raiz. Confirmei o `gpt-oss-120b` a $0,35/$0,75 de forma independente contra a fonte pública, o que corrobora o resto da tabela. **4/4** no teste-guarda e **25/25** somando free-tier-catalog e free-models; typecheck:core limpo. Obrigado, @HouMinXi. --- README.md | 18 ++++---- changelog.d/fixes/11773-cerebras-free-tier.md | 1 + docs/diagrams/README.md | 2 +- docs/diagrams/free-tier-budget.svg | 10 ++-- docs/diagrams/promise-pillars.svg | 4 +- docs/diagrams/readme-hero.svg | 4 +- docs/getting-started/FREE-TIERS-GUIDE.md | 4 +- docs/getting-started/PROVIDERS-GUIDE.md | 2 +- docs/reference/FREE_TIERS.md | 18 ++++---- docs/reference/PROVIDER_REFERENCE.md | 2 +- docs/screenshots/free-tier-budget-card.svg | 4 +- open-sse/config/freeModelCatalog.data.ts | 9 ++-- open-sse/config/freeTierCatalog.ts | 1 - .../services/__tests__/tierResolver.test.ts | 7 ++- open-sse/services/tierConfig.ts | 1 - open-sse/services/tierDefaults.json | 1 - src/i18n/messages/en.json | 2 +- src/i18n/messages/es.json | 2 +- .../constants/pricing/inference-hosts.ts | 21 +++++---- .../providers/apikey/inference-hosts.ts | 7 ++- tests/unit/cerebras-free-tier-11773.test.ts | 46 +++++++++++++++++++ tests/unit/check-docs-counts-sync.test.ts | 12 ++--- tests/unit/free-note-freshness.test.ts | 7 ++- tests/unit/free-tier-catalog.test.ts | 13 +++--- 24 files changed, 129 insertions(+), 69 deletions(-) create mode 100644 changelog.d/fixes/11773-cerebras-free-tier.md create mode 100644 tests/unit/cerebras-free-tier-11773.test.ts diff --git a/README.md b/README.md index 9814b0e62d..ca92188728 100644 --- a/README.md +++ b/README.md @@ -7,19 +7,19 @@ # 🚀 OmniRoute — The Free AI Gateway -OmniRoute — Never stop coding. Every AI tool → 356 providers — 150+ free — through one endpoint. Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini with auto-fallback. RTK + Caveman stacked compression saves 15–95% tokens (~89% avg) — never hit limits. 356 AI providers · 150+ free tiers · ~1.51B free tokens/mo · 19 routing strategies · $0 to start. +OmniRoute — Never stop coding. Every AI tool → 356 providers — 150+ free — through one endpoint. Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini with auto-fallback. RTK + Caveman stacked compression saves 15–95% tokens (~89% avg) — never hit limits. 356 AI providers · 150+ free tiers · ~1.48B free tokens/mo · 19 routing strategies · $0 to start.
-## 💰 ~1.51B Free Tokens / Month +## 💰 ~1.48B Free Tokens / Month
-> Stacking free tiers by hand is painful — dozens of SDKs, dozens of rate limits, and no idea how much you actually have. OmniRoute catalogs **437 free-tier entries across 38 recurring pool keys** and computes the token headline from the **20 pools with a published positive monthly budget**, deduplicated by shared pool. The result stays visible on the dashboard (`/dashboard/free-tiers`). +> Stacking free tiers by hand is painful — dozens of SDKs, dozens of rate limits, and no idea how much you actually have. OmniRoute catalogs **437 free-tier entries across 37 recurring pool keys** and computes the token headline from the **20 pools with a published positive monthly budget**, deduplicated by shared pool. The result stays visible on the dashboard (`/dashboard/free-tiers`). -OmniRoute free-tier budget card: ~1.51B free tokens per month steady, up to ~2.13B in the first month with signup credits, from 38 documented recurring pool keys covering 437 cataloged free-tier entries behind one endpoint. Honest pool-deduped math — each shared pool counted once, including 20 recurring pools with a published positive monthly token budget; 13 providers are marked avoid in the terms-risk catalog so you decide. Budget bar includes Mistral 1B, LLM7 150M, Nara 150M, Gemini 60M and smaller pools, plus first-month signup credits and permanently-free no-token-cap providers surfaced separately so they never inflate the headline. Live used/remaining on /dashboard/free-tiers. +OmniRoute free-tier budget card: ~1.48B free tokens per month steady, up to ~2.10B in the first month with signup credits, from 37 documented recurring pool keys covering 437 cataloged free-tier entries behind one endpoint. Honest pool-deduped math — each shared pool counted once, including 20 recurring pools with a published positive monthly token budget; 13 providers are marked avoid in the terms-risk catalog so you decide. Budget bar includes Mistral 1B, LLM7 150M, Nara 150M, Gemini 60M and smaller pools, plus first-month signup credits and permanently-free no-token-cap providers surfaced separately so they never inflate the headline. Live used/remaining on /dashboard/free-tiers. > Animated summary of the live `/dashboard/free-tiers` page. Full methodology (pool dedupe, credit tiers, provider terms): **[docs/reference/FREE_TIERS.md](docs/reference/FREE_TIERS.md)**. > @@ -209,7 +209,7 @@ curl http://localhost:20128/v1/chat/completions \ -The Promise — One endpoint and 356 providers. Automatic fallback keeps routing while another healthy target is available. Six pillars: resilient fallback across 356 providers · up to 95% token savings on eligible workloads · $0 to start with 150+ free tiers and 53 recurring/keyless free-forever providers · 36 CLI/agent integrations through one config · OpenAI, Claude, Gemini and Responses API compatibility at /v1 · production controls including circuit breakers, TLS stealth, MCP 110 tools, A2A, memory, guardrails, evals and 39,000+ static test declarations across 5,100+ tracked test files. +The Promise — One endpoint and 356 providers. Automatic fallback keeps routing while another healthy target is available. Six pillars: resilient fallback across 356 providers · up to 95% token savings on eligible workloads · $0 to start with 150+ free tiers and 52 recurring/keyless free-forever providers · 36 CLI/agent integrations through one config · OpenAI, Claude, Gemini and Responses API compatibility at /v1 · production controls including circuit breakers, TLS stealth, MCP 110 tools, A2A, memory, guardrails, evals and 39,000+ static test declarations across 5,100+ tracked test files.

@@ -518,9 +518,9 @@ Pix copia-e-cola: ## 📡 OmniRoute Radar -The main free-tier headline remains **~1.51B tokens/month** from the documented, +The main free-tier headline remains **~1.48B tokens/month** from the documented, pool-deduplicated catalog above. Temporary provider signup credits can separately lift the first -month to **~2.13B**. Radar is an optional, signed catalog overlay for people who want fresher +month to **~2.10B**. Radar is an optional, signed catalog overlay for people who want fresher free-model availability between OmniRoute releases; the community catalog and every existing free feature remain free. @@ -648,7 +648,7 @@ of your shell history. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md) -> **352 registered providers** across the canonical chat, media, search, local, cloud-agent and system collections, including **152 carrying `hasFree: true` discovery metadata**. The chat model registry covers **229 providers / 2,554 distinct provider-model pairs / 1,283 raw model IDs**; the separate free-budget catalog has **437 per-model rows**, **38 recurring pools** and **53 recurring/keyless free-forever providers**. These are different denominators by design; definitions and pool-deduped calculations live in the [Provider Reference](docs/reference/PROVIDER_REFERENCE.md) and [Free Tiers](docs/reference/FREE_TIERS.md). +> **352 registered providers** across the canonical chat, media, search, local, cloud-agent and system collections, including **152 carrying `hasFree: true` discovery metadata**. The chat model registry covers **229 providers / 2,554 distinct provider-model pairs / 1,283 raw model IDs**; the separate free-budget catalog has **437 per-model rows**, **37 recurring pools** and **52 recurring/keyless free-forever providers**. These are different denominators by design; definitions and pool-deduped calculations live in the [Provider Reference](docs/reference/PROVIDER_REFERENCE.md) and [Free Tiers](docs/reference/FREE_TIERS.md).
@@ -1307,7 +1307,7 @@ Métricas canônicas em 2026-08-24: **1.029 vídeos únicos** · **11.132.922 vi Resilience GuideCircuit breakers, cooldowns, queue, anti-thundering herd, TLS spoofing Auto-Combo Engine16-factor scoring, mode packs, self-healing Proxy Guide3-level proxy system, 1proxy marketplace, registry CRUD - Free TiersConsolidated directory: 38 documented recurring pools / 437 cataloged free-tier entries + Free TiersConsolidated directory: 37 documented recurring pools / 437 cataloged free-tier entries Features GalleryVisual dashboard tour with screenshots Codebase DocumentationBeginner-friendly codebase walkthrough diff --git a/changelog.d/fixes/11773-cerebras-free-tier.md b/changelog.d/fixes/11773-cerebras-free-tier.md new file mode 100644 index 0000000000..ecc67975de --- /dev/null +++ b/changelog.d/fixes/11773-cerebras-free-tier.md @@ -0,0 +1 @@ +- **fix(providers):** reclassify Cerebras as a one-time $5 signup credit (payment method required, 30-day validity), not a recurring no-card 1M tokens/day trial ([#11773](https://github.com/diegosouzapw/OmniRoute/issues/11773)) diff --git a/docs/diagrams/README.md b/docs/diagrams/README.md index 62ef3efff9..1382f41263 100644 --- a/docs/diagrams/README.md +++ b/docs/diagrams/README.md @@ -34,7 +34,7 @@ inside GitHub's `` sandbox: | [combo-always-on.svg](./combo-always-on.svg) | style reference | Animated priority-combo fallback (4 layers, 16s loop). Edit the SVG directly — there is no `.mmd` source. | | [cli-terminal.svg](./cli-terminal.svg) | README.md (root) | Compact half-height animated terminal (1200×350): 3 real CLI commands cycling with typewriter + scrolling subcommand ticker; first frame = completed providers screen. Edit the SVG directly — there is no `.mmd` source. | | [compression-pipeline.svg](./compression-pipeline.svg) | README.md (root) | Animated 12-engine compression funnel (8s loop). Edit the SVG directly — there is no `.mmd` source. | -| [free-tier-budget.svg](./free-tier-budget.svg) | README.md (root) | Animated free-tier budget card (~1.51B/mo quantified headline, 20-pool budget bar, per-pool grid, signup credits, 10s loop). Edit the SVG directly — there is no `.mmd` source. | +| [free-tier-budget.svg](./free-tier-budget.svg) | README.md (root) | Animated free-tier budget card (~1.48B/mo quantified headline, 20-pool budget bar, per-pool grid, signup credits, 10s loop). Edit the SVG directly — there is no `.mmd` source. | | [readme-hero.svg](./readme-hero.svg) | README.md (root) | Animated hero card (tagline, live provider/free-access headline, full-width compression bar demo, 6 stat chips). Edit the SVG directly — there is no `.mmd` source. | | [promise-pillars.svg](./promise-pillars.svg) | README.md (root) | Animated "The Promise" 6-pillar card (12s border-highlight sweep). Edit the SVG directly — there is no `.mmd` source. | | [why-pain-fix.svg](./why-pain-fix.svg) | README.md (root) | Animated "Why OmniRoute" 10-row pain-vs-fix ledger (15s green row sweep). Edit the SVG directly — there is no `.mmd` source. | diff --git a/docs/diagrams/free-tier-budget.svg b/docs/diagrams/free-tier-budget.svg index 72d1219cb2..cdee0ab39c 100644 --- a/docs/diagrams/free-tier-budget.svg +++ b/docs/diagrams/free-tier-budget.svg @@ -1,4 +1,4 @@ - + Pool-deduplicated chart of the 20 recurring free-token pools with positive published budgets, plus signup credits and uncapped providers shown separately. @@ -61,10 +61,10 @@ - ~1.51B + ~1.48B FREE TOKENS / MONTH · STEADY - up to ~2.13B in your first month — signup credits - documented free tiers · 38 recurring pools · 437 catalog entries · one endpoint + up to ~2.10B in your first month — signup credits + documented free tiers · 37 recurring pools · 437 catalog entries · one endpoint @@ -75,7 +75,7 @@ every rate limit · 24/7 we don't publish that - ~1.51B + ~1.48B each shared free pool counted once ✓ 13 providers ToS-flagged — we flag it · you decide diff --git a/docs/diagrams/promise-pillars.svg b/docs/diagrams/promise-pillars.svg index 41bcdf397c..43d5fb8381 100644 --- a/docs/diagrams/promise-pillars.svg +++ b/docs/diagrams/promise-pillars.svg @@ -1,4 +1,4 @@ - + Animated promise card: six pillar tiles fade in in reading order, then a soft colored border highlight sweeps from tile to tile in a continuous cycle. @@ -73,7 +73,7 @@ $0 to start - 150+ providers with a free tier, 53 free + 150+ providers with a free tier, 52 free forever — Qoder, Pollinations, Cloudflare, SiliconFlow… No card needed. diff --git a/docs/diagrams/readme-hero.svg b/docs/diagrams/readme-hero.svg index 2fc0a31918..7dc4d749a2 100644 --- a/docs/diagrams/readme-hero.svg +++ b/docs/diagrams/readme-hero.svg @@ -1,4 +1,4 @@ - + Animated hero card: a pulse travels the divider line and a compression bar demo repeatedly shrinks a prompt by up to 95 percent; all headline content is static and readable on the first frame. @@ -72,7 +72,7 @@ 90+ FREE TIERS - ~1.51B + ~1.48B FREE TOKENS / MO 15–95% diff --git a/docs/getting-started/FREE-TIERS-GUIDE.md b/docs/getting-started/FREE-TIERS-GUIDE.md index fe7ebbcc49..4108f0babe 100644 --- a/docs/getting-started/FREE-TIERS-GUIDE.md +++ b/docs/getting-started/FREE-TIERS-GUIDE.md @@ -161,8 +161,8 @@ The live, pool-deduplicated catalog currently reports: | Metric | Current audited value | Interpretation | | ---------------------------------------------------- | -----------------------------------------------: | ----------------------------------------------------------------------------------------- | -| Recurring quantified grant | **~1.51B tokens/month** | Shared pools counted once; excludes uncapped providers from the sum | -| First month with signup grants | **~2.13B tokens** | Recurring total plus one-time and recurring credits | +| Recurring quantified grant | **~1.48B tokens/month** | Shared pools counted once; excludes uncapped providers from the sum | +| First month with signup grants | **~2.10B tokens** | Recurring total plus one-time and recurring credits | | Audited free-model inventory | **39 recurring pool keys / 445 catalog entries** | 438 active + 7 discontinued; distinct from the 351-provider catalog | | Recurring/keyless free-forever providers represented | **55** | Unique providers across recurring daily/monthly/credit/uncapped and keyless catalog types | | Provider catalog entries marked `hasFree` | **152 / 351** | Broader provider metadata; not all have a quantifiable recurring quota | diff --git a/docs/getting-started/PROVIDERS-GUIDE.md b/docs/getting-started/PROVIDERS-GUIDE.md index 1ac7bbfecf..c65b6ad647 100644 --- a/docs/getting-started/PROVIDERS-GUIDE.md +++ b/docs/getting-started/PROVIDERS-GUIDE.md @@ -183,7 +183,7 @@ These providers offer **free access** with no credit card: | **LongCat** | 10M one-time | LongCat-2.0 | API key + KYC | | **Cloudflare AI** | 10K neurons/day | 50+ models | No auth needed | | **NVIDIA NIM** | ~40 RPM | 129 models | API key needed | -| **Cerebras** | 1M tokens/day | Qwen3 235B, GPT-OSS 120B | API key needed | +| **Cerebras** | $5 signup credit | GLM 4.7, GPT-OSS 120B | API key + card | | **Qoder** | Unlimited | Kimi-K2, DeepSeek-R1, Qwen3-coder | No auth needed | **Tip**: Connect multiple free providers for **unlimited free AI** with automatic fallback! diff --git a/docs/reference/FREE_TIERS.md b/docs/reference/FREE_TIERS.md index de03e3712b..0decf6fb3f 100644 --- a/docs/reference/FREE_TIERS.md +++ b/docs/reference/FREE_TIERS.md @@ -15,21 +15,23 @@ lastUpdated: 2026-08-31 | Metric | Tokens / month | Meaning | | ------------------------------------------- | ----------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| **Documented recurring grant (steady)** | **~1.51B** | Free-tier **pools** (per-model catalog), each shared pool counted **once**. The live source behind `/api/free-tier/summary` and the dashboard's Free-Tier Budget page. **Use this number.** | -| **+ first month with signup credits** | **~2.13B** | Steady + one-time signup credits (Together $25, Z.AI 20M, DeepSeek 5M, …), deduped per account. **First month only** — does not recur. | +| **Documented recurring grant (steady)** | **~1.48B** | Free-tier **pools** (per-model catalog), each shared pool counted **once**. The live source behind `/api/free-tier/summary` and the dashboard's Free-Tier Budget page. **Use this number.** | +| **+ first month with signup credits** | **~2.10B** | Steady + one-time signup credits (Together $25, Z.AI 20M, DeepSeek 5M, …), deduped per account. **First month only** — does not recur. | | **+ permanently free, no published cap** | _un-quantifiable_ | `siliconflow`, `glm-cn` (GLM-4-Flash), `tencent`, `baidu`, `kilo-gateway`, `opencode-zen` — real recurring access, rate/concurrency-limited, **no token cap to count**. Listed, never summed (counting them at `RPM×24/7` is the inflation we reject). | | **+ deposit-unlock boost** | **+~24M** | A one-time **$10** OpenRouter top-up raises its free pool from 50 → 1000 req/day. Reported separately so it never inflates the steady number. | | Theoretical ceiling (all rate limits, 24/7) | ~10B | Sum of every provider rate limit extrapolated to non-stop use. **Not a guarantee** — do not headline this. | -**Honest headline:** _OmniRoute aggregates **~1.51B documented free tokens per month** (up to ~2.13B in your first month with signup credits) across 38 free-tier pools — plus a long tail of permanently-free, no-cap providers — and RTK + Caveman compression (15–95% token savings) stretches that further._ +**Honest headline:** _OmniRoute aggregates **~1.48B documented free tokens per month** (up to ~2.10B in your first month with signup credits) across 37 free-tier pools — plus a long tail of permanently-free, no-cap providers — and RTK + Caveman compression (15–95% token savings) stretches that further._ > **Why this dropped from the previous ~1.94B.** The 2026-06-17 refresh is an honesty correction, not a loss: `gemini` is now pool-deduped (was inflated by counting each Flash variant separately, 462M → 60M), `cloudflare-ai` corrected to its real 10k-Neurons/day (122M → 30M), `doubao` reclassified as a one-time signup credit (not recurring), and shut-down tiers removed (`chutes`/`phind`/`kluster` discontinued). Partly offset by `llm7` (correct 5M/day → 150M) and new free providers (Kilo, OpenCode Zen, Z.AI GLM-Flash). > > **Further corrected to ~1.37B in v3.8.42:** `longcat` was reclassified from a 150M/mo recurring grant to a one-time 10M signup credit after its free preview ended. Same honesty rule — no provider was dropped by mistake. > -> **Updated on 2026-08-26 after retiring Felo Web:** the source now reports 38 recurring pool keys. Felo Web is excluded while its GPL-derived provenance/licensing remains on HOLD. This is the live, CI-gated number (`check:docs-counts` fails the build if this drifts from `computeFreeModelTotals()`). +> **Corrected to ~1.48B on 2026-09-03 (#11773):** `cerebras` was reclassified from a 30M/mo recurring grant (old no-card 1M tokens/day trial) to a one-time $5 signup credit that requires a payment method. Same honesty rule as LongCat. +> +> **Updated on 2026-08-26 after retiring Felo Web:** the source now reports 37 recurring pool keys. Felo Web is excluded while its GPL-derived provenance/licensing remains on HOLD. This is the live, CI-gated number (`check:docs-counts` fails the build if this drifts from `computeFreeModelTotals()`). -Biggest **documented** contributors: `mistral` 1.00B, `llm7` 150M, `nara` 150M, `gemini` 60M, `cerebras` 30M, `cloudflare-ai` 30M, `api-airforce` 24M. (`longcat` is excluded — its 10M LongCat-2.0 grant is a one-time, KYC-gated signup credit, not a recurring monthly budget.) +Biggest **documented** contributors: `mistral` 1.00B, `llm7` 150M, `nara` 150M, `gemini` 60M, `cloudflare-ai` 30M, `api-airforce` 24M. (`longcat` is excluded — its 10M LongCat-2.0 grant is a one-time, KYC-gated signup credit, not a recurring monthly budget.) > ⚠️ The theoretical ceiling (~10B) is inflated by rate-limit-only providers with **no published token cap** (`tencent`, `siliconflow`, `nvidia`, `baidu`, `glm-cn`, `sparkdesk`) whose figures would be `RPM/TPM × 24/7 × 30d` — a theoretical maximum no single account will sustain. They are **excluded** from the defensible number (shown in the "permanently free, no cap" row instead). This is the same inflation that makes competitors' multi-billion claims unreliable. @@ -69,7 +71,7 @@ purpose. ## Methodology & caveats - Numbers are **upper-bound estimates** from each provider's documented free-tier limits as of **2026-06-17**, gathered by web research. Free tiers change constantly — re-verify before relying on a figure. -- **What an entry actually vouches for.** No entry carries a per-row confidence rating, and the API serves none — treat every figure above as an estimate of the same, unstated quality. Two facts are different, because they are curated by hand rather than inferred: 7 entries carry an independently documented hard stop, and 13 entries carry a prompt-training disclosure. `hardStopGuaranteed` is set only when the provider's own terms say that exceeding the free allowance refuses the request rather than silently starting to bill you, with the source in a comment next to the entry; it is never defaulted to `true`, and an entry nobody has verified stays unset. So a missing hard-stop flag means "not established", not "known to bill you". +- **What an entry actually vouches for.** No entry carries a per-row confidence rating, and the API serves none — treat every figure above as an estimate of the same, unstated quality. Two facts are different, because they are curated by hand rather than inferred: 5 entries carry an independently documented hard stop, and 13 entries carry a prompt-training disclosure. `hardStopGuaranteed` is set only when the provider's own terms say that exceeding the free allowance refuses the request rather than silently starting to bill you, with the source in a comment next to the entry; it is never defaulted to `true`, and an entry nobody has verified stays unset. So a missing hard-stop flag means "not established", not "known to bill you". - `estMonthlyFreeTokens` = recurring monthly tokens only. **One-time signup credits do not recur** and count as 0. Discontinued tiers are also 0. - Daily token cap → `monthly = daily × 30`. Only RPD documented → `RPD × ~800 output tokens × 30`. Only RPM/TPM (no daily cap) → **uncapped** (see below). - **Permanently free, but no published token cap** (`siliconflow`, `glm-cn`, `tencent`, `baidu`, `kilo-gateway`, `opencode-zen`): these are real recurring free access, rate/concurrency-limited. We classify them `recurring-uncapped` and **never sum them** — multiplying `RPM × 24/7 × 30d` would produce a fantasy ceiling (the inflation we reject). They are listed so you know they exist. @@ -193,7 +195,7 @@ purpose. | `llm7` | recurring | ~150M | — | caution | 4 | | `longcat` | one-time | — | 10M | caution | 1 | | `gemini` | recurring | ~60M | — | caution | 4 | -| `cerebras` | recurring | ~30M | — | caution | 2 | +| `cerebras` | one-time | — | $5 credit | caution | 2 | | `cloudflare-ai` | recurring | ~30M | — | caution | 9 | | `api-airforce` | recurring | ~24M | — | caution | 7 | | `ollama-cloud` | recurring | ~20M | — | ambiguous | 8 | @@ -276,7 +278,7 @@ purpose. - **`bluesminds`** — Our shipped freeNote was "(none)" — but BluesMinds does have a documented free tier: 500 pi credits, 20 RPM, 300 RPD, permanent free plan. The catalog significantly understates the offering. - **`brave-search`** — The catalog notes "(none)" suggesting no free tier was tracked, but in reality there was a free 5,000 queries/month tier (no card) until February 12, 2026, which has since been replaced by a $5/month… - **`byteplus`** — Our catalog shipped "(none)" but BytePlus ModelArk does have a free tier: a one-time trial credit of 500k tokens per LLM model for new accounts. The catalog underreports this. -- **`cerebras`** — TPM appears tightened from 60K to 30K on current documented models (gpt-oss-120b, zai-glm-4.7). RPM of 5 is now explicitly documented (was not in our shipped note). Daily token cap of 1M/day is uncha… +- **`cerebras`** — The no-card 1M tokens/day trial is gone. Live cerebras.ai/pricing (2026-09-03) is a one-time $5 signup credit, payment method required, 30-day validity. Reclassified as `one-time-initial` (LongCat-shaped); dropped from `LEGACY_FREE_PROVIDERS` and the recurring budget. - **`chutes`** — The shipped freeNote says "Free tier available" but as of March 15, 2026, the free tier has been officially discontinued. The catalog note is stale and should be updated to reflect that there is no r… - **`coze`** — The shipped note "Free ByteDance agent platform" is directionally accurate but omits that the free tier is now tightly credit-capped (10 credits/day ≈ 5–100 messages depending on model), a constraint… - **`deepinfra`** — Our shipped freeNote says "Free signup credits for API testing" — this appears stale. The official pricing page now requires card/prepayment with no documented general free signup credit. The free ti… diff --git a/docs/reference/PROVIDER_REFERENCE.md b/docs/reference/PROVIDER_REFERENCE.md index c7ba75f601..e588bb04a3 100644 --- a/docs/reference/PROVIDER_REFERENCE.md +++ b/docs/reference/PROVIDER_REFERENCE.md @@ -151,7 +151,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `bluesminds` | `bm` | BluesMinds | API key | [link](https://www.bluesminds.com) | Free daily pi credits — supports 200+ models including GPT-4o, GPT-4.1, Claude Sonnet 4.5, Gemini 2.0 Flash, DeepSeek V4, Qwen, Kimi K2 | | `byteplus` | `bpm` | BytePlus ModelArk | API key | [link](https://console.byteplus.com/ark) | — | | `bytez` | `bytez` | Bytez | API key | [link](https://bytez.com) | $1 free credits, refreshes every 4 weeks | -| `cerebras` | `cerebras` | Cerebras | API key | [link](https://inference.cerebras.ai) | Free Trial: 1M tokens/day, 30K TPM, 5 RPM — no credit card. | +| `cerebras` | `cerebras` | Cerebras | API key | [link](https://inference.cerebras.ai) | One-time $5 signup credit (30-day validity); a payment method is required. Not a recurring free tier. | | `charm-hyper` | `charm-hyper` | Charm Hyper | API key | [link](https://hyper.charm.land) | 100 free monthly Hypercredits on signup | | `chat-oripe` | `chat-oripe` | Chat Oripe | API key, aggregator | [link](https://api.oriper.com) | Official metadata advertises 2M tokens/month, but the public site and documentation were blocked during audit; treat the quota and brand mapping as unconfirmed. | | `chatanywhere` | `chatanywhere` | ChatAnywhere | API key, aggregator | [link](https://chatanywhere.tech) | Personal, educational or research use only: public documentation cites 10,000 points/day and 200 requests/day per IP/key; do not use for commercial traffic. | diff --git a/docs/screenshots/free-tier-budget-card.svg b/docs/screenshots/free-tier-budget-card.svg index c56452439a..90ab6b0279 100644 --- a/docs/screenshots/free-tier-budget-card.svg +++ b/docs/screenshots/free-tier-budget-card.svg @@ -5,9 +5,9 @@ Monthly free-token budget 20 free pools · 446 models · one endpoint Steady / month -~1.51B +~1.48B First month (+ signup credits) -~2.13B +~2.10B ToS-flagged (you decide) 13 providers diff --git a/open-sse/config/freeModelCatalog.data.ts b/open-sse/config/freeModelCatalog.data.ts index f107d7fef4..a8ba20ae9c 100644 --- a/open-sse/config/freeModelCatalog.data.ts +++ b/open-sse/config/freeModelCatalog.data.ts @@ -106,9 +106,12 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "bytez", modelId: "meta-llama/Llama-3.3-70B-Instruct", displayName: "meta-llama/Llama-3.3-70B-Instruct", monthlyTokens: 0, creditTokens: 1000000, freeType: "recurring-credit", poolKey: "bytez", tos: "ambiguous" }, { provider: "bytez", modelId: "mistralai/Mistral-7B-Instruct-v0.3", displayName: "mistralai/Mistral-7B-Instruct-v0.3", monthlyTokens: 0, creditTokens: 1000000, freeType: "recurring-credit", poolKey: "bytez", tos: "ambiguous" }, { provider: "bytez", modelId: "Qwen/Qwen2.5-72B-Instruct", displayName: "Qwen/Qwen2.5-72B-Instruct", monthlyTokens: 0, creditTokens: 1000000, freeType: "recurring-credit", poolKey: "bytez", tos: "ambiguous" }, - // hardStopGuaranteed: Cerebras pricing page states "Free Trial: 1M tokens/day... no credit card" (open-sse/services/../providers/apikey/inference-hosts.ts:74-84). - { provider: "cerebras", modelId: "zai-glm-4.7", displayName: "GLM 4.7", monthlyTokens: 30000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "cerebras", tos: "caution", hardStopGuaranteed: true }, - { provider: "cerebras", modelId: "gpt-oss-120b", displayName: "GPT OSS 120B", monthlyTokens: 30000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "cerebras", tos: "caution", hardStopGuaranteed: true }, + // #11773: cerebras.ai/pricing (2026-09-03) is a one-time $5 signup credit + // gated on a payment method, 30-day expiry — not the old no-card 1M/day + // trial. creditTokens stays 0 because Cerebras publishes dollars, not a + // token grant. hardStopGuaranteed must stay unset: a stored card can bill. + { provider: "cerebras", modelId: "zai-glm-4.7", displayName: "GLM 4.7", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "cerebras", tos: "caution" }, + { provider: "cerebras", modelId: "gpt-oss-120b", displayName: "GPT OSS 120B", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "cerebras", tos: "caution" }, // #8717: drop dead Workers AI ids (400/403/410). Keep Neurons/day budget on fp8-fast. { provider: "cloudflare-ai", modelId: "@cf/mistral/mistral-7b-instruct-v0.2-lora", displayName: "Mistral 7B (🆓)", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-daily", poolKey: "cloudflare-ai", tos: "caution" }, { provider: "cloudflare-ai", modelId: "@cf/qwen/qwen2.5-coder-32b-instruct", displayName: "Qwen 2.5 Coder 32B (🆓)", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-daily", poolKey: "cloudflare-ai", tos: "caution" }, diff --git a/open-sse/config/freeTierCatalog.ts b/open-sse/config/freeTierCatalog.ts index 339f01a100..68cc240caa 100644 --- a/open-sse/config/freeTierCatalog.ts +++ b/open-sse/config/freeTierCatalog.ts @@ -16,7 +16,6 @@ export const FREE_TIER_BUDGETS: Record = { "cloudflare-ai": 122_000_000, gemini: 60_000_000, doubao: 60_000_000, - cerebras: 30_000_000, "api-airforce": 24_000_000, "ollama-cloud": 20_000_000, groq: 15_000_000, diff --git a/open-sse/services/__tests__/tierResolver.test.ts b/open-sse/services/__tests__/tierResolver.test.ts index fac0b24404..132b6e8561 100644 --- a/open-sse/services/__tests__/tierResolver.test.ts +++ b/open-sse/services/__tests__/tierResolver.test.ts @@ -60,10 +60,10 @@ describe("TierResolver", () => { expect(result.hasFreeTier).toBe(true); }); - it("classifies Cerebras as free", () => { + it("classifies Cerebras as not free after the no-card trial ended (#11773)", () => { const result = classifyTier("cerebras", "llama-3.1-70b"); - expect(result.tier).toBe(PROVIDER_TIER.FREE); - expect(result.hasFreeTier).toBe(true); + expect(result.tier).not.toBe(PROVIDER_TIER.FREE); + expect(result.hasFreeTier).toBe(false); }); it("classifies Groq as free", () => { @@ -228,7 +228,6 @@ describe("TierResolver", () => { "longcat", "cloudflare-ai", "nvidia-nim", - "cerebras", "groq", ]) { expect(LEGACY_FREE_PROVIDERS.includes(id), `expected ${id} in LEGACY_FREE_PROVIDERS`).toBe( diff --git a/open-sse/services/tierConfig.ts b/open-sse/services/tierConfig.ts index 2119029e13..b02234a6b0 100644 --- a/open-sse/services/tierConfig.ts +++ b/open-sse/services/tierConfig.ts @@ -52,7 +52,6 @@ export const LEGACY_FREE_PROVIDERS: readonly string[] = [ "longcat", "cloudflare-ai", "nvidia-nim", - "cerebras", "groq", ]; diff --git a/open-sse/services/tierDefaults.json b/open-sse/services/tierDefaults.json index 5e1e4a23cf..1e74212f3b 100644 --- a/open-sse/services/tierDefaults.json +++ b/open-sse/services/tierDefaults.json @@ -19,7 +19,6 @@ "longcat", "cloudflare-ai", "nvidia-nim", - "cerebras", "groq" ] } diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index e9643f3511..89659dc23c 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -6139,7 +6139,7 @@ "bluesminds": "Get your API key at https://www.bluesminds.com — OpenAI-compatible endpoint at https://api.bluesminds.com/v1 with free daily credits. VIP models (Claude Opus 4.5, Gemini 2.5 Pro) consume pi credits.", "byteplus": "Connect BytePlus ModelArk with an API key.", "bytez": "$1 free credits, refreshes every 4 weeks", - "cerebras": "Free Trial: 1M tokens/day, 30K TPM, 5 RPM — no credit card.", + "cerebras": "One-time $5 signup credit (30-day validity); a payment method is required. Not a recurring free tier.", "charm-hyper": "Create an API key at https://hyper.charm.land, then paste it here as a Bearer token.", "chutes": "Bearer API key for the Chutes OpenAI-compatible gateway.", "clarifai": "Clarifai exposes OpenAI-compatible chat, responses and /models on /v2/ext/openai/v1. Public/community models typically require a PAT; app-scoped keys only work for resources inside that app.", diff --git a/src/i18n/messages/es.json b/src/i18n/messages/es.json index ba61b59daa..09b8296a82 100644 --- a/src/i18n/messages/es.json +++ b/src/i18n/messages/es.json @@ -6136,7 +6136,7 @@ "bluesminds": "Get your API key at https://www.bluesminds.com — OpenAI-compatible endpoint at https://api.bluesminds.com/v1 with free daily credits. VIP models (Claude Opus 4.5, Gemini 2.5 Pro) consume pi credits.", "byteplus": "Connect BytePlus ModelArk with an API key.", "bytez": "$1 free credits, refreshes every 4 weeks", - "cerebras": "Free Trial: 1M tokens/day, 30K TPM, 5 RPM — no credit card.", + "cerebras": "One-time $5 signup credit (30-day validity); a payment method is required. Not a recurring free tier.", "charm-hyper": "Create an API key at https://hyper.charm.land, then paste it here as a Bearer token.", "chutes": "Bearer API key for the Chutes OpenAI-compatible gateway.", "clarifai": "Clarifai exposes OpenAI-compatible chat, responses and /models on /v2/ext/openai/v1. Public/community models typically require a PAT; app-scoped keys only work for resources inside that app.", diff --git a/src/shared/constants/pricing/inference-hosts.ts b/src/shared/constants/pricing/inference-hosts.ts index e09715b942..3551548bdd 100644 --- a/src/shared/constants/pricing/inference-hosts.ts +++ b/src/shared/constants/pricing/inference-hosts.ts @@ -340,26 +340,29 @@ export const DEFAULT_PRICING_INFERENCE = { cache_creation: 0, }, }, + // #11773: Developer-tier $/1M from cerebras.ai/pricing (2026-09-03). + // Signup is a one-time $5 credit, not a $0 token grant — keep paid rates + // so classifyTier cannot treat Cerebras as the free routing tier. cerebras: { - "gpt-oss-120b": { input: 0, output: 0, cached: 0, reasoning: 0, cache_creation: 0 }, - "gemma-4-31b": { input: 0, output: 0, cached: 0, reasoning: 0, cache_creation: 0 }, - "zai-glm-4.7": { input: 0, output: 0, cached: 0, reasoning: 0, cache_creation: 0 }, - "llama-3.3-70b": { input: 0, output: 0, cached: 0, reasoning: 0, cache_creation: 0 }, + "gpt-oss-120b": { input: 0.35, output: 0.75, cached: 0, reasoning: 0, cache_creation: 0 }, + "gemma-4-31b": { input: 0.4, output: 0.8, cached: 0, reasoning: 0, cache_creation: 0 }, + "zai-glm-4.7": { input: 2.25, output: 2.75, cached: 0, reasoning: 0, cache_creation: 0 }, + "llama-3.3-70b": { input: 0.85, output: 1.2, cached: 0, reasoning: 0, cache_creation: 0 }, "llama-4-scout-17b-16e-instruct": { - input: 0, - output: 0, + input: 0.2, + output: 0.2, cached: 0, reasoning: 0, cache_creation: 0, }, "qwen-3-235b-a22b-instruct-2507": { - input: 0, - output: 0, + input: 0.6, + output: 1.2, cached: 0, reasoning: 0, cache_creation: 0, }, - "qwen-3-32b": { input: 0, output: 0, cached: 0, reasoning: 0, cache_creation: 0 }, + "qwen-3-32b": { input: 0.4, output: 0.8, cached: 0, reasoning: 0, cache_creation: 0 }, }, nvidia: { "nvidia/gpt-oss-120b": { input: 0, output: 0, cached: 0, reasoning: 0, cache_creation: 0 }, diff --git a/src/shared/constants/providers/apikey/inference-hosts.ts b/src/shared/constants/providers/apikey/inference-hosts.ts index 84cd65ad41..730ccd64f3 100644 --- a/src/shared/constants/providers/apikey/inference-hosts.ts +++ b/src/shared/constants/providers/apikey/inference-hosts.ts @@ -86,7 +86,12 @@ export const APIKEY_PROVIDERS_INFERENCE = { textIcon: "CB", website: "https://inference.cerebras.ai", hasFree: true, - freeNote: "Free Trial: 1M tokens/day, 30K TPM, 5 RPM — no credit card.", + // #11773: Cerebras retired the no-card 1M tokens/day trial. Live + // cerebras.ai/pricing (2026-09-03) is a one-time $5 signup credit that + // requires a payment method and expires after 30 days — LongCat-shaped + // (hasFree stays true; not a recurring grant). + freeNote: + "One-time $5 signup credit (30-day validity); a payment method is required. Not a recurring free tier.", }, nvidia: { id: "nvidia", diff --git a/tests/unit/cerebras-free-tier-11773.test.ts b/tests/unit/cerebras-free-tier-11773.test.ts new file mode 100644 index 0000000000..1d1f0881b7 --- /dev/null +++ b/tests/unit/cerebras-free-tier-11773.test.ts @@ -0,0 +1,46 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { APIKEY_PROVIDERS } from "../../src/shared/constants/providers/apikey/index.ts"; +import { getProviderById } from "../../src/shared/constants/providers.ts"; +import { FREE_MODEL_BUDGETS } from "../../open-sse/config/freeModelCatalog.ts"; +import { FREE_TIER_BUDGETS } from "../../open-sse/config/freeTierCatalog.ts"; +import { LEGACY_FREE_PROVIDERS } from "../../open-sse/services/tierConfig.ts"; +import { classifyTier, clearTierCache } from "../../open-sse/services/tierResolver.ts"; +import { PROVIDER_TIER } from "../../open-sse/services/tierTypes.ts"; + +const CEREBRAS_MODELS = ["zai-glm-4.7", "gpt-oss-120b"] as const; + +test("#11773 cerebras stays catalogued, but not as a recurring zero-cost tier", () => { + const entry = APIKEY_PROVIDERS.cerebras; + assert.ok(entry, "APIKEY_PROVIDERS.cerebras must remain registered"); + assert.equal(entry.hasFree, true); + assert.equal(Object.hasOwn(FREE_TIER_BUDGETS, "cerebras"), false); + assert.equal(LEGACY_FREE_PROVIDERS.includes("cerebras"), false); +}); + +test("#11773 cerebras freeNote describes the $5 card-gated signup credit", () => { + const note = getProviderById("cerebras")?.freeNote ?? ""; + assert.match(note, /\$5/); + assert.match(note, /30.?day|30 days/i); + assert.match(note, /payment method|credit card/i); + assert.equal(/1M tokens\/day|30K TPM/.test(note), false); +}); + +test("#11773 cerebras catalog rows are one-time signup credits, not a hard-stop free trial", () => { + const rows = FREE_MODEL_BUDGETS.filter((row) => row.provider === "cerebras"); + assert.ok(rows.length >= CEREBRAS_MODELS.length, "catalog must keep the live Cerebras models"); + for (const modelId of CEREBRAS_MODELS) { + const row = rows.find((entry) => entry.modelId === modelId); + assert.ok(row, `missing catalog row for ${modelId}`); + assert.equal(row.freeType, "one-time-initial"); + assert.equal(row.monthlyTokens, 0); + assert.notEqual(row.hardStopGuaranteed, true); + } +}); + +test("#11773 cerebras is not classified as the free routing tier", () => { + clearTierCache(); + const result = classifyTier("cerebras", "zai-glm-4.7"); + assert.notEqual(result.tier, PROVIDER_TIER.FREE); +}); diff --git a/tests/unit/check-docs-counts-sync.test.ts b/tests/unit/check-docs-counts-sync.test.ts index 5f786f327a..2cea8b52db 100644 --- a/tests/unit/check-docs-counts-sync.test.ts +++ b/tests/unit/check-docs-counts-sync.test.ts @@ -485,8 +485,8 @@ const TRAINING_CLAIM = { }; test("the hard-stop claim passes on the real sentence and fails on a stale count", () => { - const v = makeValidator(7, HARD_STOP_CLAIM); - assert.equal(v("7 entries carry an independently documented hard stop, and").ok, true); + const v = makeValidator(5, HARD_STOP_CLAIM); + assert.equal(v("5 entries carry an independently documented hard stop, and").ok, true); assert.equal(v("99 entries carry an independently documented hard stop, and").ok, false); }); @@ -500,10 +500,10 @@ test("the training claim passes on the real sentence and fails on a stale count" test("a reworded or deleted sentence fails, instead of passing as absent", () => { // The gate's real failure mode is not a stale number, it is silence: reword the // sentence past the pattern and "no claim in this file" used to read green. - const required = makeValidator(7, { ...HARD_STOP_CLAIM, requireClaim: true }); - assert.equal(required("7 entries have a provider-documented hard-stop guarantee.").ok, false); + const required = makeValidator(5, { ...HARD_STOP_CLAIM, requireClaim: true }); + assert.equal(required("5 entries have a provider-documented hard-stop guarantee.").ok, false); assert.equal(required("the page no longer mentions it at all").ok, false); - assert.equal(required("7 entries carry an independently documented hard stop.").ok, true); + assert.equal(required("5 entries carry an independently documented hard stop.").ok, true); const trainingRequired = makeValidator(13, { ...TRAINING_CLAIM, requireClaim: true }); assert.equal(trainingRequired("13 entries disclose training use.").ok, false); @@ -514,7 +514,7 @@ test("the live page actually satisfies both required gates", () => { // A unit test on synthetic strings proves the validator; this one proves the // document. Without it, the two could drift apart and both stay green. const page = readFileSync(path.resolve(here, "../../docs/reference/FREE_TIERS.md"), "utf8"); - assert.equal(makeValidator(7, { ...HARD_STOP_CLAIM, requireClaim: true })(page).ok, true); + assert.equal(makeValidator(5, { ...HARD_STOP_CLAIM, requireClaim: true })(page).ok, true); assert.equal(makeValidator(13, { ...TRAINING_CLAIM, requireClaim: true })(page).ok, true); }); diff --git a/tests/unit/free-note-freshness.test.ts b/tests/unit/free-note-freshness.test.ts index 1f24629218..67769aa00b 100644 --- a/tests/unit/free-note-freshness.test.ts +++ b/tests/unit/free-note-freshness.test.ts @@ -14,6 +14,9 @@ test("longcat freeNote reflects the post-2026-05-29 5M tokens/day reality", () = assert.match(note("longcat"), /5M tokens\/day|LongCat-2\.0/i); }); -test("cerebras freeNote reflects the tightened 30K TPM", () => { - assert.match(note("cerebras"), /30K TPM|1M tokens\/day/i); +test("cerebras freeNote reflects the $5 card-gated signup credit (#11773)", () => { + const n = note("cerebras"); + assert.match(n, /\$5/); + assert.match(n, /payment method|credit card/i); + assert.equal(/1M tokens\/day|30K TPM/.test(n), false); }); diff --git a/tests/unit/free-tier-catalog.test.ts b/tests/unit/free-tier-catalog.test.ts index 6f4b23e356..25282d5ef1 100644 --- a/tests/unit/free-tier-catalog.test.ts +++ b/tests/unit/free-tier-catalog.test.ts @@ -7,13 +7,14 @@ import { } from "../../open-sse/config/freeTierCatalog.ts"; test("FREE_TIER_BUDGETS holds positive integer monthly-token budgets", () => { - assert.ok(Object.keys(FREE_TIER_BUDGETS).length >= 19); + assert.ok(Object.keys(FREE_TIER_BUDGETS).length >= 18); for (const [id, tokens] of Object.entries(FREE_TIER_BUDGETS)) { assert.ok(Number.isInteger(tokens) && tokens > 0, `${id} must be a positive integer`); } assert.equal(FREE_TIER_BUDGETS.mistral, 1_000_000_000); assert.equal(FREE_TIER_BUDGETS["cloudflare-ai"], 122_000_000); - assert.equal(FREE_TIER_BUDGETS.cerebras, 30_000_000); + // #11773: Cerebras is a one-time $5 signup credit, not a recurring monthly grant. + assert.equal(FREE_TIER_BUDGETS.cerebras, undefined); // LongCat is excluded from this recurring-monthly catalog: its free tier is a // one-time 10M-token signup grant (not recurring), so it must not appear here. assert.equal(FREE_TIER_BUDGETS.longcat, undefined); @@ -27,9 +28,9 @@ test("FREE_TIER_TOS marks proxy-prohibited providers as avoid", () => { test("computeFreeTierTotals sums the documented budgets", () => { const t = computeFreeTierTotals(); - assert.equal(t.providerCount, 19); - assert.ok(t.documentedMonthlyTokens >= 1_350_000_000); - assert.ok(t.documentedMonthlyTokens <= 1_450_000_000); + assert.equal(t.providerCount, 18); + assert.ok(t.documentedMonthlyTokens >= 1_320_000_000); + assert.ok(t.documentedMonthlyTokens <= 1_420_000_000); assert.equal(typeof t.headline, "string"); assert.match(t.headline, /1\.3/); }); @@ -38,5 +39,5 @@ test("computeFreeTierTotals can exclude ToS-avoid providers", () => { const all = computeFreeTierTotals(); const clean = computeFreeTierTotals({ excludeTosAvoid: true }); assert.equal(all.documentedMonthlyTokens - clean.documentedMonthlyTokens, 25_000); - assert.equal(clean.providerCount, 18); + assert.equal(clean.providerCount, 17); }); From 2265ce761f76b6c81e34515cc8483ca127d4b414 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 21:31:13 -0300 Subject: [PATCH 073/143] fix(security): harden public error boundaries (#12506) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado sobre o tip de `release/v3.8.51` depois de reconciliar com o #12620, que entrou primeiro nesta mesma sessão e ataca a mesma classe de problema por outra arquitetura. **A colisão e como foi resolvida.** O #12620 consertou o GHSA-qv45-56jc-4wmj adicionando `RAW_CREDENTIAL_PATTERNS` a `error.ts` e importando-os em `upstreamErrorPassthrough.ts`. Este PR resolve o mesmo problema quebrando `error.ts` em `errorSanitization.ts` + `errorPathRedaction.ts`. Mantive a divisão em módulos deste PR, porque ao comparar os dois vocabulários o dele já era mais amplo: o `STRONG_CREDENTIAL_TOKEN` daqui cobre `sk-`/`sk_` **com lookbehind e uma variante para a forma embutida** (que pega `sk-proj-…`), mais Slack `xox-`, AWS `AKIA`/`ASIA`, `github_pat_`/`ghp_`/`glpat-` e JWT de três segmentos. A única forma que o #12620 carregava e este conjunto não tinha era a chave do Google (`AIza…`) — adicionada aqui, com o mesmo quantificador limitado que os irmãos usam (AGENTS.md → PII §1, já que isso roda sobre corpos upstream não confiáveis). **A verificação não foi por inspeção.** Rodei as suítes do próprio #12620 contra esta estrutura: **48/48** em `error-sanitizer-sk-key-qv45`, `bifrost-relay-response-leak-9m72`, `search-baseurl-client-override-3f8g` e `search-baseurl-ssrf-guard` — incluindo a asserção anti-drift daquela suíte, que é o oráculo certo aqui: *para todo corpo que a camada de passthrough recusa como vazante, o sanitizador de fallback não pode devolvê-lo intacto*. Ela passa, então a propriedade de segurança dos três GHSAs sobrevive à troca de arquitetura. Os 21 arquivos de teste deste PR: **259/259**. `typecheck:core` limpo. --- .../0000-public-error-boundary-hardening.md | 1 + docs/security/ERROR_SANITIZATION.md | 81 +- open-sse/executors/claude-web.ts | 7 +- open-sse/executors/claude-web/stream.ts | 7 +- open-sse/executors/ninerouter.ts | 3 +- open-sse/handlers/chatCore.ts | 79 +- open-sse/handlers/chatCore/failureUsage.ts | 15 + .../handlers/chatCore/streamErrorResult.ts | 8 +- .../handlers/chatCore/translationFailure.ts | 24 + open-sse/handlers/moderations.ts | 20 +- open-sse/handlers/ocr.ts | 18 +- open-sse/mcp-server/errorMessage.ts | 13 + open-sse/mcp-server/server.ts | 55 +- .../translator/response/openai-responses.ts | 6 +- open-sse/utils/credentialPatterns.ts | 79 ++ open-sse/utils/error.ts | 516 +++++++--- open-sse/utils/errorPathRedaction.ts | 905 ++++++++++++++++++ open-sse/utils/errorSanitization.ts | 895 +++++++++++++++++ open-sse/utils/passthroughTailProcessor.ts | 18 +- open-sse/utils/responsesFailureOutput.ts | 70 ++ open-sse/utils/stream.ts | 183 ++-- open-sse/utils/streamErrorFormat.ts | 219 ++++- open-sse/utils/streamFailureBoundary.ts | 76 ++ open-sse/utils/streamFailureFinalization.ts | 13 +- open-sse/utils/upstreamErrorPassthrough.ts | 76 +- open-sse/utils/upstreamErrorResponse.ts | 46 + src/app/api/logs/[id]/route.ts | 35 +- .../[id]/models/staleEncryptionGuard.ts | 6 +- .../[id]/test/publicErrorBoundary.ts | 155 +++ src/app/api/providers/[id]/test/route.ts | 168 +--- src/app/api/providers/validate/route.ts | 18 +- src/lib/guardrails/credentialMasker.ts | 88 +- src/lib/logPayloads.ts | 313 +++++- src/lib/providers/validation/transport.ts | 52 +- src/lib/proxyLogger.ts | 45 +- src/lib/skills/executor.ts | 179 +++- src/lib/skills/interception.ts | 33 +- src/lib/usage/callLogs.ts | 30 +- src/lib/usage/callLogs/format.ts | 60 +- src/lib/usage/usageHistory.ts | 3 +- src/lib/usage/usageStats.ts | 8 +- src/shared/utils/apiKeyPolicy.ts | 3 +- src/shared/utils/terminalStatus.ts | 26 +- src/sse/services/auth.ts | 10 +- tests/unit/calllogs-format-split.test.ts | 10 +- .../unit/chatcore-stream-error-result.test.ts | 17 + tests/unit/chatcore-translation-paths.test.ts | 62 +- tests/unit/combo-diagnostics-trace.test.ts | 47 +- tests/unit/error-message-sanitization.test.ts | 21 +- .../error-public-boundaries-hardening.test.ts | 11 + tests/unit/error-sensitive-redaction.test.ts | 128 ++- ...ror-public-boundaries-hardening.fixture.ts | 608 ++++++++++++ .../mcp-public-error-boundaries.fixture.ts | 184 ++++ ...onnection-test-error-boundaries.fixture.ts | 262 +++++ ...rovider-last-error-sanitization.fixture.ts | 109 +++ ...request-log-management-boundary.fixture.ts | 108 +++ ...ilure-persistent-classification.fixture.ts | 91 ++ .../gemini-responses-error-redaction.test.ts | 39 + .../helpers/runIsolatedBoundaryFixture.ts | 73 ++ .../unit/mcp-public-error-boundaries.test.ts | 11 + tests/unit/moderations-handler.test.ts | 75 +- tests/unit/ocr-handler-dispatch.test.ts | 80 ++ ...r-connection-test-error-boundaries.test.ts | 14 + .../provider-last-error-sanitization.test.ts | 11 + ...ider-validation-error-sanitization.test.ts | 101 ++ .../request-log-management-boundary.test.ts | 11 + tests/unit/request-log-payloads.test.ts | 421 ++++++++ tests/unit/skills-executor.test.ts | 143 ++- tests/unit/skills-interception.test.ts | 101 +- ...-failure-persistent-classification.test.ts | 14 + ...stream-passthrough-error-redaction.test.ts | 446 +++++++++ tests/unit/upstream-error-passthrough.test.ts | 96 +- 72 files changed, 7173 insertions(+), 786 deletions(-) create mode 100644 changelog.d/fixes/0000-public-error-boundary-hardening.md create mode 100644 open-sse/handlers/chatCore/translationFailure.ts create mode 100644 open-sse/mcp-server/errorMessage.ts create mode 100644 open-sse/utils/credentialPatterns.ts create mode 100644 open-sse/utils/errorPathRedaction.ts create mode 100644 open-sse/utils/errorSanitization.ts create mode 100644 open-sse/utils/responsesFailureOutput.ts create mode 100644 open-sse/utils/streamFailureBoundary.ts create mode 100644 open-sse/utils/upstreamErrorResponse.ts create mode 100644 src/app/api/providers/[id]/test/publicErrorBoundary.ts create mode 100644 tests/unit/error-public-boundaries-hardening.test.ts create mode 100644 tests/unit/fixtures/error-public-boundaries-hardening.fixture.ts create mode 100644 tests/unit/fixtures/mcp-public-error-boundaries.fixture.ts create mode 100644 tests/unit/fixtures/provider-connection-test-error-boundaries.fixture.ts create mode 100644 tests/unit/fixtures/provider-last-error-sanitization.fixture.ts create mode 100644 tests/unit/fixtures/request-log-management-boundary.fixture.ts create mode 100644 tests/unit/fixtures/stream-failure-persistent-classification.fixture.ts create mode 100644 tests/unit/gemini-responses-error-redaction.test.ts create mode 100644 tests/unit/helpers/runIsolatedBoundaryFixture.ts create mode 100644 tests/unit/mcp-public-error-boundaries.test.ts create mode 100644 tests/unit/provider-connection-test-error-boundaries.test.ts create mode 100644 tests/unit/provider-last-error-sanitization.test.ts create mode 100644 tests/unit/provider-validation-error-sanitization.test.ts create mode 100644 tests/unit/request-log-management-boundary.test.ts create mode 100644 tests/unit/stream-failure-persistent-classification.test.ts create mode 100644 tests/unit/stream-passthrough-error-redaction.test.ts diff --git a/changelog.d/fixes/0000-public-error-boundary-hardening.md b/changelog.d/fixes/0000-public-error-boundary-hardening.md new file mode 100644 index 0000000000..c0531eba1d --- /dev/null +++ b/changelog.d/fixes/0000-public-error-boundary-hardening.md @@ -0,0 +1 @@ +- **fix(security):** Sanitize provider and runtime failures before public API, SSE and MCP responses and before persistent request, proxy and usage logs, preventing credentials, stack traces and host filesystem paths from crossing those boundaries while preserving stable error codes and useful diagnostics. diff --git a/docs/security/ERROR_SANITIZATION.md b/docs/security/ERROR_SANITIZATION.md index 898ca209d9..e3ffe04c77 100644 --- a/docs/security/ERROR_SANITIZATION.md +++ b/docs/security/ERROR_SANITIZATION.md @@ -1,14 +1,16 @@ --- title: "Error Message Sanitization" -version: 3.8.40 -lastUpdated: 2026-06-28 +version: 3.8.51 +lastUpdated: 2026-09-02 --- # Error Message Sanitization -> **Source of truth:** `open-sse/utils/error.ts` — `sanitizeErrorMessage`, `buildErrorBody`, `createErrorResult` -> **Tests:** `tests/unit/error-message-sanitization.test.ts` -> **Last updated:** 2026-06-28 — v3.8.40 +> **Source of truth:** `open-sse/utils/errorSanitization.ts`, +> `open-sse/utils/errorPathRedaction.ts`, and the public builders in `open-sse/utils/error.ts` +> **Tests:** `tests/unit/error-message-sanitization.test.ts`, +> `tests/unit/error-public-boundaries-hardening.test.ts` +> **Last updated:** 2026-09-02 — v3.8.51 > **Audience:** Any engineer touching error responses (HTTP routes, SSE streams, executors, MCP handlers). > **Status:** **MANDATORY** for every code path that returns an error message to a client. @@ -20,10 +22,18 @@ CodeQL rule `js/stack-trace-exposure` (CWE-209) flags any code path where an err - Library / framework versions inferred from stack frames → targeted exploit selection. - Sensitive runtime values that may be string-interpolated into errors (DB queries, config values). -The `sanitizeErrorMessage` helper in `open-sse/utils/error.ts` strips both classes of leakage: +The `sanitizeErrorMessage` helper exported by `open-sse/utils/error.ts` strips these classes of +leakage: -1. Multi-line stack traces — only the first line (the actual error message) is kept. -2. Absolute paths (`/...*.{ts,js,tsx,jsx,mjs,cjs}[:line[:col]]` and `C:\...`) — replaced with ``. +1. Physical, serialized, and unambiguously inline JavaScript stack-frame tails. +2. Absolute POSIX, Windows, UNC, and `file://` filesystem paths, while preserving safe HTTPS URLs + and explicitly marked API routes. +3. Credential assignments, common provider token formats, private-key PEM blocks, and base64 data + URLs. + +The sanitizer caps input length and fails closed when a thrown value rejects string coercion. +Recursive upstream JSON sanitization also drops unsafe credential/path keys, session aliases, and +prototype-control keys before a response is serialized. ## The mandatory pattern @@ -59,7 +69,10 @@ import { } from "@omniroute/open-sse/utils/error.ts"; ``` -All of these route through `buildErrorBody` and therefore through `sanitizeErrorMessage`. **You never need to call `sanitizeErrorMessage` manually** when using these helpers. +All of these apply the canonical public-error boundary. `errorResponse`, `writeStreamError`, and +`createErrorResult` route through `buildErrorBody`; the three specialized retry/circuit helpers +project and sanitize their public context directly. **You never need to call +`sanitizeErrorMessage` manually** when using these helpers. ### 2. Custom error envelopes (rare) @@ -81,17 +94,25 @@ This is the only sanctioned way to assemble a custom error body. See `open-sse/e ### 3. Logging vs. responding -`sanitizeErrorMessage` should **only** wrap the value that crosses the network boundary. Internal logs (`pino`, `console`) should keep the full message, including stack, so operators can debug. Pattern: +Trusted internal exceptions may keep their full message and stack so operators can debug. Values +originating at provider, validation, browser-session, or credential-adjacent boundaries must be +sanitized before they enter console output, audit metadata, or persistent call logs. Pattern: ```ts try { // ... } catch (err) { - log.error({ err }, "handler failed"); // full err with stack — internal log + log.error({ err }, "handler failed"); // trusted internal exception only return errorResponse(500, getErrorMessage(err)); // sanitized — sent to client } ``` +For provider-controlled failures, project the logged value too: + +```ts +log.error({ message: sanitizeErrorMessage(err) || "Provider request failed" }); +``` + ### 4. Forbidden patterns ❌ **Never** put raw exception output in a Response body: @@ -112,7 +133,9 @@ const safe = String(err).split("\n")[0]; ❌ **Never** sanitize in the route and forget the SSE path. Anything that writes to a stream goes through `writeStreamError` (or its underlying `buildErrorBody`). -❌ **Never** include `process.cwd()`, `__filename`, `__dirname`, env-derived paths in error messages — they bypass the path regex and reveal the deployment topology. +❌ **Never** intentionally include `process.cwd()`, `__filename`, `__dirname`, or env-derived paths +in error messages. The sanitizer covers absolute paths as defense in depth, but callers must not +construct topology-bearing messages in the first place. ## Coverage in CI @@ -129,7 +152,9 @@ When adding a new route or executor, copy the assertion pattern from this file. ## Related controls - `js/stack-trace-exposure` CodeQL alerts in `.github/security` should always be **either** fixed via these helpers **or** dismissed with a comment citing this doc. -- The `pino` redaction config (`src/shared/utils/logRedaction.ts`) handles structured log redaction separately. This doc covers only the response-message surface. +- The `pino` redaction config (`src/shared/utils/logRedaction.ts`) handles trusted structured logs + separately. This document covers public response messages and provider-controlled values that + cross persistent call/proxy-log boundaries. - Upstream-header denylist (`src/shared/constants/upstreamHeaders.ts`) covers header leakage — keep both files aligned when adding a new exfiltration concern. ## Upstream details passthrough @@ -138,27 +163,39 @@ When adding a new route or executor, copy the assertion pattern from this file. parsed body from the upstream provider). When provided, it is sanitized by `sanitizeUpstreamDetails` before inclusion in the response as `upstream_details`. -An optional fourth argument `classification` (`{ type?: string; code?: string }`) -preserves an explicit error type/code instead of re-deriving both from the -status-code table — used when the caller already classified the failure (e.g. -HTTP 499 → `client_disconnected`). +An optional fourth argument `classification` +(`{ type?: string; code?: string; reason?: string }`) accepts an explicit public classification. +Every field is projected onto the bounded public-identifier vocabulary. Unsafe, credential-shaped, +control-character, or overlong values fall back to the status-derived type/code; an unsafe optional +reason is omitted. Three-digit HTTP status identifiers (`100` through `599`) remain valid for +provider contracts that expose the numeric upstream status as a machine-readable code. The same +bounded range is accepted in the locally generated HTTP-status placeholder form; arbitrary provider +numbers and names remain outside the vocabulary. + +Pass every explicit classification in that fourth argument. Never overwrite +`body.error.code`, `body.error.type`, or `body.error.reason` after `buildErrorBody()` returns; +post-builder mutation bypasses the public projection. Sanitization rules applied to `upstreamDetails`: 1. String leaves: run through `sanitizeErrorMessage` (strips stacks + absolute paths). -2. Key blocklist: keys matching `/stack|trace|path|file|cwd|dir|password|secret|token|key/i` - are removed. +2. Unsafe path, credential, session-alias, and prototype-control keys are removed. 3. Depth cap: nesting beyond 4 levels is replaced with the string `"[truncated]"`. 4. Arrays are capped at 32 elements. -Only the seven upstream-error `createErrorResult` call sites in `chatCore.ts` pass -`upstreamErrorBody`. Internal OmniRoute errors (SSE parse failures, empty content, -guardrail blocks) do not include `upstream_details`. +Only call sites with a parsed provider error body should pass `upstreamDetails`. Internal OmniRoute +errors (SSE parse failures, empty content, guardrail blocks) must not include it. Do NOT pass raw `err.stack`, `err.message`, or any string from a runtime exception to `upstreamDetails`. Those must still go through `errorResponse` / `buildErrorBody(code, msg)` without an upstream body. +Selective upstream 4xx passthrough preserves the provider's safe JSON shape and wording required by +client auto-recovery, but it is not byte-for-byte passthrough: the recursive sanitizer always runs +before serialization. Cyclic, BigInt-bearing, or hostile `toJSON()` bodies fail closed and are not +eligible for passthrough. OCR and moderation apply the same rule; non-JSON, blank, or mislabeled +upstream bodies are converted to the canonical OmniRoute JSON error envelope. + ## Known CodeQL limitation: custom sanitizers not recognized The CodeQL query [`js/stack-trace-exposure`](https://codeql.github.com/codeql-query-help/javascript/js-stack-trace-exposure/) uses a fixed allowlist of sanitizer patterns (e.g. inline `.split("\n")[0]`, `String#replace` with specific regex shapes, access to `.message` on `Error`). It does **not** recognize indirection through a custom helper like our `sanitizeErrorMessage()`. diff --git a/open-sse/executors/claude-web.ts b/open-sse/executors/claude-web.ts index 5034b0da6c..77c06b5b14 100644 --- a/open-sse/executors/claude-web.ts +++ b/open-sse/executors/claude-web.ts @@ -216,9 +216,10 @@ function makeErrorResponse( extraHeaders?: Record; } ): Response { - const body = buildErrorBody(status, message, options?.details); - if (options?.type) body.error.type = options.type; - if (options?.code) body.error.code = options.code; + const body = buildErrorBody(status, message, options?.details, { + type: options?.type, + code: options?.code, + }); const headers: Record = { "Content-Type": "application/json" }; if (options?.extraHeaders) { for (const [key, value] of Object.entries(options.extraHeaders)) { diff --git a/open-sse/executors/claude-web/stream.ts b/open-sse/executors/claude-web/stream.ts index 264f8218e8..e3b6c724e9 100644 --- a/open-sse/executors/claude-web/stream.ts +++ b/open-sse/executors/claude-web/stream.ts @@ -453,9 +453,10 @@ function makeChunk( } function protocolErrorBody(): Record { - const body = buildErrorBody(502, "Claude Web stream protocol error"); - body.error.type = "upstream_protocol_error"; - body.error.code = "claude_web_protocol_error"; + const body = buildErrorBody(502, "Claude Web stream protocol error", undefined, { + type: "upstream_protocol_error", + code: "claude_web_protocol_error", + }); return body as unknown as Record; } diff --git a/open-sse/executors/ninerouter.ts b/open-sse/executors/ninerouter.ts index 0c6bfd8bc6..74fa30a48e 100644 --- a/open-sse/executors/ninerouter.ts +++ b/open-sse/executors/ninerouter.ts @@ -72,8 +72,7 @@ export class NineRouterExecutor extends BaseExecutor { * Message goes through buildErrorBody to satisfy hard rule #12 (no raw err.message). */ private buildServiceUnavailableResponse(message: string): Response { - const body = buildErrorBody(503, message); - body.error.code = "service_not_running"; + const body = buildErrorBody(503, message, undefined, { code: "service_not_running" }); return new Response(JSON.stringify(body), { status: 503, headers: { diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 65a986d186..622e934084 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -5,7 +5,8 @@ import { import { injectMemoryAndSkills } from "./chatCore/memorySkillsInjection.ts"; import { resolveChatCoreRequestSetup } from "./chatCore/requestSetup.ts"; import { normalizeOpenAICompatibleTools } from "./chatCore/openAICompatibleTools.ts"; -import { buildFailureUsageRecord } from "./chatCore/failureUsage.ts"; +import { buildFailureUsageRecord, projectFailureUsageErrorCode } from "./chatCore/failureUsage.ts"; +import { createTranslationFailureResult } from "./chatCore/translationFailure.ts"; import { estimateFinalInputTokens } from "./chatCore/contextEstimation.ts"; import { extractSystemRoleMessages, @@ -2513,35 +2514,11 @@ export async function handleChatCore({ : HTTP_STATUS.SERVER_ERROR; const message = error?.message || "Invalid request"; const errorType = typeof error?.errorType === "string" ? error.errorType : null; - - log?.warn?.("TRANSLATE", `Request translation failed: ${message}`); - - if (errorType) { - trackPendingRequest(model, provider, connectionId, false); - return { - success: false, - status: statusCode, - error: message, - response: new Response( - JSON.stringify({ - error: { - message, - type: errorType, - code: errorType, - }, - }), - { - status: statusCode, - headers: { - "Content-Type": "application/json", - }, - } - ), - }; - } + const result = createTranslationFailureResult(statusCode, message, errorType); + log?.warn?.("TRANSLATE", `Request translation failed: ${result.error}`); trackPendingRequest(model, provider, connectionId, false); - return createErrorResult(statusCode, message); + return result; } // The latest OmniGlyph release has protocol-native OpenAI transforms. Run @@ -3924,10 +3901,14 @@ export async function handleChatCore({ streamController.handleError(error); return createErrorResult(499, "Request aborted"); } - persistFailureUsage( - failureStatus, - upstreamErrorCode || (error instanceof Error && error.name ? error.name : "upstream_error") - ); + const persistentErrorCode = projectFailureUsageErrorCode({ + statusCode: failureStatus, + message: failureMessage, + errorCode: + upstreamErrorCode || (error instanceof Error && error.name ? error.name : "upstream_error"), + errorType: upstreamErrorType, + }); + persistFailureUsage(failureStatus, persistentErrorCode); console.log(`${COLORS.red}[ERROR] ${failureMessage}${COLORS.reset}`); if (stream && upstreamErrorCode) { const result = createStreamingErrorResult( @@ -4253,6 +4234,9 @@ export async function handleChatCore({ `${decision.kind} (model remaining: ${decision.snapshot.modelRemaining ?? "unknown"}, total remaining: ${decision.snapshot.totalRemaining ?? "unknown"})` ); } + // Classifiers and recovery paths above consume the raw provider wording. + // Project a separate value only at persistent connection-state boundaries. + const persistentMessage = sanitizeErrorMessage(message) || "Provider request failed"; const errorConnectionId = getCurrentConnectionId(); if (errorConnectionId && errorType) { try { @@ -4264,7 +4248,7 @@ export async function handleChatCore({ { testStatus: "banned", isActive: false, - lastError: message, + lastError: persistentMessage, lastErrorType: errorType, errorCode: String(statusCode), }, @@ -4295,7 +4279,7 @@ export async function handleChatCore({ ) { await updateProviderConnection(errorConnectionId, { lastErrorType: errorType, - lastError: message, + lastError: persistentMessage, errorCode: statusCode, }); console.warn( @@ -4308,7 +4292,7 @@ export async function handleChatCore({ { testStatus: "deactivated", isActive: false, - lastError: message, + lastError: persistentMessage, lastErrorType: errorType, errorCode: String(statusCode), }, @@ -4332,7 +4316,7 @@ export async function handleChatCore({ errorConnectionId, { testStatus: "credits_exhausted", - lastError: message, + lastError: persistentMessage, lastErrorType: errorType, errorCode: String(statusCode), }, @@ -4418,7 +4402,7 @@ export async function handleChatCore({ rateLimitedUntil: kimiRateLimitResetAt, backoffLevel: 0, lastErrorType: PROVIDER_ERROR_TYPES.RATE_LIMITED, - lastError: message, + lastError: persistentMessage, errorCode: statusCode, }); console.warn( @@ -4447,7 +4431,7 @@ export async function handleChatCore({ errorConnectionId, { testStatus: "credits_exhausted", - lastError: message, + lastError: persistentMessage, lastErrorType: errorType, errorCode: String(statusCode), }, @@ -4463,14 +4447,14 @@ export async function handleChatCore({ // Normal 401 (token/session auth issue): keep account active for refresh/re-auth. await updateProviderConnection(errorConnectionId, { lastErrorType: errorType, - lastError: message, + lastError: persistentMessage, errorCode: statusCode, }); } else if (errorType === PROVIDER_ERROR_TYPES.OAUTH_INVALID_TOKEN) { // OAuth 401 with invalid credentials - token refresh can recover await updateProviderConnection(errorConnectionId, { lastErrorType: errorType, - lastError: message, + lastError: persistentMessage, errorCode: statusCode, }); console.warn( @@ -4480,7 +4464,7 @@ export async function handleChatCore({ // Cloud Code 403 with stale project: not a ban, keep account active. await updateProviderConnection(errorConnectionId, { lastErrorType: errorType, - lastError: message, + lastError: persistentMessage, errorCode: statusCode, }); console.warn( @@ -4496,7 +4480,7 @@ export async function handleChatCore({ const geoCooldownMs = COOLDOWN_MS.geoBlocked ?? 24 * 60 * 60 * 1000; await updateProviderConnection(errorConnectionId, { lastErrorType: errorType, - lastError: message, + lastError: persistentMessage, errorCode: statusCode, }); // T-PROBE: the 24h exclusion is a routing mutation — a probe must @@ -4521,7 +4505,7 @@ export async function handleChatCore({ const byopCooldownMs = COOLDOWN_MS.gcpProjectRequired ?? 24 * 60 * 60 * 1000; await updateProviderConnection(errorConnectionId, { lastErrorType: errorType, - lastError: message, + lastError: persistentMessage, errorCode: statusCode, }); try { @@ -5305,9 +5289,12 @@ export async function handleChatCore({ }).catch(() => {}); const malformed = describeMalformedNonStream(translatedResponse, malformedTranslatedReason); const malformedMessage = `[${provider}/${model}] ${malformed.message}`; - const malformedClientBody = buildErrorBody(HTTP_STATUS.BAD_GATEWAY, malformedMessage); - malformedClientBody.error.code = malformed.code; - malformedClientBody.error.type = malformed.type; + const malformedClientBody = buildErrorBody( + HTTP_STATUS.BAD_GATEWAY, + malformedMessage, + undefined, + { code: malformed.code, type: malformed.type } + ); persistAttemptLogs({ status: HTTP_STATUS.BAD_GATEWAY, tokens: usage, diff --git a/open-sse/handlers/chatCore/failureUsage.ts b/open-sse/handlers/chatCore/failureUsage.ts index 52f70fbfee..d9fff0ae33 100644 --- a/open-sse/handlers/chatCore/failureUsage.ts +++ b/open-sse/handlers/chatCore/failureUsage.ts @@ -8,6 +8,21 @@ * `latencyMs` (Date.now() - startTime) and fires the fire-and-forget saveRequestUsage(...).catch(). */ +import { buildErrorBody } from "../../utils/error.ts"; + +export function projectFailureUsageErrorCode(opts: { + statusCode: number; + message: string; + errorCode?: string | null; + errorType?: string | null; +}): string { + const errorBody = buildErrorBody(opts.statusCode, opts.message, undefined, { + code: opts.errorCode || undefined, + type: opts.errorType || undefined, + }); + return errorBody.error.code || String(opts.statusCode); +} + export function buildFailureUsageRecord(opts: { provider: string | null | undefined; model: string | null | undefined; diff --git a/open-sse/handlers/chatCore/streamErrorResult.ts b/open-sse/handlers/chatCore/streamErrorResult.ts index 77244b611d..04e041e55c 100644 --- a/open-sse/handlers/chatCore/streamErrorResult.ts +++ b/open-sse/handlers/chatCore/streamErrorResult.ts @@ -25,13 +25,7 @@ export function createStreamingErrorResult( code?: string, type?: string ) { - const errorBody = buildErrorBody(statusCode, message); - if (code) { - errorBody.error.code = code; - } - if (type) { - errorBody.error.type = type; - } + const errorBody = buildErrorBody(statusCode, message, undefined, { code, type }); const body = `data: ${JSON.stringify(errorBody)}\n\ndata: [DONE]\n\n`; diff --git a/open-sse/handlers/chatCore/translationFailure.ts b/open-sse/handlers/chatCore/translationFailure.ts new file mode 100644 index 0000000000..622b5c8c47 --- /dev/null +++ b/open-sse/handlers/chatCore/translationFailure.ts @@ -0,0 +1,24 @@ +import { buildErrorBody, createErrorResult } from "../../utils/error.ts"; + +export function createTranslationFailureResult( + status: number, + message: string, + errorType: string | null +) { + if (!errorType) return createErrorResult(status, message); + const body = buildErrorBody( + status, + message, + undefined, + { type: errorType, code: errorType } + ); + return { + success: false as const, + status, + error: body.error.message, + response: new Response(JSON.stringify(body), { + status, + headers: { "Content-Type": "application/json" }, + }), + }; +} diff --git a/open-sse/handlers/moderations.ts b/open-sse/handlers/moderations.ts index c153e10eea..fa5cb72a16 100644 --- a/open-sse/handlers/moderations.ts +++ b/open-sse/handlers/moderations.ts @@ -6,7 +6,8 @@ import { CORS_HEADERS } from "../utils/cors.ts"; */ import { getModerationProvider, parseModerationModel } from "../config/moderationRegistry.ts"; -import { errorResponse, redactSensitiveErrorText } from "../utils/error.ts"; +import { errorResponse, sanitizeErrorMessage } from "../utils/error.ts"; +import { buildSanitizedUpstreamErrorResponse } from "../utils/upstreamErrorResponse.ts"; import { attachOmniRouteMetaHeaders } from "@/domain/omnirouteResponseMeta"; import { generateRequestId } from "@/shared/utils/requestId"; @@ -57,14 +58,11 @@ export async function handleModeration({ body, credentials }) { if (!res.ok) { const errText = await res.text(); - // secret-leak hardening: redact any credential the upstream echoed back - // before relaying the error body to the client (structure-preserving). - return new Response(redactSensitiveErrorText(errText), { + return buildSanitizedUpstreamErrorResponse({ status: res.status, - headers: { - "Content-Type": "application/json", - ...CORS_HEADERS, - }, + rawBody: errText, + fallbackMessage: `Moderation provider returned HTTP ${res.status}`, + headers: CORS_HEADERS, }); } @@ -79,6 +77,10 @@ export async function handleModeration({ body, credentials }) { }); return new Response(JSON.stringify(data), { status: 200, headers }); } catch (err) { - return errorResponse(500, `Moderation request failed: ${err.message}`); + const safeDetail = + sanitizeErrorMessage(err) + .replace(/^[A-Za-z]*Error:\s*/, "") + .trim() || "unknown upstream failure"; + return errorResponse(500, `Moderation request failed: ${safeDetail}`); } } diff --git a/open-sse/handlers/ocr.ts b/open-sse/handlers/ocr.ts index f5d52f0106..16538e08cc 100644 --- a/open-sse/handlers/ocr.ts +++ b/open-sse/handlers/ocr.ts @@ -11,7 +11,8 @@ import { parseOcrModel, OCR_PROVIDERS, } from "../config/ocrRegistry.ts"; -import { errorResponse, redactSensitiveErrorText } from "../utils/error.ts"; +import { errorResponse, sanitizeErrorMessage } from "../utils/error.ts"; +import { buildSanitizedUpstreamErrorResponse } from "../utils/upstreamErrorResponse.ts"; import { attachOmniRouteMetaHeaders } from "@/domain/omnirouteResponseMeta"; import { generateRequestId } from "@/shared/utils/requestId"; import { @@ -151,15 +152,11 @@ export async function handleOcr({ if (!res.ok) { const errText = await res.text(); - // secret-leak hardening: an upstream OCR provider can echo the offending - // request (Authorization header / api key) inside its error text. Redact - // secret patterns (structure-preserving) before relaying to the client. - return new Response(redactSensitiveErrorText(errText), { + return buildSanitizedUpstreamErrorResponse({ status: res.status, - headers: { - "Content-Type": "application/json", - ...CORS_HEADERS, - }, + rawBody: errText, + fallbackMessage: `OCR provider returned HTTP ${res.status}`, + headers: CORS_HEADERS, }); } @@ -184,7 +181,8 @@ export async function handleOcr({ }); return new Response(JSON.stringify(parsed), { status: 200, headers }); } catch (err) { - console.error("[OCR]", err); + const safeErrorMessage = sanitizeErrorMessage(err).trim() || "OCR request failed"; + console.error("[OCR]", safeErrorMessage); return errorResponse(500, "OCR request failed"); } } diff --git a/open-sse/mcp-server/errorMessage.ts b/open-sse/mcp-server/errorMessage.ts new file mode 100644 index 0000000000..f90c570166 --- /dev/null +++ b/open-sse/mcp-server/errorMessage.ts @@ -0,0 +1,13 @@ +import { sanitizeErrorMessage } from "../utils/error.ts"; + +export function toSafeMcpErrorMessage( + value: unknown, + fallback = "MCP tool execution failed" +): string { + try { + const raw = value instanceof Error ? value.message : value; + return sanitizeErrorMessage(raw) || fallback; + } catch { + return fallback; + } +} diff --git a/open-sse/mcp-server/server.ts b/open-sse/mcp-server/server.ts index 73e3a387dc..4c30e608f9 100644 --- a/open-sse/mcp-server/server.ts +++ b/open-sse/mcp-server/server.ts @@ -93,7 +93,7 @@ import { import { getDbInstance, ensureDbInitialized } from "../../src/lib/db/core.ts"; import { normalizeQuotaResponse } from "../../src/shared/contracts/quota.ts"; import { resolveOmniRouteBaseUrl } from "../../src/shared/utils/resolveOmniRouteBaseUrl.ts"; -import { sanitizeErrorMessage } from "../utils/error.ts"; +import { toSafeMcpErrorMessage } from "./errorMessage.ts"; import { mcpFetchTimeoutSignal } from "./fetchTimeout.ts"; import { getMcpModelsCatalog } from "./catalog.ts"; import { registerRadarCatalogTool } from "./radarCatalog.ts"; @@ -328,9 +328,7 @@ async function handleGetHealth() { .filter(({ settled }) => settled.status === "rejected") .map(({ source, settled }) => ({ source, - error: sanitizeErrorMessage( - settled.status === "rejected" ? (settled as PromiseRejectedResult).reason : undefined - ), + error: toSafeMcpErrorMessage((settled as PromiseRejectedResult).reason, ""), })); const result = { @@ -378,7 +376,7 @@ async function handleGetHealth() { await logToolCall("omniroute_get_health", {}, result, Date.now() - start, true); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err); await logToolCall("omniroute_get_health", {}, null, Date.now() - start, false, msg); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } @@ -420,7 +418,7 @@ async function handleListCombos(args: { includeMetrics?: boolean }) { await logToolCall("omniroute_list_combos", args, result, Date.now() - start, true); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err); await logToolCall("omniroute_list_combos", args, null, Date.now() - start, false, msg); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } @@ -435,7 +433,7 @@ async function handleGetComboMetrics(args: { comboId: string }) { await logToolCall("omniroute_get_combo_metrics", args, result, Date.now() - start, true); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err); await logToolCall("omniroute_get_combo_metrics", args, null, Date.now() - start, false, msg); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } @@ -451,7 +449,7 @@ async function handleSwitchCombo(args: { comboId: string; active: boolean }) { await logToolCall("omniroute_switch_combo", args, result, Date.now() - start, true); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err); await logToolCall("omniroute_switch_combo", args, null, Date.now() - start, false, msg); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } @@ -472,7 +470,7 @@ async function handleCreateCombo(args: { await logToolCall("omniroute_create_combo", args, result, Date.now() - start, true); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err); await logToolCall("omniroute_create_combo", args, null, Date.now() - start, false, msg); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } @@ -493,7 +491,7 @@ async function handleCheckQuota(args: { provider?: string; connectionId?: string await logToolCall("omniroute_check_quota", args, result, Date.now() - start, true); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err); await logToolCall("omniroute_check_quota", args, null, Date.now() - start, false, msg); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } @@ -562,7 +560,7 @@ async function handleRouteRequest(args: { ); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err); await logToolCall( "omniroute_route_request", { model: args.model }, @@ -611,7 +609,7 @@ async function handleCostReport(args: { period?: string }) { await logToolCall("omniroute_cost_report", args, result, Date.now() - start, true); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err); await logToolCall("omniroute_cost_report", args, null, Date.now() - start, false, msg); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } @@ -631,7 +629,7 @@ async function handleListModelsCatalog(args: { provider?: string; capability?: s ); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err); await logToolCall("omniroute_list_models_catalog", args, null, Date.now() - start, false, msg); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } @@ -660,7 +658,7 @@ async function handleWebSearch(args: { await logToolCall("omniroute_web_search", args, result, Date.now() - start, true); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err); await logToolCall("omniroute_web_search", args, null, Date.now() - start, false, msg); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } @@ -686,7 +684,7 @@ async function handleXSearch(args: { await logToolCall("omniroute_x_search", args, result, Date.now() - start, true); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err); await logToolCall("omniroute_x_search", args, null, Date.now() - start, false, msg); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } @@ -726,7 +724,7 @@ async function handleWebFetch(args: { await logToolCall("omniroute_web_fetch", args, result, Date.now() - start, true); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err); await logToolCall("omniroute_web_fetch", args, null, Date.now() - start, false, msg); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } @@ -1182,7 +1180,7 @@ export function createMcpServer(options?: CreateMcpServerOptions): McpServer { const result = await toolDef.handler(parsedArgs, extra); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err, "Memory tool execution failed"); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } }, @@ -1209,7 +1207,7 @@ export function createMcpServer(options?: CreateMcpServerOptions): McpServer { const result = await toolDef.handler(parsedArgs, extra); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err, "Skill tool execution failed"); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } }, @@ -1234,7 +1232,7 @@ export function createMcpServer(options?: CreateMcpServerOptions): McpServer { const result = await toolDef.handler(parsedArgs, extra); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err, "Agent skill tool execution failed"); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } }) @@ -1259,7 +1257,7 @@ export function createMcpServer(options?: CreateMcpServerOptions): McpServer { const result = await toolDef.handler(parsedArgs); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err, "GitHub skill tool execution failed"); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } }, @@ -1286,7 +1284,7 @@ export function createMcpServer(options?: CreateMcpServerOptions): McpServer { const result = await toolDef.handler(parsedArgs, extra); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err, "Plugin tool execution failed"); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } }, @@ -1313,7 +1311,7 @@ export function createMcpServer(options?: CreateMcpServerOptions): McpServer { const result = await toolDef.handler(parsedArgs, extra); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err, "Compression tool execution failed"); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } }, @@ -1350,7 +1348,7 @@ export function createMcpServer(options?: CreateMcpServerOptions): McpServer { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }], }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err, "Pool tool execution failed"); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } }, @@ -1378,7 +1376,7 @@ export function createMcpServer(options?: CreateMcpServerOptions): McpServer { const result = await toolDef.handler(parsedArgs, extra); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err, "Gamification tool execution failed"); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } }, @@ -1405,7 +1403,7 @@ export function createMcpServer(options?: CreateMcpServerOptions): McpServer { const result = await toolDef.handler(parsedArgs, extra); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err, "Notion tool execution failed"); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } }, @@ -1432,8 +1430,9 @@ export function createMcpServer(options?: CreateMcpServerOptions): McpServer { const result = await toolDef.handler(parsedArgs, extra); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (error) { + const msg = toSafeMcpErrorMessage(error, "Local corpus tool execution failed"); return { - content: [{ type: "text" as const, text: `Error: ${sanitizeErrorMessage(error)}` }], + content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true, }; } @@ -1461,7 +1460,7 @@ export function createMcpServer(options?: CreateMcpServerOptions): McpServer { const result = await toolDef.handler(parsedArgs, extra); return { content: [{ type: "text" as const, text: JSON.stringify(result, null, 2) }] }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err, "Obsidian tool execution failed"); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true }; } }, @@ -1502,7 +1501,7 @@ export function createMcpServer(options?: CreateMcpServerOptions): McpServer { ], }; } catch (err) { - const msg = err instanceof Error ? err.message : String(err); + const msg = toSafeMcpErrorMessage(err, "Skill execution failed"); return { content: [{ type: "text" as const, text: `Error: ${msg}` }], isError: true, diff --git a/open-sse/translator/response/openai-responses.ts b/open-sse/translator/response/openai-responses.ts index a2244f01f5..43453c0b28 100644 --- a/open-sse/translator/response/openai-responses.ts +++ b/open-sse/translator/response/openai-responses.ts @@ -5,6 +5,7 @@ import { register } from "../registry.ts"; import { FORMATS } from "../formats.ts"; import { appendToolCallArgumentDelta } from "../../utils/toolCallArguments.ts"; +import { projectCompletedStreamError } from "../../utils/streamErrorFormat.ts"; import { fallbackToolCallId } from "../helpers/toolCallHelper.ts"; import { shouldParseTextualReasoningTags } from "../../handlers/responseSanitizer.ts"; import { getReadableReasoningValue } from "../../utils/reasoningFields.ts"; @@ -746,6 +747,7 @@ function sendCompleted(state, emit) { // translator or the OpenAI-Responses translator itself when the upstream // SSE stream emits a JSON error object after partial content. const upstreamErr = state.upstreamError; + const publicUpstreamError = projectCompletedStreamError(upstreamErr); const response: Record = { id: state.responseId, @@ -753,9 +755,7 @@ function sendCompleted(state, emit) { created_at: state.created, status: upstreamErr ? "failed" : "completed", background: false, - error: upstreamErr - ? { code: String(upstreamErr.status ?? ""), message: upstreamErr.message ?? "" } - : null, + error: publicUpstreamError, output, }; diff --git a/open-sse/utils/credentialPatterns.ts b/open-sse/utils/credentialPatterns.ts new file mode 100644 index 0000000000..02784a4ae5 --- /dev/null +++ b/open-sse/utils/credentialPatterns.ts @@ -0,0 +1,79 @@ +/** Pure credential signatures shared by guardrails and public error sanitization. */ +export interface CredentialPattern { + name: string; + regex: RegExp; + replacement: string; +} + +export const CREDENTIAL_PATTERNS: CredentialPattern[] = [ + { name: "openai_proj", regex: /sk-proj-[A-Za-z0-9_-]{20,}/g, replacement: "[REDACTED:openai]" }, + { name: "openai", regex: /\bsk-[A-Za-z0-9]{48}\b/g, replacement: "[REDACTED:openai]" }, + { + name: "anthropic", + regex: /sk-ant-api[0-9]?-[A-Za-z0-9_-]{20,}/g, + replacement: "[REDACTED:anthropic]", + }, + { + name: "anthropic_alt", + regex: /sk-ant-[A-Za-z0-9_-]{20,}/g, + replacement: "[REDACTED:anthropic]", + }, + { name: "google", regex: /AIza[0-9A-Za-z_-]{35}/g, replacement: "[REDACTED:google]" }, + { name: "huggingface", regex: /hf_[A-Za-z0-9]{34}/g, replacement: "[REDACTED:hf]" }, + { name: "replicate", regex: /r8_[A-Za-z0-9]{37}/g, replacement: "[REDACTED:replicate]" }, + { name: "github", regex: /gh[pousr]_[A-Za-z0-9]{36,}/g, replacement: "[REDACTED:github]" }, + { name: "slack", regex: /xox[bpoa]-[A-Za-z0-9-]{10,}/g, replacement: "[REDACTED:slack]" }, + { name: "linear", regex: /lin_api_[A-Za-z0-9]{40}/g, replacement: "[REDACTED:linear]" }, + { name: "notion", regex: /secret_[A-Za-z0-9]{43}/g, replacement: "[REDACTED:notion]" }, + { name: "npm", regex: /npm_[A-Za-z0-9]{36}/g, replacement: "[REDACTED:npm]" }, + { + name: "postman", + regex: /PMAK-[a-f0-9]{8}-[a-f0-9]{32}/g, + replacement: "[REDACTED:postman]", + }, + { + name: "discord", + regex: /\b[MN][A-Za-z0-9]{23}\.[A-Za-z0-9]{6}\.[A-Za-z0-9]{27}\b/g, + replacement: "[REDACTED:discord]", + }, + { + name: "stripe", + regex: /(?:sk|rk)_(?:live|test)_[0-9a-zA-Z]{24,}/g, + replacement: "[REDACTED:stripe]", + }, + { + name: "square", + regex: /sq0(?:atp-[0-9A-Za-z_-]{22}|csp-[0-9A-Za-z_-]{43})/g, + replacement: "[REDACTED:square]", + }, + { name: "aws_access_key", regex: /AKIA[0-9A-Z]{16}/g, replacement: "[REDACTED:aws]" }, + { name: "twilio", regex: /\bSK[0-9a-fA-F]{32}\b/g, replacement: "[REDACTED:twilio]" }, + { + name: "sendgrid", + regex: /SG\.[A-Za-z0-9_-]{22}\.[A-Za-z0-9_-]{43}/g, + replacement: "[REDACTED:sendgrid]", + }, + { name: "mailgun", regex: /key-[a-f0-9]{32}/g, replacement: "[REDACTED:mailgun]" }, + { + name: "private_key", + regex: + /-----BEGIN (?:RSA |EC |DSA |OPENSSH |PGP )?PRIVATE KEY-----[\s\S]*?-----END (?:RSA |EC |DSA |OPENSSH |PGP )?PRIVATE KEY-----/g, + replacement: "[REDACTED:private_key]", + }, + { + name: "jwt", + regex: /\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\b/g, + replacement: "[REDACTED:jwt]", + }, + { + name: "connection_string", + regex: /(?:mongodb(?:\+srv)?|postgres(?:ql)?|mysql|redis|amqp):\/\/[^:/@\s"']+:[^:/@\s"']+@/g, + replacement: "[REDACTED:connection_string]", + }, + { + name: "auth_header", + regex: + /((?:["\x27]?(?:Authorization|x-api-key|api-key|apikey)["\x27]?\s*[:=]\s*["\x27]?)(?:(?:Bearer|Basic|Token)\s+)?)[A-Za-z0-9._~+/=-]{10,}/gi, + replacement: "$1[REDACTED:auth_header]", + }, +]; diff --git a/open-sse/utils/error.ts b/open-sse/utils/error.ts index 66563c618f..5d7357eb1c 100644 --- a/open-sse/utils/error.ts +++ b/open-sse/utils/error.ts @@ -1,15 +1,18 @@ import { CORS_HEADERS } from "./cors.ts"; import { unwrapClinepassEnvelope } from "./clinepassEnvelope.ts"; +import { + redactSensitiveErrorText, + sanitizeErrorMessage, + sanitizeUpstreamDetails, +} from "./errorSanitization.ts"; import { getDefaultErrorMessage, getErrorInfo } from "../config/errorConfig.ts"; import { normalizePayloadForLog } from "@/lib/logPayloads"; import type { ModelCooldownErrorPayload } from "@/types"; import { buildPassthroughErrorResponse } from "./upstreamErrorPassthrough.ts"; -/** - * Sanitize an error message to prevent stack trace exposure in API responses. - * Strips stack traces, file paths, and absolute Windows/POSIX paths from - * error messages before they reach the client. - */ +export { redactSensitiveErrorText, sanitizeErrorMessage, sanitizeUpstreamDetails }; + +/** Client-visible error shape; dynamic fields are projected through canonical boundaries. */ interface ErrorResponseBody { error: { message: string; @@ -20,119 +23,6 @@ interface ErrorResponseBody { upstream_details?: Record | null; // sanitized upstream provider body } -// Length cap protects against pathological inputs even before tokenization. -const MAX_ERROR_LEN = 4096; -const SOURCE_EXT = ["ts", "tsx", "js", "jsx", "mjs", "cjs"] as const; - -function looksLikeAbsolutePath(tok: string): boolean { - // POSIX: "/<...>.ts" (optionally followed by :line[:col]). - // Windows: "C:\<...>.ts" or "C:/<...>.ts". - if (tok.length < 4 || tok.length > 2048) return false; - const isPosix = tok.charCodeAt(0) === 0x2f; // '/' - const isWindows = tok.length > 2 && tok.charCodeAt(1) === 0x3a && /[A-Za-z]/.test(tok[0]); - if (!isPosix && !isWindows) return false; - const dot = tok.lastIndexOf("."); - if (dot <= 0 || dot === tok.length - 1) return false; - const ext = tok - .slice(dot + 1) - .split(":", 1)[0] - .toLowerCase(); - return (SOURCE_EXT as readonly string[]).includes(ext); -} - -/** - * Raw credential shapes that carry no `key=` label to key off — the token IS the - * whole match, so the only way to redact them is to recognize the shape. - * - * GHSA-qv45-56jc-4wmj: `upstreamErrorPassthrough.ts` already recognized `sk-` - * and refused verbatim passthrough for bodies containing it, then handed those - * bodies to THIS sanitizer — which had no such pattern, so the key came back to - * the caller anyway. The passthrough file's comment claimed to "mirror the - * vocabulary of redactSensitiveErrorText"; the mirror had drifted. It now - * imports this array instead of keeping a second copy, so the two cannot drift - * again. - * - * Quantifiers are upper-bounded (AGENTS.md → PII learnings §1, ReDoS): these run - * over untrusted upstream error bodies. - */ -export const RAW_CREDENTIAL_PATTERNS: ReadonlyArray = [ - // OpenAI/Anthropic/Stripe-style secret keys: sk-…, sk-ant-…, sk_live_… - /\bsk[-_][A-Za-z0-9._-]{8,200}/g, - // Google API keys - /\bAIza[A-Za-z0-9_-]{20,200}/g, - // JWTs (three base64url segments) - /\beyJ[A-Za-z0-9_-]{8,400}\.[A-Za-z0-9_-]{8,800}\.[A-Za-z0-9_-]{8,800}/g, -]; - -export function redactSensitiveErrorText(value: string): string { - let out = value; - for (const pattern of RAW_CREDENTIAL_PATTERNS) { - out = out.replace(pattern, "[REDACTED_CREDENTIAL]"); - } - return out - .replace(/data:[^,\s]+;base64,[A-Za-z0-9+/=_-]+/gi, "[REDACTED_DATA_URL]") - .replace(/\b(Bearer|Basic)\s+[A-Za-z0-9._~+/=-]+/gi, "$1 [REDACTED]") - .replace( - /(["']?(?:api[_-]?key|access[_-]?token|authorization|cookie|secret)["']?\s*[:=]\s*["'])[^"']*(["'])/gi, - "$1[REDACTED]$2" - ) - .replace( - /(["']?(?:api[_-]?key|access[_-]?token|authorization|cookie|secret)["']?\s*[:=]\s*)[^"',\s}]+/gi, - "$1[REDACTED]" - ); -} - -/** - * Strip stack-trace tail and absolute source paths from error messages. - * - * Implemented via simple whitespace tokenization (linear time) instead of a - * single complex regex, so CodeQL `js/polynomial-redos` stays clean even when - * the runtime error message is attacker-controlled. - */ -export function sanitizeErrorMessage(message: unknown): string { - let str = typeof message === "string" ? message : String(message ?? ""); - if (str.length > MAX_ERROR_LEN) str = str.slice(0, MAX_ERROR_LEN); - const nl = str.indexOf("\n"); - const firstLine = nl >= 0 ? str.slice(0, nl) : str; - // Preserve original whitespace by splitting on captured separator. - const parts = firstLine.split(/(\s+)/); - for (let i = 0; i < parts.length; i++) { - if (looksLikeAbsolutePath(parts[i])) parts[i] = ""; - } - return redactSensitiveErrorText(parts.join("")); -} - -const BLOCKED_KEYS = - /stack|trace|path|file|cwd|dir|password|secret|token|key|authorization|cookie/i; -const MAX_DEPTH = 4; - -/** - * Recursively sanitize an arbitrary JSON value from an upstream provider body. - * - Strings: run through sanitizeErrorMessage (strips stacks + absolute paths). - * - Keys matching BLOCKED_KEYS are dropped (credential/path guards). - * - Depth capped at MAX_DEPTH to prevent pathological nesting. - * - Arrays capped at 32 elements. - * - Returns null for null/undefined/non-JSON-serializable values. - */ -export function sanitizeUpstreamDetails(value: unknown, depth = 0): unknown { - if (depth > MAX_DEPTH) return "[truncated]"; - if (value === null || value === undefined) return null; - if (typeof value === "string") return sanitizeErrorMessage(value); - if (typeof value === "number" || typeof value === "boolean") return value; - if (Array.isArray(value)) { - return value.slice(0, 32).map((v) => sanitizeUpstreamDetails(v, depth + 1)); - } - if (typeof value === "object") { - const out: Record = {}; - for (const [k, v] of Object.entries(value as Record)) { - if (BLOCKED_KEYS.test(k)) continue; - out[k] = sanitizeUpstreamDetails(v, depth + 1); - } - return out; - } - return null; -} - /** Optional caller classification; when set, wins over status-derived defaults. */ export type ErrorBodyClassification = { type?: string; @@ -140,6 +30,279 @@ export type ErrorBodyClassification = { reason?: string; }; +const PUBLIC_ERROR_IDENTIFIER = /^[A-Za-z0-9][A-Za-z0-9._-]{0,127}$/; +const SAFE_PUBLIC_ERROR_IDENTIFIERS = new Set([ + "abort", + "aborted", + "account_semaphore_capacity", + "acp_cancelled", + "acp_early_exit", + "acp_error", + "acp_output_too_large", + "acp_session_mismatch", + "acp_timeout", + "admission_aborted", + "admission_deadline", + "admission_lane_evicted", + "admission_oversized", + "admission_queue_full", + "admission_shutdown", + "admission_unavailable", + "all_accounts_inactive", + "all_targets_skipped", + "antigravity_pre_response_timeout", + "api_error", + "authentication_error", + "authentication_required", + "auth_error", + "bad_gateway", + "bad_request", + "bedrock_stream_error", + "billing_error", + "blackbox_auth_required", + "blackbox_rate_limit", + "blackbox_subscription_required", + "body_exceeds_budget", + "browser_stream_inconsistent", + "capability_mismatch", + "cf_mitigated_challenge", + "chat_admission_busy", + "chat_history_too_large", + "chatgpt_web_codex_error", + "chatgpt_web_codex_turn_failed", + "chatgpt_session_expired", + "chatgpt_submission_ambiguous", + "chatgpt_submitted_turn_failed", + "chatgpt_subscription_unavailable", + "client_cancelled", + "client_closed_request", + "client_disconnected", + "cli_not_found", + "cloudflare_challenge", + "cloudflare_or_bot", + "codex_app_server_unconfigured", + "codex_app_server_turn_failed", + "combo_target_timeout", + "combo_timeout", + "compaction_control_unavailable", + "compaction_handoff_failed", + "connector_error", + "connector_not_found", + "connection_error", + "context_length_exceeded", + "context_window", + "chipotle_error", + "devin_agentic_error", + "devin_cli_error", + "devin_desktop_error", + "devin_internal_tool_execution", + "duplicate_tool_use_id", + "direct_response_start_timeout", + "eai_again", + "econnrefused", + "econnreset", + "empty_acp_output", + "empty_content", + "empty_messages", + "empty_response", + "executor_contract_violation", + "error", + "etimedout", + "executor_error", + "feature_disabled", + "gateway_timeout", + "gemini_tpm_exhausted", + "gcp_project_required", + "grok_error", + "insufficient_quota", + "incompatible_reasoning_effort", + "internal_server_error", + "invalid_acp_frame", + "invalid_acp_upstream", + "invalid_api_key", + "invalid_kiro_tool_call", + "invalid_request", + "invalid_request_error", + "invalid_previous_response_binding", + "invalid_tool_arguments", + "invalid_tool_choice", + "invalid_tool_json", + "invalid_tool_name", + "invalid_tools", + "invalid_trailer", + "lease_action_invalid", + "lease_api_key_invalid", + "lease_authentication_required", + "lease_authorization_mismatch", + "lease_capacity_unavailable", + "lease_connection_mismatch", + "lease_content_type_required", + "lease_context_invalid", + "lease_context_required", + "lease_error", + "lease_fence_stale", + "lease_key_configuration_invalid", + "lease_key_policy_invalid", + "lease_model_invalid", + "lease_no_eligible_connection", + "lmarena_error", + "lease_required", + "lease_scope_required", + "lease_service_unavailable", + "lease_eligibility_unavailable", + "lease_unsupported_route", + "lease_unsupported_transport", + "message_limit", + "missing_credits", + "meta_ai_empty_response", + "meta_ai_mode_switch_failed", + "meta_ai_warmup_failed", + "meta_ai_ws_error", + "missing_tool_name", + "missing_tool_use_id", + "mixed_tool_narrative", + "missing_authorization", + "missing_cookie", + "missing_project_id", + "missing_credentials", + "missing_session_id", + "model_not_found", + "model_not_supported", + "model_shutdown", + "multipart_protocol_violation", + "multiple_tool_requests", + "native_codex_pinned_model_unavailable", + "network_error", + "no_free_eligible_connection", + "not_found", + "oauth_missing_project_id", + "orphan_tool_result", + "payload_too_large", + "payment_required", + "permission_error", + "premium_model_requires_key", + "prompt_attachment_integrity", + "provider_error", + "provider_retired", + "provider_unavailable", + "pplx_error", + "proxy_unavailable", + "proxy_family_unavailable", + "proxy_request_failed", + "proxy_unreachable", + "quota_exhausted", + "quota_not_allocated", + "quota_only", + "rate_limit_error", + "rate_limit_execution_timeout", + "rate_limit_exceeded", + "rate_limit_queue_full", + "rate_limit_queue_timeout", + "rate_limit_queue_wedged", + "rate_limit_longer_reached", + "rate_limit_reached", + "rate_limited", + "reached_limit", + "relay_timeout", + "resource_pressure", + "resource_exhausted", + "request_failed", + "risk_session_stale", + "server_error", + "semaphore_queue_full", + "semaphore_timeout", + "service_unavailable", + "service_not_running", + "session_expired", + "session_pool_exhausted", + "spawn_failed", + "stream_error", + "stream_disconnected", + "stream_early_eof", + "stream_idle_timeout", + "stream_pipeline_error", + "stream_readiness_timeout", + "stream_terminated", + "stream_timeout", + "storage_encryption_stale", + "structure_limit", + "structured_output", + "structured_output_validation_failed", + "timeout_error", + "timeout", + "token_limit_exceeded", + "token_required", + "tls_client_unavailable", + "tls_circuit_open", + "tls_fingerprint_failed", + "tls_session_capacity", + "tool_calling_not_supported", + "tools", + "undeclared_historical_tool", + "und_err_body_timeout", + "und_err_connect_timeout", + "und_err_headers_timeout", + "und_err_socket", + "unexpected_acp_response", + "unexecuted_tool_intent", + "unavailable", + "unknown_devin_model", + "unknown_tool", + "unverified_codex_client", + "unsafe_devin_home", + "unsupported_acp_version", + "unsupported_content_block", + "unsupported_control_for_provider", + "unsupported_endpoint", + "unsupported_image_block", + "unsupported_role", + "unsupported_system_block", + "upstream_error", + "upstream_access_denied", + "upstream_auth_error", + "upstream_empty_response", + "upstream_response_failed", + "upstream_response_error", + "upstream_server_error", + "upstream_protocol_error", + "upstream_timeout", + "upstream_websocket_connect_failed", + "upstream_websocket_error", + "usage_limit_reached", + "unsupported_feature", + "unsupported_runtime", + "video_artifact_content_type_invalid", + "video_artifact_download_failed", + "video_artifact_not_ready", + "video_artifact_signature_invalid", + "video_artifact_too_large", + "video_artifact_unavailable", + "video_artifact_url_blocked", + "video_artifact_url_invalid", + "vision", + "claude_web_protocol_error", + "wreq_unavailable", +]); + +function isSafePublicErrorIdentifier(value: string): boolean { + if (!PUBLIC_ERROR_IDENTIFIER.test(value)) return false; + if (/^[1-5]\d{2}$/.test(value)) return true; + if (/^HTTP_[1-5]\d{2}$/i.test(value)) return true; + return SAFE_PUBLIC_ERROR_IDENTIFIERS.has(value.toLowerCase()); +} + +/** Project an internal classification onto the bounded client-visible identifier vocabulary. */ +export function projectPublicErrorIdentifier(value: unknown, fallback: unknown): string { + const safeFallback = + fallback === "" + ? "" + : typeof fallback === "string" && isSafePublicErrorIdentifier(fallback) + ? fallback + : "error"; + if (typeof value !== "string") return safeFallback; + return isSafePublicErrorIdentifier(value) ? value : safeFallback; +} + /** * Build OpenAI-compatible error response body. Message is always sanitized * so callers do not need to remember to strip stack traces themselves. @@ -156,13 +319,17 @@ export function buildErrorBody( ): ErrorResponseBody { const errorInfo = getErrorInfo(statusCode); const safeMessage = sanitizeErrorMessage(message) || getDefaultErrorMessage(statusCode); + const safeReason = + typeof classification?.reason === "string" && isSafePublicErrorIdentifier(classification.reason) + ? classification.reason + : undefined; const body: ErrorResponseBody = { error: { message: safeMessage, - type: classification?.type ?? errorInfo.type, - code: classification?.code ?? errorInfo.code, - reason: classification?.reason, + type: projectPublicErrorIdentifier(classification?.type, errorInfo.type), + code: projectPublicErrorIdentifier(classification?.code, errorInfo.code), + reason: safeReason, }, }; @@ -211,7 +378,7 @@ export interface ComboRecoveryHint { action: ComboRecoveryAction; /** Seconds the client should wait before retrying. Only meaningful when action="wait". */ retry_after_seconds?: number; - /** Human-readable next step — included verbatim in the error body for non-MCP clients. */ + /** Human-readable next step — sanitized and length-capped for non-MCP clients. */ next_step: string; } @@ -231,21 +398,36 @@ export interface ComboDiagnostics { } function clampDiagStr(v: unknown, max = 128): string { - return typeof v === "string" ? v.slice(0, max).replace(/[\r\n]+/g, " ") : ""; + return typeof v === "string" ? sanitizeErrorMessage(v).slice(0, max) : ""; +} + +const RECOVERY_ROUTE_PLACEHOLDERS = [ + ["/dashboard/providers", "OMNIROUTE_SAFE_DASHBOARD_PROVIDERS_ROUTE"], +] as const; + +function clampRecoveryStr(value: unknown, max: number): string { + if (typeof value !== "string") return ""; + let projected = value; + for (const [route, placeholder] of RECOVERY_ROUTE_PLACEHOLDERS) { + projected = projected.replaceAll(route, placeholder); + } + projected = sanitizeErrorMessage(projected); + for (const [route, placeholder] of RECOVERY_ROUTE_PLACEHOLDERS) { + projected = projected.replaceAll(placeholder, route); + } + return projected.slice(0, max); } /** - * HTTP header values must be Latin1/ByteString (undici throws a TypeError - * otherwise — see #6612). Replace any codepoint outside the Latin1 range - * (0-255) with "?" so header construction never throws. Only used for the - * literal header value; the JSON body keeps the original, unsanitized - * readable text via `sanitizeComboDiagnostics`. + * HTTP header values must exclude controls and remain ByteString-compatible + * (undici throws a TypeError otherwise — see #6612). Replace every codepoint + * outside printable ASCII with "?" so header construction never throws. */ function toHeaderSafeAscii(v: string): string { let out = ""; for (let i = 0; i < v.length; i++) { const code = v.charCodeAt(i); - out += code > 255 ? "?" : v[i]; + out += code < 0x20 || code > 0x7e ? "?" : v[i]; } return out; } @@ -270,7 +452,7 @@ export function sanitizeRecoveryHint( if (!action || !RECOVERY_ACTIONS.has(action)) return undefined; // Reject empty OR whitespace-only next_step — the value must render usefully as a // header and as a body field. A whitespace-only string would print as a blank hint. - const next_step = clampDiagStr(r.next_step, 200).trim(); + const next_step = clampRecoveryStr(r.next_step, 200).trim(); if (!next_step) return undefined; const hint: ComboRecoveryHint = { action, next_step }; if (typeof r.retry_after_seconds === "number" && Number.isFinite(r.retry_after_seconds)) { @@ -321,12 +503,10 @@ export function errorResponseWithComboDiagnostics( opts: { code?: string; type?: string } = {} ): Response { const safe = sanitizeComboDiagnostics(diagnostics); - const body = buildErrorBody(statusCode, message) as ErrorResponseBody & { + const body = buildErrorBody(statusCode, message, undefined, opts) as ErrorResponseBody & { diagnostics?: ComboDiagnostics; recovery_hint?: ComboRecoveryHint; }; - if (opts.code) body.error.code = opts.code; - if (opts.type) body.error.type = opts.type; body.diagnostics = safe; if (safe.recovery) body.recovery_hint = safe.recovery; const excludedHeader = toHeaderSafeAscii( @@ -427,6 +607,29 @@ function normalizeRetryAfterSeconds(retryAfter?: string | number | Date | null): return 1; } +const MAX_PUBLIC_CONTEXT_LABEL_LENGTH = 256; + +function projectPublicContextLabel(value: unknown): string | null { + if (typeof value !== "string") return null; + const label = value.trim(); + if ( + label.length === 0 || + label.length > MAX_PUBLIC_CONTEXT_LABEL_LENGTH || + /[\u0000-\u001f\u007f]/.test(label) + ) { + return null; + } + return sanitizeErrorMessage(label) === label ? label : null; +} + +function projectPublicRetryTimestamp(value: unknown): string | null { + if (typeof value !== "string") return null; + const timestamp = value.trim(); + if (!/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{3}Z$/.test(timestamp)) return null; + const parsed = Date.parse(timestamp); + return Number.isFinite(parsed) && new Date(parsed).toISOString() === timestamp ? timestamp : null; +} + /** * Parse Antigravity error message to extract retry time * Example: "You have exhausted your capacity on this model. Your quota will reset after 2h7m23s." @@ -470,7 +673,7 @@ export function parseAntigravityRetryTime(message: unknown): number | null { * @returns {Promise<{statusCode: number, message: string, retryAfterMs: number|null, responseBody: unknown}>} */ export async function parseUpstreamError(response: Response, provider: string | null = null) { - let message: unknown = ""; + let message = ""; let retryAfterMs: number | null = null; let responseBody: unknown = null; let errorCode: unknown = undefined; @@ -490,9 +693,15 @@ export async function parseUpstreamError(response: Response, provider: string | // stack) — still routed through sanitizeErrorMessage/buildErrorBody by // every consumer below (Rule #12). const { error: clinepassEnvError } = unwrapClinepassEnvelope(json, provider); - message = clinepassEnvError + const extractedMessage = clinepassEnvError ? clinepassEnvError.message - : json.error?.message || json.message || json.error || text; + : json.error?.message || + json.message || + (typeof json.error === "string" ? json.error : null); + message = + typeof extractedMessage === "string" + ? extractedMessage + : `Upstream error: ${response.status}`; errorCode = json.error?.code || json.code; errorType = json.error?.type || json.type; } catch { @@ -503,7 +712,7 @@ export async function parseUpstreamError(response: Response, provider: string | responseBody = { _rawText: message }; } - const messageStr = typeof message === "string" ? message : JSON.stringify(message); + const messageStr = message; const retryAfterHeader = response.headers?.get?.("retry-after"); if (retryAfterHeader && !retryAfterMs) { @@ -573,13 +782,10 @@ export function createErrorResult( upstreamDetails?: unknown, opts?: { passthrough?: boolean } ) { - const body = buildErrorBody(statusCode, message, upstreamDetails); - if (errorCode) { - body.error.code = errorCode; - } - if (errorType) { - body.error.type = errorType; - } + const body = buildErrorBody(statusCode, message, upstreamDetails, { + code: errorCode, + type: errorType, + }); const result: { success: false; @@ -619,8 +825,8 @@ export function createErrorResult( result.retryAfterMs = retryAfterMs; } - // Opt-in relay of the verbatim upstream error body (Claude Code auto-recover - // contract — see upstreamErrorPassthrough.ts). Only swaps `result.response`; + // Opt-in relay of the recursively sanitized upstream JSON shape (Claude Code + // auto-recover contract — see upstreamErrorPassthrough.ts). Only swaps `result.response`; // `result.error`/`rawMessage`/`errorType`/`errorCode` stay untouched so // server-side classification (checkFallbackError, combo retry logic, etc.) // never sees a different value depending on this flag. @@ -653,7 +859,9 @@ export function unavailableResponse( retryAfterHuman?: string ) { const retryAfterSec = normalizeRetryAfterSeconds(retryAfter); - const msg = retryAfterHuman ? `${message} (${retryAfterHuman})` : message; + const safeMessage = sanitizeErrorMessage(message) || getDefaultErrorMessage(statusCode); + const safeRetryAfterHuman = retryAfterHuman ? sanitizeErrorMessage(retryAfterHuman) : ""; + const msg = safeRetryAfterHuman ? `${safeMessage} (${safeRetryAfterHuman})` : safeMessage; return new Response(JSON.stringify({ error: { message: msg } }), { status: statusCode, headers: { @@ -668,13 +876,14 @@ export function providerCircuitOpenResponse( retryAfter?: string | number | Date | null ) { const retryAfterSec = normalizeRetryAfterSeconds(retryAfter); + const safeProvider = projectPublicContextLabel(provider) ?? "unknown"; return new Response( JSON.stringify({ error: { - message: `Provider ${provider} circuit breaker is open`, + message: `Provider ${safeProvider} circuit breaker is open`, type: "server_error", code: "provider_circuit_open", - provider, + provider: safeProvider, retry_after: retryAfterSec, }, }), @@ -700,9 +909,10 @@ export function buildModelCooldownBody({ retryAfterAt?: string | null; credentialsCoolingCount?: number | null; }): ModelCooldownErrorPayload { - const resolvedModel = typeof model === "string" && model.trim().length > 0 ? model.trim() : null; - const resolvedRetryAfterAt = - typeof retryAfterAt === "string" && retryAfterAt.length > 0 ? retryAfterAt : null; + const resolvedModel = projectPublicContextLabel(model); + const resolvedRetryAfterAt = projectPublicRetryTimestamp(retryAfterAt); + const resolvedResetSeconds = + Number.isFinite(retryAfterSec) && retryAfterSec > 0 ? Math.max(Math.ceil(retryAfterSec), 1) : 1; const resolvedCoolingCount = typeof credentialsCoolingCount === "number" && Number.isFinite(credentialsCoolingCount) && @@ -718,7 +928,7 @@ export function buildModelCooldownBody({ type: "rate_limit_error", code: "model_cooldown", ...(resolvedModel ? { model: resolvedModel } : {}), - reset_seconds: Math.max(Math.ceil(retryAfterSec), 1), + reset_seconds: resolvedResetSeconds, ...(resolvedRetryAfterAt ? { retry_after: resolvedRetryAfterAt } : {}), ...(resolvedCoolingCount ? { credentials_cooling: resolvedCoolingCount } : {}), }, diff --git a/open-sse/utils/errorPathRedaction.ts b/open-sse/utils/errorPathRedaction.ts new file mode 100644 index 0000000000..2b372af583 --- /dev/null +++ b/open-sse/utils/errorPathRedaction.ts @@ -0,0 +1,905 @@ +const SOURCE_EXT = ["ts", "tsx", "js", "jsx", "mjs", "cjs", "mts", "cts"] as const; +const NATIVE_EXT = ["node", "so", "dylib", "dll"] as const; +const LEADING_PATH_PUNCTUATION = "'\"`([{<"; +const TRAILING_PATH_PUNCTUATION = "'\"`)]}>.,;:!?"; +const PATH_SPAN_END_PUNCTUATION = "'\"`)]}>.,;:!?"; +const FILE_URI_PREFIX = "file://"; +const HTTP_METHODS = [ + "GET", + "POST", + "PUT", + "PATCH", + "DELETE", + "OPTIONS", + "HEAD", + "CONNECT", + "TRACE", +] as const; +const CLEAR_PROSE_BOUNDARIES = [ + "after", + "because", + "before", + "but", + "crashed", + "denied", + "eacces", + "enoent", + "expired", + "failed", + "rejected", + "retry", + "then", + "when", + "while", +] as const; +const POSIX_FILESYSTEM_ROOTS = [ + "/Users", + "/app", + "/boot", + "/data", + "/dev", + "/etc", + "/home", + "/media", + "/mnt", + "/nix", + "/opt", + "/private", + "/proc", + "/root", + "/run", + "/srv", + "/sys", + "/tmp", + "/usr", + "/var", + "/workspace", +] as const; +const WINDOWS_ROOT_RELATIVE_ROOTS = new Set([ + "program files", + "programdata", + "temp", + "users", + "windows", +]); + +function isWindowsAbsolutePathAt(value: string, start: number): boolean { + const remaining = value.length - start; + if (remaining > 2) { + const first = value.charCodeAt(start); + const second = value.charCodeAt(start + 1); + if ((first === 0x5c && second === 0x5c) || (first === 0x2f && second === 0x2f)) { + return true; + } + } + if (remaining < 3 || value.charCodeAt(start + 1) !== 0x3a) return false; + const driveLetter = value.charCodeAt(start); + const isAsciiLetter = + (driveLetter >= 0x41 && driveLetter <= 0x5a) || (driveLetter >= 0x61 && driveLetter <= 0x7a); + return ( + isAsciiLetter && (value.charCodeAt(start + 2) === 0x2f || value.charCodeAt(start + 2) === 0x5c) + ); +} + +function isWindowsAbsolutePath(value: string): boolean { + return isWindowsAbsolutePathAt(value, 0); +} + +function isWindowsRootRelativePathAt(value: string, start: number): boolean { + if ( + value.charCodeAt(start) !== 0x5c || + value.charCodeAt(start + 1) === 0x5c || + isWhitespace(value[start + 1]) + ) { + return false; + } + + const tokenEnd = findTokenEnd(value, start); + let firstSeparator = start + 1; + while (firstSeparator < tokenEnd && value.charCodeAt(firstSeparator) !== 0x5c) { + firstSeparator++; + } + const root = value.slice(start + 1, firstSeparator).toLowerCase(); + if (WINDOWS_ROOT_RELATIVE_ROOTS.has(root)) return true; + return ( + firstSeparator < tokenEnd - 1 || tokenContainsPathExtensionEvidence(value, start + 1, tokenEnd) + ); +} + +function hasAbsoluteFileUriAt(value: string, start: number): boolean { + const prefixEnd = start + FILE_URI_PREFIX.length; + return ( + value.length > prefixEnd && + value.slice(start, prefixEnd).toLowerCase() === FILE_URI_PREFIX && + !isWhitespace(value[prefixEnd]) + ); +} + +function hasAbsoluteFileUri(value: string): boolean { + return hasAbsoluteFileUriAt(value, 0); +} + +function isSyntacticallyAbsolutePathAt(value: string, start: number): boolean { + return ( + value.charCodeAt(start) === 0x2f || + isWindowsAbsolutePathAt(value, start) || + isWindowsRootRelativePathAt(value, start) || + hasAbsoluteFileUriAt(value, start) + ); +} + +function isAsciiDigit(code: number): boolean { + return code >= 0x30 && code <= 0x39; +} + +function isAsciiLetter(code: number): boolean { + return (code >= 0x41 && code <= 0x5a) || (code >= 0x61 && code <= 0x7a); +} + +function isAsciiAlphaNumeric(code: number): boolean { + return isAsciiDigit(code) || isAsciiLetter(code); +} + +function hasHttpUrlSchemeBefore(value: string, slashIndex: number): boolean { + for (const scheme of ["http:", "https:"]) { + const schemeStart = slashIndex - scheme.length; + if (schemeStart < 0 || value.slice(schemeStart, slashIndex).toLowerCase() !== scheme) continue; + if (schemeStart === 0 || !isAsciiAlphaNumeric(value.charCodeAt(schemeStart - 1))) return true; + } + return false; +} + +function isWhitespace(value: string): boolean { + return /\s/.test(value); +} + +function isRouteContextWord(value: string): boolean { + return value === "Route" || (HTTP_METHODS as readonly string[]).includes(value); +} + +function hasRouteContextBefore(value: string, candidateIndex: number): boolean { + let index = candidateIndex - 1; + while ( + index >= 0 && + (isWhitespace(value[index]) || + value.charCodeAt(index) === 0x28 || + value.charCodeAt(index) === 0x3a) + ) { + index--; + } + + const contextEnd = index + 1; + while (index >= 0 && isAsciiAlphaNumeric(value.charCodeAt(index))) index--; + return isRouteContextWord(value.slice(index + 1, contextEnd)); +} + +function isRouteContextToken(value: string): boolean { + let end = value.length; + while (end > 0 && !isAsciiAlphaNumeric(value.charCodeAt(end - 1))) end--; + let start = end; + while (start > 0 && isAsciiAlphaNumeric(value.charCodeAt(start - 1))) start--; + return isRouteContextWord(value.slice(start, end)); +} + +function matchesPosixFilesystemRootAt(value: string, start: number, root: string): boolean { + if (!value.startsWith(root, start)) return false; + const rootEnd = start + root.length; + return ( + rootEnd === value.length || + value.charCodeAt(rootEnd) === 0x2f || + PATH_SPAN_END_PUNCTUATION.includes(value[rootEnd]) + ); +} + +function isKnownPosixFilesystemPathAt(value: string, start: number): boolean { + return POSIX_FILESYSTEM_ROOTS.some((root) => matchesPosixFilesystemRootAt(value, start, root)); +} + +function isKnownPosixFilesystemPath(value: string): boolean { + return isKnownPosixFilesystemPathAt(value, 0); +} + +function looksLikeAbsolutePath(token: string): boolean { + // POSIX: common filesystem roots, with or without a source extension. + // Windows: drive-letter, UNC, or extended-length absolute paths. + // Source-file paths rooted elsewhere remain covered by SOURCE_EXT below. + if (token.length < 4 || token.length > 2048) return false; + const isPosix = token.charCodeAt(0) === 0x2f; + const isWindows = isWindowsAbsolutePath(token) || isWindowsRootRelativePathAt(token, 0); + if (!isPosix && !isWindows) return false; + if (isWindows) return true; + if (isKnownPosixFilesystemPath(token)) return true; + const dot = token.lastIndexOf("."); + if (dot <= 0 || dot === token.length - 1) return false; + const extension = token + .slice(dot + 1) + .split(":", 1)[0] + .toLowerCase(); + return ( + (SOURCE_EXT as readonly string[]).includes(extension) || + (NATIVE_EXT as readonly string[]).includes(extension) + ); +} + +function redactAbsolutePathToken(token: string, followsRouteContext: boolean): string { + let start = 0; + let end = token.length; + + while (start < end && LEADING_PATH_PUNCTUATION.includes(token[start])) start++; + while (end > start && TRAILING_PATH_PUNCTUATION.includes(token[end - 1])) end--; + + const candidate = token.slice(start, end); + const isFileUri = hasAbsoluteFileUri(candidate); + const pathCandidate = isFileUri ? candidate.slice(FILE_URI_PREFIX.length) : candidate; + + if ( + !isFileUri && + !isWindowsAbsolutePath(pathCandidate) && + !isWindowsRootRelativePathAt(pathCandidate, 0) && + pathCandidate.charCodeAt(0) === 0x2f && + followsRouteContext + ) { + return token; + } + if (!isFileUri && !looksLikeAbsolutePath(pathCandidate)) return token; + return `${token.slice(0, start)}${token.slice(end)}`; +} + +function findPathQuote(value: string, start: number, quote: string, takeFirst: boolean): number { + let candidate = value.indexOf(quote, start); + if (takeFirst || candidate < 0) return candidate < 0 ? value.length : candidate; + + while (candidate < value.length) { + const nextQuote = value.indexOf(quote, candidate + 1); + if (nextQuote < 0) return candidate; + // Two separately quoted absolute paths are unambiguous. Close the first + // candidate so the second one is scanned on its own; otherwise keep + // consuming quotes fail-closed because POSIX filenames may contain them. + if (isSyntacticallyAbsolutePathAt(value, nextQuote + 1)) return candidate; + candidate = nextQuote; + } + return value.length; +} + +function redactQuotedAbsolutePaths(value: string): string { + const parts: string[] = []; + let copyStart = 0; + let index = 0; + + while (index < value.length) { + const quote = value[index]; + if (quote !== "'" && quote !== '"' && quote !== "`") { + index++; + continue; + } + const candidateStart = index + 1; + if (!isSyntacticallyAbsolutePathAt(value, candidateStart)) { + index++; + continue; + } + + const isShieldedRoute = + value.charCodeAt(candidateStart) === 0x2f && + !isWindowsAbsolutePathAt(value, candidateStart) && + hasRouteContextBefore(value, index); + // Route/API contexts use their first closing quote so a later quoted + // filesystem path is still scanned independently. Filesystem candidates + // take the last matching quote on the line: POSIX filenames may themselves + // contain quote characters, whitespace, and punctuation, so earlier + // matches are ambiguous and must fail closed rather than expose a suffix. + const closingQuote = findPathQuote(value, candidateStart, quote, isShieldedRoute); + if (isShieldedRoute) { + if (closingQuote >= value.length) break; + index = closingQuote + 1; + continue; + } + parts.push(value.slice(copyStart, candidateStart), ""); + copyStart = closingQuote; + + if (closingQuote >= value.length) break; + index = closingQuote + 1; + } + + if (parts.length === 0) return value; + parts.push(value.slice(copyStart)); + return parts.join(""); +} + +function findPathExtensionEnd(value: string, dot: number): number { + let end = dot + 1; + const maxExtensionEnd = Math.min(value.length, end + 16); + while (end < maxExtensionEnd && isAsciiAlphaNumeric(value.charCodeAt(end))) end++; + if (end === dot + 1 || (end === maxExtensionEnd && isAsciiAlphaNumeric(value.charCodeAt(end)))) { + return -1; + } + let hasLetter = false; + for (let index = dot + 1; index < end; index++) { + if (isAsciiLetter(value.charCodeAt(index))) hasLetter = true; + } + if (!hasLetter) return -1; + + while (value.charCodeAt(end) === 0x3a) { + let coordinateEnd = end + 1; + if (!isAsciiDigit(value.charCodeAt(coordinateEnd))) break; + while (coordinateEnd < value.length && isAsciiDigit(value.charCodeAt(coordinateEnd))) { + coordinateEnd++; + } + end = coordinateEnd; + } + + if ( + end === value.length || + isWhitespace(value[end]) || + PATH_SPAN_END_PUNCTUATION.includes(value[end]) + ) { + return end; + } + return -1; +} + +function findTokenEnd(value: string, start: number): number { + let end = start; + while (end < value.length && !isWhitespace(value[end])) end++; + return end; +} + +function findExtensionEndInToken(value: string, start: number, end: number): number { + let lastExtensionEnd = -1; + for (let index = start; index < end; index++) { + const code = value.charCodeAt(index); + if (code === 0x2f || code === 0x5c) { + lastExtensionEnd = -1; + continue; + } + if (code !== 0x2e) continue; + const extensionEnd = findPathExtensionEnd(value, index); + if (extensionEnd >= 0 && extensionEnd <= end) lastExtensionEnd = extensionEnd; + } + return lastExtensionEnd; +} + +function tokenContainsPathExtensionEvidence(value: string, start: number, end: number): boolean { + for (let dot = start; dot < end; dot++) { + if (value.charCodeAt(dot) !== 0x2e) continue; + let extensionEnd = dot + 1; + const maxExtensionEnd = Math.min(end, extensionEnd + 16); + let hasLetter = false; + while (extensionEnd < maxExtensionEnd && isAsciiAlphaNumeric(value.charCodeAt(extensionEnd))) { + if (isAsciiLetter(value.charCodeAt(extensionEnd))) hasLetter = true; + extensionEnd++; + } + if ( + extensionEnd === dot + 1 || + !hasLetter || + (extensionEnd === maxExtensionEnd && + extensionEnd < end && + isAsciiAlphaNumeric(value.charCodeAt(extensionEnd))) + ) { + continue; + } + if ( + extensionEnd === end || + value.charCodeAt(extensionEnd) === 0x2f || + value.charCodeAt(extensionEnd) === 0x5c || + PATH_SPAN_END_PUNCTUATION.includes(value[extensionEnd]) + ) { + return true; + } + } + return false; +} + +function tokenContainsPathSeparator(value: string, start: number, end: number): boolean { + for (let index = start; index < end; index++) { + const code = value.charCodeAt(index); + if (code === 0x2f || code === 0x5c) return true; + } + return false; +} + +function remainderContainsFilesystemSeparator(value: string, start: number): boolean { + let tokenStart = start; + let previousToken = ""; + while (tokenStart < value.length) { + while (tokenStart < value.length && isWhitespace(value[tokenStart])) tokenStart++; + if (tokenStart >= value.length) return false; + + const tokenEnd = findTokenEnd(value, tokenStart); + const token = value.slice(tokenStart, tokenEnd).toLowerCase(); + const isHttpUrl = token.includes("http://") || token.includes("https://"); + let separatorIndex = tokenStart; + while ( + separatorIndex < tokenEnd && + value.charCodeAt(separatorIndex) !== 0x2f && + value.charCodeAt(separatorIndex) !== 0x5c + ) { + separatorIndex++; + } + const precedingSeparatorCode = + separatorIndex > tokenStart ? value.charCodeAt(separatorIndex - 1) : -1; + const contextIndex = + precedingSeparatorCode === 0x27 || + precedingSeparatorCode === 0x22 || + precedingSeparatorCode === 0x60 + ? separatorIndex - 1 + : separatorIndex; + const isShieldedRoute = + separatorIndex < tokenEnd && + value.charCodeAt(separatorIndex) === 0x2f && + !isWindowsAbsolutePathAt(value, separatorIndex) && + (isRouteContextToken(previousToken) || hasRouteContextBefore(value, contextIndex)); + if (!isHttpUrl && separatorIndex < tokenEnd && !isShieldedRoute) return true; + previousToken = value.slice(tokenStart, tokenEnd); + tokenStart = tokenEnd; + } + return false; +} + +function trimPathSpanEnd(value: string, start: number, end: number): number { + while (end > start && PATH_SPAN_END_PUNCTUATION.includes(value[end - 1])) end--; + return end; +} + +function isClearProseBoundaryToken(value: string, start: number, end: number): boolean { + while (start < end && LEADING_PATH_PUNCTUATION.includes(value[start])) start++; + end = trimPathSpanEnd(value, start, end); + return (CLEAR_PROSE_BOUNDARIES as readonly string[]).includes( + value.slice(start, end).toLowerCase() + ); +} + +function findUnquotedPathEnd( + value: string, + start: number, + acceptFirstTokenPunctuation: boolean, + acceptEndpointBeforeAnotherAbsolute: boolean, + failClosedAmbiguity: boolean +): number { + let tokenStart = start; + let isFirstToken = true; + let firstTokenEnd = -1; + let firstTrimmedTokenEnd = -1; + let lastPathTokenEnd = -1; + let resolvedExtensionEnd = -1; + let hasFilesystemEvidence = false; + let hasUnresolvedFragments = false; + + const resolveEndpoint = (): number => { + if (hasUnresolvedFragments) { + return failClosedAmbiguity || hasFilesystemEvidence ? value.length : -1; + } + if (resolvedExtensionEnd >= 0) return resolvedExtensionEnd; + if (hasFilesystemEvidence && lastPathTokenEnd >= 0) return lastPathTokenEnd; + if ( + acceptFirstTokenPunctuation && + firstTrimmedTokenEnd >= 0 && + firstTrimmedTokenEnd < firstTokenEnd + ) { + return firstTrimmedTokenEnd; + } + return -1; + }; + + while (tokenStart < value.length) { + const tokenEnd = findTokenEnd(value, tokenStart); + const extensionEnd = findExtensionEndInToken(value, tokenStart, tokenEnd); + const trimmedTokenEnd = trimPathSpanEnd(value, tokenStart, tokenEnd); + + if (isFirstToken) { + firstTokenEnd = tokenEnd; + firstTrimmedTokenEnd = trimmedTokenEnd; + lastPathTokenEnd = trimmedTokenEnd; + // A prose-looking token may itself be a directory name. It is a safe + // boundary only when no later token carries path-separator evidence; + // otherwise keep scanning so a filesystem suffix cannot survive. + } else if ( + isClearProseBoundaryToken(value, tokenStart, tokenEnd) && + (!remainderContainsFilesystemSeparator(value, tokenEnd) || + (!failClosedAmbiguity && !hasFilesystemEvidence)) + ) { + return resolveEndpoint(); + } + + const containsSeparator = tokenContainsPathSeparator(value, tokenStart, tokenEnd); + const containsExtensionEvidence = tokenContainsPathExtensionEvidence( + value, + tokenStart, + tokenEnd + ); + if (containsSeparator) { + lastPathTokenEnd = trimmedTokenEnd; + hasFilesystemEvidence = true; + hasUnresolvedFragments = false; + resolvedExtensionEnd = extensionEnd >= 0 ? extensionEnd : -1; + if (extensionEnd < 0 && containsExtensionEvidence) { + resolvedExtensionEnd = trimmedTokenEnd; + } + } else if (extensionEnd >= 0) { + resolvedExtensionEnd = extensionEnd; + hasFilesystemEvidence = true; + hasUnresolvedFragments = false; + } else if (containsExtensionEvidence) { + resolvedExtensionEnd = trimmedTokenEnd; + hasFilesystemEvidence = true; + hasUnresolvedFragments = false; + } else if (!isFirstToken) { + hasUnresolvedFragments = true; + } + + let nextTokenStart = tokenEnd; + while (nextTokenStart < value.length && isWhitespace(value[nextTokenStart])) nextTokenStart++; + if (nextTokenStart >= value.length) return resolveEndpoint(); + if (isSyntacticallyAbsolutePathAt(value, nextTokenStart)) { + const endpoint = resolveEndpoint(); + if (endpoint >= 0) return endpoint; + return acceptEndpointBeforeAnotherAbsolute ? lastPathTokenEnd : -1; + } + + tokenStart = nextTokenStart; + isFirstToken = false; + } + return resolveEndpoint(); +} + +function isUnquotedPosixSpanCandidateAt(value: string, start: number): boolean { + const tokenEnd = findTokenEnd(value, start); + const token = value.slice(start, tokenEnd); + if (isKnownPosixFilesystemPath(token)) return true; + if ( + findExtensionEndInToken(value, start, tokenEnd) >= 0 || + tokenContainsPathExtensionEvidence(value, start, tokenEnd) + ) { + return true; + } + + let slashCount = 0; + for (let index = start; index < tokenEnd; index++) { + if (value.charCodeAt(index) === 0x2f) slashCount++; + } + // Any boundary-delimited absolute POSIX token is filesystem-sensitive by + // default. Explicit Route/HTTP context is shielded by the caller before this + // candidate check, so `/vault` is redacted while `Route /vault` is retained. + return slashCount >= 1 && token.length > 1; +} + +function redactUnquotedAbsolutePathSpans(value: string): string { + const parts: string[] = []; + let copyStart = 0; + let index = 0; + + while (index < value.length) { + const previous = index > 0 ? value[index - 1] : ""; + const followsQuote = previous === "'" || previous === '"' || previous === "`"; + const hasCommonBoundary = + index === 0 || + isWhitespace(previous) || + LEADING_PATH_PUNCTUATION.includes(previous) || + previous === "=" || + previous === ":" || + previous === "," || + previous === ";" || + previous === "." || + previous === ">" || + previous === "|"; + const startsForwardSlashUnc = + value.charCodeAt(index) === 0x2f && value.charCodeAt(index + 1) === 0x2f; + const startsHttpUrl = + startsForwardSlashUnc && previous === ":" && hasHttpUrlSchemeBefore(value, index); + const isWindowsPath = + !followsQuote && + (isWindowsAbsolutePathAt(value, index) || isWindowsRootRelativePathAt(value, index)) && + !startsHttpUrl; + const isFileUriPath = !followsQuote && hasAbsoluteFileUriAt(value, index); + const isPosixPath = + !followsQuote && + value.charCodeAt(index) === 0x2f && + value.charCodeAt(index + 1) !== 0x2f && + !hasRouteContextBefore(value, index) && + isUnquotedPosixSpanCandidateAt(value, index); + const hasBoundary = hasCommonBoundary || (isWindowsPath && previous === ":"); + if (!hasBoundary || (!isWindowsPath && !isFileUriPath && !isPosixPath)) { + index++; + continue; + } + + // Whitespace makes an unquoted path ambiguous. Extend through adjacent + // separator-bearing tokens or to a deterministic filename extension. + // Unequivocal Windows, file-URI, and known-root candidates fail closed; + // arbitrary extensionless POSIX text falls back to token-level handling so + // ordinary `/x/y` route text is not redacted indiscriminately. + const isKnownPosixPath = isKnownPosixFilesystemPathAt(value, index); + const pathEnd = findUnquotedPathEnd( + value, + index, + isWindowsPath || isFileUriPath || isKnownPosixPath, + isWindowsPath || isFileUriPath || isKnownPosixPath, + isWindowsPath || isFileUriPath || isKnownPosixPath + ); + if (pathEnd < 0) { + const mustFailClosed = isWindowsPath || isFileUriPath || isKnownPosixPath; + if (mustFailClosed) { + // An unequivocal filesystem prefix with an unknowable endpoint must + // fail closed over the rest of the first line rather than expose a + // suffix such as `Files\\secret` or `My Project`. + parts.push(value.slice(copyStart, index), ""); + copyStart = value.length; + index = value.length; + break; + } + index++; + continue; + } + parts.push(value.slice(copyStart, index), ""); + copyStart = pathEnd; + index = pathEnd; + } + + if (parts.length === 0) return value; + parts.push(value.slice(copyStart)); + return parts.join(""); +} + +function isPhysicalLineSeparator(code: number): boolean { + return code === 0x0a || code === 0x0d || code === 0x2028 || code === 0x2029; +} + +function serializedLineSeparatorLengthAt(value: string, start: number): number { + if (value.charCodeAt(start) !== 0x5c) return 0; + const marker = value[start + 1]?.toLowerCase(); + if (marker === "n" || marker === "r") return 2; + const unicodeMarker = value.slice(start + 1, start + 6).toLowerCase(); + return unicodeMarker === "u000a" || + unicodeMarker === "u000d" || + unicodeMarker === "u2028" || + unicodeMarker === "u2029" + ? 6 + : 0; +} + +function looksLikeRelativeStackLocation(token: string): boolean { + if (token.length < 6 || token.length > 2048) return false; + + const lastForwardSlash = token.lastIndexOf("/"); + const lastBackslash = token.lastIndexOf("\\"); + const lastSeparator = Math.max(lastForwardSlash, lastBackslash); + if (lastSeparator === token.length - 1) return false; + + const columnSeparator = token.lastIndexOf(":"); + const lineSeparator = token.lastIndexOf(":", columnSeparator - 1); + if (lineSeparator < 0 || !hasNumericLineColumnSuffix(token, lineSeparator)) return false; + const queryIndex = token.indexOf("?", lastSeparator + 1); + const fragmentIndex = token.indexOf("#", lastSeparator + 1); + const metadataIndexes = [queryIndex, fragmentIndex].filter( + (index) => index >= 0 && index < lineSeparator + ); + const extensionEnd = metadataIndexes.length > 0 ? Math.min(...metadataIndexes) : lineSeparator; + const dot = token.lastIndexOf(".", extensionEnd - 1); + if (dot <= lastSeparator || dot === extensionEnd - 1) return false; + const extension = token.slice(dot + 1, extensionEnd).toLowerCase(); + if (!(SOURCE_EXT as readonly string[]).includes(extension)) return false; + return true; +} + +function looksLikeUrlStackLocation(token: string): boolean { + if (token.length < 12 || token.length > 2048) return false; + const lower = token.toLowerCase(); + if (!lower.startsWith("http://") && !lower.startsWith("https://")) return false; + const columnSeparator = token.lastIndexOf(":"); + const lineSeparator = token.lastIndexOf(":", columnSeparator - 1); + return lineSeparator > 0 && hasNumericLineColumnSuffix(token, lineSeparator); +} + +function hasNumericLineColumnSuffix(value: string, separator: number): boolean { + if (value.charCodeAt(separator) !== 0x3a) return false; + let index = separator + 1; + if (!isAsciiDigit(value.charCodeAt(index))) return false; + while (index < value.length && isAsciiDigit(value.charCodeAt(index))) index++; + if (value.charCodeAt(index) !== 0x3a) return false; + + index++; + if (!isAsciiDigit(value.charCodeAt(index))) return false; + while (index < value.length && isAsciiDigit(value.charCodeAt(index))) index++; + return index === value.length; +} + +function isNodeModulePathCode(code: number): boolean { + return ( + isAsciiAlphaNumeric(code) || code === 0x2e || code === 0x2f || code === 0x5f || code === 0x2d + ); +} + +function looksLikeNodeStackLocation(token: string): boolean { + if (token.length < 10 || token.length > 2048 || !token.startsWith("node:")) return false; + const columnSeparator = token.lastIndexOf(":"); + const lineSeparator = token.lastIndexOf(":", columnSeparator - 1); + if (lineSeparator <= 5 || !hasNumericLineColumnSuffix(token, lineSeparator)) return false; + for (let index = 5; index < lineSeparator; index++) { + if (!isNodeModulePathCode(token.charCodeAt(index))) return false; + } + return true; +} + +function looksLikeEvalStackLocation(token: string): boolean { + return token.length <= 64 && token.startsWith("[eval]") && hasNumericLineColumnSuffix(token, 6); +} + +function isRecognizedStackPathAt(value: string, start: number): boolean { + if (hasAbsoluteFileUriAt(value, start)) return true; + const tokenEnd = trimPathSpanEnd(value, start, findTokenEnd(value, start)); + const token = value.slice(start, tokenEnd); + return ( + looksLikeAbsolutePath(token) || + looksLikeRelativeStackLocation(token) || + looksLikeUrlStackLocation(token) || + looksLikeNodeStackLocation(token) || + looksLikeEvalStackLocation(token) + ); +} + +function isStackFrameLabel(value: string, start: number, end: number): boolean { + const label = value.slice(start, end).trim(); + if (label.length === 0 || label.length > 256) return false; + if (!/^[A-Za-z_$<]/.test(label) || /[^A-Za-z0-9_$.[\]<>:/ -]/.test(label)) return false; + if (!/\s/.test(label)) return true; + return /^(?:async|new)\s+\S+$/.test(label) || /^\S+\s+\[as\s+\S+\]$/.test(label); +} + +function skipAsyncStackPrefix(value: string, start: number): number { + if (value.slice(start, start + 5) !== "async" || !isWhitespace(value[start + 5])) return start; + let locationStart = start + 6; + while (locationStart < value.length && isWhitespace(value[locationStart])) locationStart++; + return locationStart; +} + +function isAggregateIndexLocationAt(value: string, start: number): boolean { + if (value.slice(start, start + 5) !== "index" || !isWhitespace(value[start + 5])) return false; + let index = start + 6; + while (index < value.length && isWhitespace(value[index])) index++; + if (!isAsciiDigit(value.charCodeAt(index))) return false; + while (index < value.length && isAsciiDigit(value.charCodeAt(index))) index++; + while (index < value.length && isWhitespace(value[index])) index++; + return value.charCodeAt(index) === 0x29; +} + +function looksLikeStackFrameAt(value: string, atIndex: number, allowDirectPath: boolean): boolean { + if (value.slice(atIndex, atIndex + 2).toLowerCase() !== "at") return false; + let labelStart = atIndex + 2; + if (!isWhitespace(value[labelStart])) return false; + while (labelStart < value.length && isWhitespace(value[labelStart])) labelStart++; + labelStart = skipAsyncStackPrefix(value, labelStart); + if (allowDirectPath && isRecognizedStackPathAt(value, labelStart)) return true; + + const openParen = value.indexOf("(", labelStart); + if (openParen < 0 || openParen - labelStart > 256) return false; + let pathStart = openParen + 1; + while (pathStart < value.length && isWhitespace(value[pathStart])) pathStart++; + return ( + isStackFrameLabel(value, labelStart, openParen) && + (isRecognizedStackPathAt(value, pathStart) || + (allowDirectPath && isAggregateIndexLocationAt(value, pathStart))) + ); +} + +function looksLikeAtSignStackFrameAt(value: string, frameStart: number): boolean { + const tokenEnd = trimPathSpanEnd(value, frameStart, findTokenEnd(value, frameStart)); + const atSign = value.indexOf("@", frameStart); + if (atSign <= frameStart || atSign >= tokenEnd || atSign - frameStart > 256) return false; + return isStackFrameLabel(value, frameStart, atSign) && isRecognizedStackPathAt(value, atSign + 1); +} + +function findSerializedStackFrameStart(value: string): number { + for (let index = 0; index < value.length; index++) { + const separatorLength = serializedLineSeparatorLengthAt(value, index); + if (separatorLength === 0) continue; + let frameStart = index + separatorLength; + while (frameStart < value.length) { + while (frameStart < value.length && isWhitespace(value[frameStart])) frameStart++; + const adjacentSeparatorLength = serializedLineSeparatorLengthAt(value, frameStart); + if (adjacentSeparatorLength === 0) break; + frameStart += adjacentSeparatorLength; + } + if ( + looksLikeStackFrameAt(value, frameStart, true) || + looksLikeAtSignStackFrameAt(value, frameStart) + ) { + let separatorStart = index; + while (separatorStart > 0 && value.charCodeAt(separatorStart - 1) === 0x5c) { + separatorStart--; + } + return separatorStart; + } + } + return -1; +} + +function findInlineStackFrameStart(value: string): number { + let marker = value.indexOf(" at "); + while (marker >= 0) { + if (looksLikeStackFrameAt(value, marker + 1, false)) return marker; + marker = value.indexOf(" at ", marker + 4); + } + return -1; +} + +function findInlineAtSignStackFrameStart(value: string): number { + let frameStart = 0; + while (frameStart < value.length) { + if (looksLikeAtSignStackFrameAt(value, frameStart)) { + return frameStart > 0 && isWhitespace(value[frameStart - 1]) ? frameStart - 1 : frameStart; + } + const tokenEnd = findTokenEnd(value, frameStart); + frameStart = tokenEnd; + while (frameStart < value.length && isWhitespace(value[frameStart])) frameStart++; + } + return -1; +} + +function physicalLineSeparatorLengthAt(value: string, start: number): number { + const code = value.charCodeAt(start); + if (!isPhysicalLineSeparator(code)) return 0; + return code === 0x0d && value.charCodeAt(start + 1) === 0x0a ? 2 : 1; +} + +function findPhysicalStackFrameStart(value: string): number { + for (let index = 0; index < value.length; index++) { + const separatorLength = physicalLineSeparatorLengthAt(value, index); + if (separatorLength === 0) continue; + let frameStart = index + separatorLength; + while (frameStart < value.length && isWhitespace(value[frameStart])) frameStart++; + if ( + looksLikeStackFrameAt(value, frameStart, true) || + looksLikeAtSignStackFrameAt(value, frameStart) + ) { + return index; + } + index += separatorLength - 1; + } + return -1; +} + +/** Strip only recognized physical, serialized, and inline JavaScript stack-frame tails. */ +export function stripRecognizedErrorStackTail(value: string): string { + const candidates = [ + findPhysicalStackFrameStart(value), + findSerializedStackFrameStart(value), + findInlineStackFrameStart(value), + findInlineAtSignStackFrameStart(value), + ].filter((candidate) => candidate >= 0); + if (candidates.length === 0) return value; + return value.slice(0, Math.min(...candidates)); +} + +/** + * Public exception messages remain fail-closed at the first physical line. + * Provider passthroughs that require multiline capability wording use the + * narrower recognized-frame helper above instead. + */ +export function stripErrorStackTail(value: string): string { + let firstLineEnd = value.length; + for (let index = 0; index < value.length; index++) { + if (isPhysicalLineSeparator(value.charCodeAt(index))) { + firstLineEnd = index; + break; + } + } + return stripRecognizedErrorStackTail(value.slice(0, firstLineEnd)); +} + +/** + * Redact absolute filesystem paths while preserving URLs, explicitly marked + * API routes, and punctuation around determinable endpoints. Unequivocal + * filesystem prefixes fail closed when an unquoted endpoint is ambiguous. + */ +export function redactErrorPaths(value: string): string { + const quotedPathsRedacted = redactQuotedAbsolutePaths(value); + const pathSpansRedacted = redactUnquotedAbsolutePathSpans(quotedPathsRedacted); + const parts = pathSpansRedacted.split(/(\s+)/); + let previousToken = ""; + for (let index = 0; index < parts.length; index++) { + const token = parts[index]; + if (isWhitespace(token)) continue; + parts[index] = redactAbsolutePathToken(token, isRouteContextToken(previousToken)); + previousToken = token; + } + return parts.join(""); +} diff --git a/open-sse/utils/errorSanitization.ts b/open-sse/utils/errorSanitization.ts new file mode 100644 index 0000000000..1b7601d4ee --- /dev/null +++ b/open-sse/utils/errorSanitization.ts @@ -0,0 +1,895 @@ +import { + redactErrorPaths, + stripErrorStackTail, + stripRecognizedErrorStackTail, +} from "./errorPathRedaction.ts"; +import { CREDENTIAL_PATTERNS } from "./credentialPatterns.ts"; + +// Length cap protects against pathological inputs even before tokenization. +const MAX_ERROR_LEN = 4096; +const MAX_ERROR_SCAN_HEADROOM = 512; +const MAX_SECURITY_ESCAPE_LAYERS = 3; +const STRONG_CREDENTIAL_TOKEN_SOURCE = + "(?:eyJ[A-Za-z0-9_-]{5,}\\.[A-Za-z0-9_-]{8,}\\.[A-Za-z0-9_-]{8,}|" + + "github_pat_[A-Za-z0-9_]{20,}|ghp_[A-Za-z0-9]{20,}|glpat-[A-Za-z0-9_-]{20,}|" + + "xox[a-z]-[A-Za-z0-9-]{10,}|(?:AKIA|ASIA)[A-Z0-9]{16}|" + + "(?= 0x30 && code <= 0x39) || + (code >= 0x41 && code <= 0x5a) || + (code >= 0x61 && code <= 0x7a) + ); +} + +function asciiHexValue(code: number): number { + if (code >= 0x30 && code <= 0x39) return code - 0x30; + if (code >= 0x41 && code <= 0x46) return code - 0x41 + 10; + if (code >= 0x61 && code <= 0x66) return code - 0x61 + 10; + return -1; +} + +function unicodeEscapeCodeAt(value: string, start: number): number | null { + if ( + value.charCodeAt(start) !== 0x5c || + (value[start + 1] !== "u" && value[start + 1] !== "U") || + start + 5 >= value.length + ) { + return null; + } + + let decoded = 0; + for (let digit = start + 2; digit <= start + 5; digit++) { + const nibble = asciiHexValue(value.charCodeAt(digit)); + if (nibble < 0) return null; + decoded = decoded * 16 + nibble; + } + return decoded; +} + +function isPrintableAscii(code: number | null): code is number { + return code !== null && code >= 0x20 && code <= 0x7e; +} + +function isSecurityWhitespaceCode(code: number | null): boolean { + return code === 0x08 || code === 0x09 || code === 0x0a || code === 0x0c || code === 0x0d; +} + +function isEscapeTokenBoundary(code: number): boolean { + return !isAsciiAlphaNumericCode(code) && code !== 0x2e && code !== 0x5f && code !== 0x2d; +} + +function shouldPreserveUnicodeUncEvidence( + value: string, + runStart: number, + runEnd: number, + decoded: number +): boolean { + if ( + runEnd - runStart < 2 || + decoded === 0x2f || + decoded === 0x5c || + decoded === 0x3a || + (runStart > 0 && !isEscapeTokenBoundary(value.charCodeAt(runStart - 1))) + ) { + return false; + } + + const afterEscape = runEnd + 5; + let tokenEnd = afterEscape; + while (tokenEnd < value.length && !/\s/.test(value[tokenEnd])) tokenEnd++; + if (value.slice(afterEscape, tokenEnd).includes("=")) return false; + return afterEscape < tokenEnd; +} + +function decodeSecurityEscapesOnce( + value: string, + decodeQuotes: boolean, + maxLength: number +): string { + const output: string[] = []; + let changed = false; + + for (let index = 0; index < value.length; index++) { + if (value.charCodeAt(index) !== 0x5c) { + output.push(value[index]); + continue; + } + + const runStart = index; + while (index < value.length && value.charCodeAt(index) === 0x5c) index++; + const runEnd = index; + if (runEnd >= value.length) { + output.push(value.slice(runStart)); + break; + } + + const escaped = value[runEnd]; + if (escaped === "u" || escaped === "U") { + const decoded = unicodeEscapeCodeAt(value, runEnd - 1); + const isQuote = decoded === 0x22 || decoded === 0x27; + if (isSecurityWhitespaceCode(decoded)) { + output.push(" "); + index = runEnd + 4; + changed = true; + continue; + } + if ( + isPrintableAscii(decoded) && + (decodeQuotes || !isQuote) && + !shouldPreserveUnicodeUncEvidence(value, runStart, runEnd, decoded) + ) { + output.push(String.fromCharCode(decoded)); + index = runEnd + 4; + changed = true; + continue; + } + output.push(value.slice(runStart, runEnd + 5)); + index = runEnd + 4; + continue; + } + + if ( + escaped === "b" || + escaped === "f" || + escaped === "n" || + escaped === "r" || + escaped === "t" + ) { + output.push(" "); + index = runEnd; + changed = true; + continue; + } + + if (escaped === "/" || (decodeQuotes && (escaped === '"' || escaped === "'"))) { + output.push(escaped); + index = runEnd; + changed = true; + continue; + } + + output.push(value.slice(runStart, runEnd)); + index = runEnd - 1; + } + + return changed ? output.join("").slice(0, maxLength) : value; +} + +function hasResidualSecurityEscape(value: string): boolean { + for (let index = 0; index < value.length; index++) { + if (value.charCodeAt(index) !== 0x5c) continue; + while (index < value.length && value.charCodeAt(index) === 0x5c) index++; + if (index >= value.length) return false; + const escaped = value[index]; + if ( + escaped === "b" || + escaped === "f" || + escaped === "n" || + escaped === "r" || + escaped === "t" + ) { + return true; + } + if (escaped === "/" || escaped === '"' || escaped === "'") return true; + if (escaped === "u" || escaped === "U") { + const decoded = unicodeEscapeCodeAt(value, index - 1); + if (isPrintableAscii(decoded) || isSecurityWhitespaceCode(decoded)) return true; + } + } + return false; +} + +/** Decode bounded security ASCII/JSON escapes while never materializing arbitrary Unicode. */ +function normalizeSecurityEscapes( + value: string, + decodeQuotes: boolean, + maxLength = MAX_ERROR_LEN +): string { + let normalized = value.slice(0, maxLength); + for (let layer = 0; layer < MAX_SECURITY_ESCAPE_LAYERS; layer++) { + const decoded = decodeSecurityEscapesOnce(normalized, decodeQuotes, maxLength); + if (decoded === normalized) break; + normalized = decoded.slice(0, maxLength); + } + return normalized; +} + +function isCredentialLabelBoundary(code: number): boolean { + return !isAsciiAlphaNumericCode(code) && code !== 0x5f && code !== 0x2d; +} + +function matchCredentialAssignmentAt(value: string, start: number): CredentialAssignment | null { + const keyQuote = value[start] === '"' || value[start] === "'" ? value[start] : ""; + const labelStart = start + (keyQuote ? 1 : 0); + const cliFlag = + !keyQuote && + labelStart >= 2 && + value.slice(labelStart - 2, labelStart) === "--" && + (labelStart === 2 || isCredentialLabelBoundary(value.charCodeAt(labelStart - 3))); + + for (const [label, failClosed] of CREDENTIAL_LABELS) { + const labelEnd = labelStart + label.length; + if (value.slice(labelStart, labelEnd).toLowerCase() !== label) continue; + let index = labelEnd; + if ( + (label === "arena-auth-prod-v1" || label === "__secure-next-auth.session-token") && + value[index] === "." + ) { + const chunkStart = ++index; + while (index < value.length && /\d/.test(value[index])) index++; + if (index === chunkStart) continue; + } + if (keyQuote) { + if (value[index] !== keyQuote) continue; + index++; + } else if (!isCredentialLabelBoundary(value.charCodeAt(index))) { + continue; + } else if (value[index] === '"' || value[index] === "'") { + index++; + } + const separatorStart = index; + while (/\s/.test(value[index])) index++; + if (value[index] === ":" || value[index] === "=") { + index++; + while (/\s/.test(value[index])) index++; + } else if (!(cliFlag && index > separatorStart)) { + continue; + } + return { valueStart: index, failClosed }; + } + return null; +} + +function findQuotedCredentialEnd(value: string, start: number, quote: string): number { + let index = start + 1; + while (index < value.length) { + if (value.charCodeAt(index) === 0x5c) { + index += 2; + continue; + } + if (value[index] === quote) return index; + index++; + } + return -1; +} + +function findUnquotedCredentialEnd(value: string, start: number): number { + let end = start; + while (end < value.length) { + const char = value[end]; + if (/\s/.test(char) || char === '"' || char === "'" || char === "," || char === "}") break; + end++; + } + return end; +} + +function redactLabeledCredentialAssignments(value: string): string { + const parts: string[] = []; + let copyStart = 0; + let index = 0; + + while (index < value.length) { + const assignment = matchCredentialAssignmentAt(value, index); + if (!assignment) { + index++; + continue; + } + + const { valueStart, failClosed } = assignment; + const quote = value[valueStart] === '"' || value[valueStart] === "'" ? value[valueStart] : ""; + if (quote) { + const closingQuote = findQuotedCredentialEnd(value, valueStart, quote); + parts.push(value.slice(copyStart, valueStart + 1), "[REDACTED]"); + if (closingQuote < 0) { + copyStart = value.length; + index = value.length; + } else { + parts.push(quote); + copyStart = closingQuote + 1; + index = copyStart; + } + continue; + } + + // A leading backslash may be a serialized quote or another encoded + // delimiter. Do not redact only that prefix and leave the value behind. + const valueEnd = + failClosed || value.charCodeAt(valueStart) === 0x5c + ? value.length + : findUnquotedCredentialEnd(value, valueStart); + parts.push(value.slice(copyStart, valueStart), "[REDACTED]"); + copyStart = valueEnd; + index = Math.max(valueEnd, valueStart + 1); + } + + if (parts.length === 0) return value; + parts.push(value.slice(copyStart)); + return parts.join(""); +} + +function redactPrivateKeyPemBlocks(value: string): string { + // ASCII-only fold keeps offsets aligned even when the surrounding message + // contains Unicode characters whose full uppercase form expands in length. + const upperValue = value.replace(/[a-z]/g, (char) => char.toUpperCase()); + const beginPrefix = "-----BEGIN "; + const parts: string[] = []; + let copyStart = 0; + let searchStart = 0; + + while (searchStart < value.length) { + const blockStart = upperValue.indexOf(beginPrefix, searchStart); + if (blockStart < 0) break; + const labelStart = blockStart + beginPrefix.length; + const headerEnd = upperValue.indexOf("-----", labelStart); + if (headerEnd < 0) break; + const label = upperValue.slice(labelStart, headerEnd).trim(); + if (!/^(?:[A-Z0-9]+ )*PRIVATE KEY(?: BLOCK)?$/.test(label)) { + searchStart = headerEnd + 5; + continue; + } + + const endMarker = `-----END ${label}-----`; + const closingStart = upperValue.indexOf(endMarker, headerEnd + 5); + const blockEnd = closingStart < 0 ? value.length : closingStart + endMarker.length; + parts.push(value.slice(copyStart, blockStart), "[REDACTED]"); + copyStart = blockEnd; + searchStart = blockEnd; + } + + if (parts.length === 0) return value; + parts.push(value.slice(copyStart)); + return parts.join(""); +} + +const DATA_URL_PREFIX = "data:"; +const BASE64_DATA_URL_MARKER = ";base64"; +const REDACTED_DATA_URL = "[REDACTED_DATA_URL]"; + +function matchesAsciiCaseInsensitiveAt(value: string, start: number, expected: string): boolean { + if (start < 0 || start + expected.length > value.length) return false; + for (let offset = 0; offset < expected.length; offset++) { + const code = value.charCodeAt(start + offset); + const foldedCode = code >= 0x41 && code <= 0x5a ? code + 0x20 : code; + if (foldedCode !== expected.charCodeAt(offset)) return false; + } + return true; +} + +function isBase64DataUrlPayloadCode(code: number): boolean { + return ( + isAsciiAlphaNumericCode(code) || + code === 0x2b || + code === 0x2f || + code === 0x3d || + code === 0x5f || + code === 0x2d + ); +} + +function isEcmaScriptWhitespaceCode(code: number): boolean { + return ( + (code >= 0x09 && code <= 0x0d) || + code === 0x20 || + code === 0xa0 || + code === 0x1680 || + (code >= 0x2000 && code <= 0x200a) || + code === 0x2028 || + code === 0x2029 || + code === 0x202f || + code === 0x205f || + code === 0x3000 || + code === 0xfeff + ); +} + +/** Redact base64 data URLs in one pass, including input with many repeated `data:` prefixes. */ +function redactBase64DataUrls(value: string): string { + const parts: string[] = []; + let copyStart = 0; + let index = 0; + + while (index < value.length) { + if (!matchesAsciiCaseInsensitiveAt(value, index, DATA_URL_PREFIX)) { + index++; + continue; + } + + const dataUrlStart = index; + const mediaTypeStart = dataUrlStart + DATA_URL_PREFIX.length; + let delimiter = mediaTypeStart; + while ( + delimiter < value.length && + value[delimiter] !== "," && + !isEcmaScriptWhitespaceCode(value.charCodeAt(delimiter)) + ) { + delimiter++; + } + + const markerStart = delimiter - BASE64_DATA_URL_MARKER.length; + const hasBase64Marker = + delimiter < value.length && + value[delimiter] === "," && + markerStart >= mediaTypeStart && + matchesAsciiCaseInsensitiveAt(value, markerStart, BASE64_DATA_URL_MARKER); + if (!hasBase64Marker) { + index = delimiter < value.length ? delimiter + 1 : value.length; + continue; + } + + let payloadEnd = delimiter + 1; + while (payloadEnd < value.length && isBase64DataUrlPayloadCode(value.charCodeAt(payloadEnd))) { + payloadEnd++; + } + if (payloadEnd === delimiter + 1) { + index = delimiter + 1; + continue; + } + + parts.push(value.slice(copyStart, dataUrlStart), REDACTED_DATA_URL); + copyStart = payloadEnd; + index = payloadEnd; + } + + if (parts.length === 0) return value; + parts.push(value.slice(copyStart)); + return parts.join(""); +} + +const HTTP_URL_RE = /https?:\/\//gi; +const URL_QUERY_PARAM_RE = /([?&])([^=&#]+)=([^&#]*)/g; + +function isUrlTerminator(char: string): boolean { + return ( + /\s/.test(char) || + char === '"' || + char === "'" || + char === "`" || + char === "<" || + char === ">" || + char === ")" || + char === "]" || + char === "}" || + char === "," || + char === ";" + ); +} + +function normalizeUrlQueryKey(key: string): string { + let decoded = key.replace(/\+/g, " "); + try { + decoded = decodeURIComponent(decoded); + } catch { + // Malformed percent escapes stay visible to the conservative ASCII fold. + } + return decoded.replace(/[^A-Za-z0-9]/g, "").toLowerCase(); +} + +function isSensitiveUrlQueryKey(key: string): boolean { + const normalized = normalizeUrlQueryKey(key); + return ( + normalized === "sig" || + normalized === "signature" || + normalized === "key" || + normalized === "apikey" || + normalized === "token" || + normalized === "accesstoken" || + normalized === "refreshtoken" || + normalized === "credential" || + normalized === "password" || + normalized === "secret" || + normalized === "awsaccesskeyid" || + normalized === "googleaccessid" || + normalized === "xamzcredential" || + normalized === "xamzsignature" || + normalized === "xamzsecuritytoken" || + normalized === "xgoogcredential" || + normalized === "xgoogsignature" + ); +} + +function redactUrlSegment(segment: string): string { + const schemeEnd = segment.indexOf("//") + 2; + let authorityEnd = segment.length; + for (const delimiter of ["/", "?", "#"]) { + const candidate = segment.indexOf(delimiter, schemeEnd); + if (candidate >= 0) authorityEnd = Math.min(authorityEnd, candidate); + } + + let redacted = segment; + const userInfoEnd = segment.lastIndexOf("@", authorityEnd); + if (userInfoEnd >= schemeEnd) { + redacted = `${segment.slice(0, schemeEnd)}[REDACTED]@${segment.slice(userInfoEnd + 1)}`; + } + + URL_QUERY_PARAM_RE.lastIndex = 0; + return redacted.replace(URL_QUERY_PARAM_RE, (match, separator: string, key: string) => + isSensitiveUrlQueryKey(key) ? `${separator}redacted=[REDACTED]` : match + ); +} + +function redactSensitiveUrlCredentials(value: string): string { + HTTP_URL_RE.lastIndex = 0; + const parts: string[] = []; + let copyStart = 0; + let match = HTTP_URL_RE.exec(value); + while (match) { + const start = match.index; + let end = HTTP_URL_RE.lastIndex; + while (end < value.length && !isUrlTerminator(value[end])) end++; + const segment = value.slice(start, end); + const redacted = redactUrlSegment(segment); + if (redacted !== segment) { + parts.push(value.slice(copyStart, start), redacted); + copyStart = end; + } + HTTP_URL_RE.lastIndex = Math.max(end, HTTP_URL_RE.lastIndex); + match = HTTP_URL_RE.exec(value); + } + if (parts.length === 0) return value; + parts.push(value.slice(copyStart)); + return parts.join(""); +} + +function redactKnownCredentialPatterns(value: string): string { + let redacted = value; + for (const pattern of CREDENTIAL_PATTERNS) { + if (pattern.name === "auth_header") continue; + pattern.regex.lastIndex = 0; + redacted = redacted.replace(pattern.regex, "[REDACTED]"); + } + return redacted; +} + +export function redactSensitiveErrorText(value: string): string { + const normalized = normalizeSecurityEscapes( + value, + false, + MAX_ERROR_LEN + MAX_ERROR_SCAN_HEADROOM + ); + const catalogRedacted = redactKnownCredentialPatterns(redactSensitiveUrlCredentials(normalized)); + const commonCredentialsRedacted = redactBase64DataUrls(redactPrivateKeyPemBlocks(catalogRedacted)) + .replace(/\b(Bearer|Basic)\s+[A-Za-z0-9._~+/=-]+/gi, "$1 [REDACTED]") + .replace(STRONG_CREDENTIAL_TOKEN_GLOBAL, "[REDACTED]"); + return redactLabeledCredentialAssignments(commonCredentialsRedacted); +} + +export function containsSensitiveErrorCredential(value: string): boolean { + const normalized = normalizeSecurityEscapes( + value, + false, + MAX_ERROR_LEN + MAX_ERROR_SCAN_HEADROOM + ); + const directRedacted = redactKnownCredentialPatterns(redactSensitiveUrlCredentials(normalized)) + .replace(/\b(Bearer|Basic)\s+[A-Za-z0-9._~+/=-]+/gi, "$1 [REDACTED]") + .replace(STRONG_CREDENTIAL_TOKEN_GLOBAL, "[REDACTED]"); + if (directRedacted !== normalized) return true; + if ( + /(?:^|\s)--(?:api[-_]?key|token|password|secret)\s+(?:"[^"]*"|'[^']*'|\S+)/i.test(normalized) + ) { + return true; + } + return /(?:api[_-]?key|access[_-]?token|refresh[_-]?token|authorization|cookie|secret)["']?\s*[:=]\s*["']?[^"'\\,\s}]{6,}/i.test( + normalized + ); +} + +function coerceErrorText(value: unknown): string { + if (typeof value === "string") return value; + if (value === null || value === undefined) return ""; + try { + return String(value); + } catch { + // Fail closed when an attacker-controlled toString/valueOf accessor throws. + return ""; + } +} + +function truncateSanitizedErrorText(value: string): string { + if (value.length <= MAX_ERROR_LEN) return value; + const markerStart = value.lastIndexOf("[REDACTED", MAX_ERROR_LEN); + const markerEnd = markerStart >= 0 ? value.indexOf("]", markerStart) : -1; + if ( + markerStart >= 0 && + markerStart < MAX_ERROR_LEN && + markerEnd >= MAX_ERROR_LEN && + markerEnd - markerStart <= 128 + ) { + const marker = value.slice(markerStart, markerEnd + 1); + return `${value.slice(0, MAX_ERROR_LEN - marker.length)}${marker}`; + } + return value.slice(0, MAX_ERROR_LEN); +} + +/** + * Strip stack-trace tails, credentials, and absolute source paths from a + * client-visible error message. + */ +function sanitizeErrorMessageWithStackPolicy( + message: unknown, + stripStackTail: (value: string) => string +): string { + let str = coerceErrorText(message); + if (str.length > MAX_ERROR_LEN + MAX_ERROR_SCAN_HEADROOM) { + str = str.slice(0, MAX_ERROR_LEN + MAX_ERROR_SCAN_HEADROOM); + } + // Preserve quote provenance until hidden labels/delimiters have been + // exposed and redacted, then decode safe quote escapes in the clean text. + // Raw URI credentials must be projected before the path tokenizer consumes + // the URI tail; Windows path evidence still stays intact until after this + // credential-only pass and is redacted before escape normalization. + str = redactKnownCredentialPatterns(redactSensitiveUrlCredentials(stripStackTail(str))); + str = redactErrorPaths(str); + str = redactSensitiveErrorText(str); + str = truncateSanitizedErrorText(str); + str = normalizeSecurityEscapes(str, false); + str = redactSensitiveErrorText(redactErrorPaths(stripStackTail(str))); + str = normalizeSecurityEscapes(str, true); + str = redactSensitiveErrorText(redactErrorPaths(stripStackTail(str))); + return hasResidualSecurityEscape(str) ? "[REDACTED]" : str.trimEnd(); +} + +export function sanitizeErrorMessage(message: unknown): string { + return sanitizeErrorMessageWithStackPolicy(message, stripErrorStackTail); +} + +function sanitizePassthroughErrorMessage(message: unknown): string { + return sanitizeErrorMessageWithStackPolicy(message, stripRecognizedErrorStackTail); +} + +const BLOCKED_KEYS = + /stack|trace|path|file|cwd|dir|password|secret|token|key|authorization|cookie|credential|session(?!_?(?:count|status)$)/i; +const BLOCKED_CREDENTIAL_ALIAS_KEYS = + /^(?:cf_clearance|__cf_bm|_cfuvid|_puid|sso|sso-rw|arena-auth-prod-v1(?:\.\d+)?)$/i; +const PROTOTYPE_CONTROL_KEYS = new Set(["__proto__", "constructor", "prototype"]); +const MAX_DEPTH = 4; +const MAX_UPSTREAM_KEY_LEN = 256; +type UpstreamClassificationKey = "code" | "reason" | "status" | "type"; +const SAFE_UPSTREAM_STATUS_IDENTIFIERS = new Set([ + "ABORTED", + "ALREADY_EXISTS", + "CANCELLED", + "DATA_LOSS", + "DEADLINE_EXCEEDED", + "FAILED_PRECONDITION", + "INTERNAL", + "INVALID_ARGUMENT", + "NOT_FOUND", + "OK", + "OUT_OF_RANGE", + "PERMISSION_DENIED", + "RESOURCE_EXHAUSTED", + "UNAUTHENTICATED", + "UNAVAILABLE", + "UNIMPLEMENTED", + "UNKNOWN", +]); +const SAFE_UPSTREAM_ERROR_IDENTIFIERS = new Set([ + "api_error", + "auth_error", + "authentication_error", + "bad_gateway", + "bad_request", + "billing_error", + "context_length_exceeded", + "error", + "gateway_timeout", + "insufficient_quota", + "invalid_api_key", + "invalid_request", + "invalid_request_error", + "model_not_found", + "not_found", + "payment_required", + "permission_error", + "provider_error", + "quota_exhausted", + "rate_limit_error", + "rate_limit_exceeded", + "server_error", + "upstream_error", + "upstream_timeout", +]); + +function describeOpaqueBinaryDetail(value: ArrayBuffer | ArrayBufferView): string { + return `[binary ${value.byteLength} bytes]`; +} + +function normalizeUpstreamClassificationKey(key: string): UpstreamClassificationKey | null { + const normalized = key.replace(/[-_]/g, "").toLowerCase(); + if (normalized === "code" || normalized === "errorcode") return "code"; + if (normalized === "reason" || normalized === "errorreason") return "reason"; + if ( + normalized === "status" || + normalized === "statuscode" || + normalized === "errorstatus" || + normalized === "errorstatuscode" + ) { + return "status"; + } + if (normalized === "type" || normalized === "errortype" || normalized === "subtype") { + return "type"; + } + return null; +} + +function projectUpstreamErrorIdentifier(key: UpstreamClassificationKey, value: unknown): unknown { + if (typeof value === "number") { + if (!Number.isInteger(value)) return undefined; + if (key === "code" && value >= 0 && value <= 16) return value; + return (key === "code" || key === "status") && value >= 100 && value <= 599 ? value : undefined; + } + if (typeof value !== "string") return undefined; + if (key === "status" && SAFE_UPSTREAM_STATUS_IDENTIFIERS.has(value.toUpperCase())) { + return value; + } + if ( + /^[1-5]\d{2}$/.test(value) || + /^HTTP_[1-5]\d{2}$/i.test(value) || + SAFE_UPSTREAM_ERROR_IDENTIFIERS.has(value.toLowerCase()) + ) { + return value; + } + if (key === "type") return "upstream_error"; + if (key === "code") return ""; + return undefined; +} + +function isSafeUpstreamDetailKey(key: string): boolean { + if ( + key.length === 0 || + key.length > MAX_UPSTREAM_KEY_LEN || + BLOCKED_KEYS.test(key) || + BLOCKED_CREDENTIAL_ALIAS_KEYS.test(key) || + PROTOTYPE_CONTROL_KEYS.has(key.toLowerCase()) + ) { + return false; + } + return sanitizeErrorMessage(key) === key; +} + +/** + * Recursively sanitize an arbitrary JSON value from an upstream provider body. + * Unsafe keys are dropped rather than renamed so sanitized-key collisions + * cannot restore a secret under a public placeholder. + */ +function sanitizeUpstreamDetailsInternal( + value: unknown, + depth: number, + preserveSafeMultiline: boolean, + projectClassification: boolean +): unknown { + if (depth > MAX_DEPTH) return "[truncated]"; + if (value === null || value === undefined) return null; + if (typeof value === "string") { + return preserveSafeMultiline + ? sanitizePassthroughErrorMessage(value) + : sanitizeErrorMessage(value); + } + if (typeof value === "number" || typeof value === "boolean") return value; + if (typeof value === "object") { + try { + if (value instanceof ArrayBuffer || ArrayBuffer.isView(value)) { + return describeOpaqueBinaryDetail(value); + } + if (Array.isArray(value)) { + return value + .slice(0, 32) + .map((entry) => + sanitizeUpstreamDetailsInternal( + entry, + depth + 1, + preserveSafeMultiline, + projectClassification + ) + ); + } + const out = Object.create(null) as Record; + for (const [key, entryValue] of Object.entries(value as Record)) { + if (!isSafeUpstreamDetailKey(key)) continue; + const normalizedKey = key.toLowerCase(); + const classificationKey = normalizeUpstreamClassificationKey(normalizedKey); + if (projectClassification && classificationKey) { + const projected = projectUpstreamErrorIdentifier(classificationKey, entryValue); + if (projected !== undefined) out[key] = projected; + continue; + } + const childProjectsClassification = + normalizedKey === "error" || + normalizedKey === "errors" || + normalizedKey === "warning" || + normalizedKey === "warnings"; + out[key] = sanitizeUpstreamDetailsInternal( + entryValue, + depth + 1, + preserveSafeMultiline, + childProjectsClassification + ); + } + return out; + } catch { + return null; + } + } + return null; +} + +export function sanitizeUpstreamDetails(value: unknown, depth = 0): unknown { + return sanitizeUpstreamDetailsInternal(value, depth, false, depth === 0); +} + +/** Provider-only projection that preserves safe multiline capability wording. */ +export function sanitizePassthroughUpstreamDetails(value: unknown, depth = 0): unknown { + return sanitizeUpstreamDetailsInternal(value, depth, true, depth === 0); +} diff --git a/open-sse/utils/passthroughTailProcessor.ts b/open-sse/utils/passthroughTailProcessor.ts index ab45fb5474..3fa75e58c6 100644 --- a/open-sse/utils/passthroughTailProcessor.ts +++ b/open-sse/utils/passthroughTailProcessor.ts @@ -9,6 +9,7 @@ import { stripResponsesLifecycleEcho, } from "./responsesStreamHelpers.ts"; import { getAnyReasoningValue } from "./reasoningFields.ts"; +import { projectStreamFailureEvent, type StreamFailurePayload } from "./streamErrorFormat.ts"; type JsonRecord = Record; @@ -47,6 +48,7 @@ export type PassthroughTailProcessorContext = { hasPassthroughToolCalls: () => boolean; toResponsesCompletedWithToolCalls: (parsed: JsonRecord) => JsonRecord; restoreOpenAIToolNames: (parsed: JsonRecord) => boolean; + abortFailure: (failure: StreamFailurePayload, publicMessage: string) => void; }; function asRecord(value: unknown): JsonRecord { @@ -284,7 +286,13 @@ export function processBufferedPassthroughLine( context.updateClaudeEmptyResponseLifecycle(parsedPassthroughData); } - const parsed = parsedPassthroughData as JsonRecord; + const projectedFailure = projectStreamFailureEvent(parsedPassthroughData); + const parsed = projectedFailure + ? projectedFailure.publicPayload + : (parsedPassthroughData as JsonRecord); + if (projectedFailure) { + output = `data: ${JSON.stringify(parsed)}\n\n`; + } if (context.sanitizeUsagePayload(parsed)) { output = `data: ${JSON.stringify(parsed)}\n\n`; } @@ -301,6 +309,14 @@ export function processBufferedPassthroughLine( } context.pushClientPayload(parsed); + + output = context.passthroughEventPrefix.prefixData(output, line); + context.emitConvertedOutput(output); + if (projectedFailure) { + context.abortFailure(projectedFailure.internalFailure, projectedFailure.publicMessage); + return true; + } + return false; } output = context.passthroughEventPrefix.prefixData(output, line); diff --git a/open-sse/utils/responsesFailureOutput.ts b/open-sse/utils/responsesFailureOutput.ts new file mode 100644 index 0000000000..e085ba3b69 --- /dev/null +++ b/open-sse/utils/responsesFailureOutput.ts @@ -0,0 +1,70 @@ +type JsonRecord = Record; + +export type ResponsesFailureOutputStringField = "id" | "text" | "refusal"; + +export type ResponsesFailureOutputStringProjector = ( + field: ResponsesFailureOutputStringField, + value: string +) => string; + +function asRecord(value: unknown): JsonRecord { + return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : {}; +} + +/** + * Retain only public assistant text/refusal output from a failed Responses payload. + * Failure envelopes may contain reasoning, tool arguments, annotations, commentary, + * or provider diagnostics, so every retained field is reconstructed explicitly. + */ +export function projectResponsesFailureOutput( + value: unknown, + projectString: ResponsesFailureOutputStringProjector +): JsonRecord[] { + if (!Array.isArray(value)) return []; + + const output: JsonRecord[] = []; + for (const item of value) { + const record = asRecord(item); + if (record.type !== "message" || record.role !== "assistant" || record.phase === "commentary") { + continue; + } + + const content: JsonRecord[] = []; + if (Array.isArray(record.content)) { + for (const part of record.content) { + const contentPart = asRecord(part); + if (contentPart.phase === "commentary") continue; + if (contentPart.type === "output_text" && typeof contentPart.text === "string") { + content.push({ + type: "output_text", + text: projectString("text", contentPart.text), + // Preserve the required Responses schema without forwarding any + // untrusted citation/file metadata supplied by the provider. + annotations: [], + }); + } else if (contentPart.type === "refusal" && typeof contentPart.refusal === "string") { + content.push({ + type: "refusal", + refusal: projectString("refusal", contentPart.refusal), + }); + } + } + } + + const projected: JsonRecord = { + type: "message", + role: "assistant", + content, + }; + if (typeof record.id === "string") projected.id = projectString("id", record.id); + if ( + record.status === "in_progress" || + record.status === "completed" || + record.status === "incomplete" + ) { + projected.status = record.status; + } + output.push(projected); + } + return output; +} diff --git a/open-sse/utils/stream.ts b/open-sse/utils/stream.ts index d33cc8a526..f7b01b1464 100644 --- a/open-sse/utils/stream.ts +++ b/open-sse/utils/stream.ts @@ -50,9 +50,11 @@ import { parseTextualToolCallCandidate, isValidToolCallHeaderPrefix } from "./te import { stripObfuscationZeroWidth } from "./zeroWidth.ts"; import { formatTranslatedStreamError, - normalizeStreamFailurePayload, + prepareTranslatedStreamFailure, + projectStreamFailureEvent, type StreamFailurePayload, } from "./streamErrorFormat.ts"; +import { createStreamFailureAborter } from "./streamFailureBoundary.ts"; import { recordToolLatency } from "../services/toolLatencyTracker.ts"; import { extractToolSchemaMap } from "../translator/response/openai-responses/toolSchemas.ts"; import { @@ -178,8 +180,6 @@ type StreamOptions = { * codex-compatible `namespace` + `name` fields. */ requestToolIdentityMap?: Map | null; - /** High water mark for the TransformStream internal buffer (default: 16384) */ - highWaterMark?: number; }; type TranslateState = ReturnType & { @@ -1175,7 +1175,39 @@ export function createSSEStream(options: StreamOptions = {}) { } }; - const highWaterMark = options.highWaterMark ?? 16384; + const abortStreamFailure = createStreamFailureAborter({ + onFailure, + onComplete, + getUsage: () => state?.usage, + timing, + buildProviderPayload: () => + providerPayloadCollector.build(providerPayloadCollector.getSummary(), { + includeEvents: false, + }), + buildClientPayload: (body) => clientPayloadCollector.build(body, { includeEvents: false }), + clearIdleTimer, + clearPendingRequest: clearPendingRequestFromStream, + markPendingRequestCleared, + model, + }); + + const emitTranslatedFailureAndAbort = ( + controller: TransformStreamDefaultController, + payload: unknown + ): boolean => { + const failure = prepareTranslatedStreamFailure(payload); + if (!failure) return false; + providerPayloadCollector.push(failure.providerPayload); + const output = formatTranslatedStreamError(failure.record, sourceFormat); + reqLogger?.appendConvertedChunk?.(output); + forward(controller, encoder.encode(output)); + upstreamErrorForwarded = true; + doneSent = true; + abortStreamFailure(controller, failure.internalFailure, failure.publicMessage, { + notifyComplete: true, + }); + return true; + }; return new TransformStream( { @@ -1241,6 +1273,7 @@ export function createSSEStream(options: StreamOptions = {}) { let injectedUsage = false; let clientPayload: unknown = null; let failurePayload: StreamFailurePayload | null = null; + let publicFailureMessage: string | null = null; if (skipPassthroughEvent) { if (!trimmed) { @@ -1328,6 +1361,14 @@ export function createSSEStream(options: StreamOptions = {}) { if (trimmed.startsWith("data:") && trimmed.slice(5).trim() !== "[DONE]") { try { let parsed = parsedPassthroughData ?? JSON.parse(trimmed.slice(5).trim()); + const projectedFailure = projectStreamFailureEvent(parsed); + if (projectedFailure) { + parsed = projectedFailure.publicPayload; + failurePayload = projectedFailure.internalFailure; + publicFailureMessage = projectedFailure.publicMessage; + output = `data: ${JSON.stringify(parsed)}\n\n`; + injectedUsage = true; + } // Some upstream Responses-compatible providers leak an initial Chat Completions // bootstrap chunk (assistant role + empty content) before emitting proper @@ -1484,9 +1525,6 @@ export function createSSEStream(options: StreamOptions = {}) { ); } } - if (parsed.type === "response.failed") { - failurePayload = normalizeStreamFailurePayload(parsed); - } if ( parsed.type === "response.reasoning_summary_text.delta" || parsed.type === "response.reasoning_summary_text.done" || @@ -1810,20 +1848,22 @@ export function createSSEStream(options: StreamOptions = {}) { const rawDelta = parsed.choices?.[0]?.delta; const hadReasoningAlias = hasUnsupportedReasoningSignal(rawDelta); - parsed = sanitizeStreamingChunk(parsed); - if ( - parsed && - typeof parsed === "object" && - !Array.isArray(parsed) && - (parsed as Record)[OMIT_STREAMING_CHUNK_MARKER] === true - ) { - continue; + if (!projectedFailure) { + parsed = sanitizeStreamingChunk(parsed); + if ( + parsed && + typeof parsed === "object" && + !Array.isArray(parsed) && + (parsed as Record)[OMIT_STREAMING_CHUNK_MARKER] === true + ) { + continue; + } } const restoredOpenAIToolName = restoreOpenAIToolNames(parsed, toolNameMap); const idFixed = hadNonStringTopLevelId ? false : fixInvalidId(parsed); - if (!hasValuableContent(parsed, FORMATS.OPENAI)) { + if (!projectedFailure && !hasValuableContent(parsed, FORMATS.OPENAI)) { continue; } @@ -2052,20 +2092,10 @@ export function createSSEStream(options: StreamOptions = {}) { reqLogger?.appendConvertedChunk?.(output); forward(controller, encoder.encode(output)); if (failurePayload) { - let failureHandled = false; - if (onFailure) { - try { - failureHandled = onFailure(failurePayload) === true; - } catch (e) { - console.debug(`[STREAM] onFailure callback error:`, e); - } - } - clearIdleTimer(); - if (!failureHandled) { - clearPendingRequestFromStream(); - } - controller.error( - markPendingRequestCleared(new Error(failurePayload.message || "Upstream failure")) + abortStreamFailure( + controller, + failurePayload, + publicFailureMessage || "Upstream failure" ); return; } @@ -2087,14 +2117,7 @@ export function createSSEStream(options: StreamOptions = {}) { if (upstreamErrorForwarded) continue; - if (parsed.error) { - const output = formatTranslatedStreamError(parsed, sourceFormat); - reqLogger?.appendConvertedChunk?.(output); - forward(controller, encoder.encode(output)); - upstreamErrorForwarded = true; - doneSent = true; - continue; - } + if (emitTranslatedFailureAndAbort(controller, parsed)) return; // #5786 — drop replayed Responses-API events (identical/lower sequence_number // re-sent on an upstream reconnect) so their deltas are not glued twice into @@ -2356,6 +2379,8 @@ export function createSSEStream(options: StreamOptions = {}) { ]) as JsonRecord, restoreOpenAIToolNames: (parsed: JsonRecord) => restoreOpenAIToolNames(parsed, toolNameMap), + abortFailure: (failure: StreamFailurePayload, publicMessage: string) => + abortStreamFailure(controller, failure, publicMessage), }; for (const line of normalizedTailLines) { @@ -2369,12 +2394,18 @@ export function createSSEStream(options: StreamOptions = {}) { clearPendingPassthroughEvent(); } else if (buffer) { let output = buffer; + let bufferedProjectedFailure: ReturnType = null; if (buffer.startsWith("data:") && !buffer.startsWith("data: ")) { output = "data: " + buffer.slice(5); } - const bufferedPayload = parseSSELine(bufferedLine); + let bufferedPayload = parseSSELine(bufferedLine); if (bufferedPayload) { providerPayloadCollector.push(bufferedPayload); + bufferedProjectedFailure = projectStreamFailureEvent(bufferedPayload); + if (bufferedProjectedFailure) { + bufferedPayload = bufferedProjectedFailure.publicPayload; + output = `data: ${JSON.stringify(bufferedPayload)}\n\n`; + } if (sanitizeUsagePayloadForRequest(bufferedPayload, body, clientResponseFormat)) output = `data: ${JSON.stringify(bufferedPayload)}\n\n`; if ( @@ -2423,6 +2454,14 @@ export function createSSEStream(options: StreamOptions = {}) { } reqLogger?.appendConvertedChunk?.(output); forward(controller, encoder.encode(output)); + if (bufferedProjectedFailure) { + abortStreamFailure( + controller, + bufferedProjectedFailure.internalFailure, + bufferedProjectedFailure.publicMessage + ); + return; + } } if (shouldInjectClaudeEmptyResponseOnFlush(claudeEmptyResponseLifecycle)) { @@ -2673,6 +2712,7 @@ export function createSSEStream(options: StreamOptions = {}) { if (buffer.trim()) { const parsed = parseSSELine(buffer.trim()); if (parsed && !parsed.done) { + if (emitTranslatedFailureAndAbort(controller, parsed)) return; providerPayloadCollector.push(parsed); // Extract usage from remaining buffer — if the usage-bearing event // (e.g. response.completed) is the last SSE line, it ends up here @@ -2737,58 +2777,9 @@ export function createSSEStream(options: StreamOptions = {}) { // terminal signal for the client. } - let failureHandled = false; - if (onFailure) { - try { - timing.markInterrupted(); - failureHandled = - onFailure({ - status: err.status, - message: err.message, - code: err.code, - type: err.type, - }) === true; - } catch (e) { - console.debug(`[STREAM] onFailure callback error (${model || "unknown"}):`, e); - } - } - const errorBody = buildErrorBody(err.status, err.message); - if (onComplete) { - try { - onComplete({ - status: err.status, - usage: state?.usage, - responseBody: errorBody, - ttft: timing.ttftMs(), - itlMs: timing.avgItlMs(), - interrupted: timing.interrupted, - error: err.message, - errorCode: err.code, - providerPayload: providerPayloadCollector.build( - providerPayloadCollector.getSummary(), - { includeEvents: false } - ), - clientPayload: clientPayloadCollector.build(errorBody, { - includeEvents: false, - }), - }); - failureHandled = true; - } catch (e) { - console.debug( - `[STREAM] onComplete callback error in error path (${model || "unknown"}):`, - e - ); - } - } - - clearIdleTimer(); - if (!failureHandled) { - clearPendingRequestFromStream(); - } - controller.error( - markPendingRequestCleared(new Error(err.message || "Upstream failure")) - ); + const publicErrorMessage = errorBody.error.message; + abortStreamFailure(controller, err, publicErrorMessage, { notifyComplete: true }); return; } @@ -2996,8 +2987,8 @@ export function createSSEStream(options: StreamOptions = {}) { clearIdleTimer(); }, }, - { highWaterMark }, - { highWaterMark } + { highWaterMark: 16384 }, + { highWaterMark: 16384 } ); } @@ -3019,8 +3010,7 @@ export function createSSETransformStreamWithLogger( copilotCompatibleReasoning = false, suppressThinkClose = false, customToolNames: ReadonlySet = new Set(), - requestToolIdentityMap: Map | null = null, - highWaterMark?: number + requestToolIdentityMap: Map | null = null ) { return createSSEStream({ mode: STREAM_MODE.TRANSLATE, @@ -3039,7 +3029,6 @@ export function createSSETransformStreamWithLogger( suppressThinkClose, customToolNames, requestToolIdentityMap, - highWaterMark, }); } @@ -3054,8 +3043,7 @@ export function createPassthroughStreamWithLogger( apiKeyInfo: unknown = null, onFailure: ((payload: StreamFailurePayload) => boolean | void | Promise) | null = null, clientResponseFormat: string | null = null, - requestToolIdentityMap: Map | null = null, - highWaterMark?: number + requestToolIdentityMap: Map | null = null ) { return createSSEStream({ mode: STREAM_MODE.PASSTHROUGH, @@ -3070,7 +3058,6 @@ export function createPassthroughStreamWithLogger( onFailure, clientResponseFormat, requestToolIdentityMap, - highWaterMark, }); } diff --git a/open-sse/utils/streamErrorFormat.ts b/open-sse/utils/streamErrorFormat.ts index 56b747f4e4..05a065a864 100644 --- a/open-sse/utils/streamErrorFormat.ts +++ b/open-sse/utils/streamErrorFormat.ts @@ -1,5 +1,6 @@ import { FORMATS } from "../translator/formats.ts"; -import { buildErrorBody } from "./error.ts"; +import { buildErrorBody, sanitizeErrorMessage } from "./error.ts"; +import { projectResponsesFailureOutput } from "./responsesFailureOutput.ts"; /** * Upstream stream-failure normalization + client-format error framing. @@ -17,10 +18,125 @@ export type StreamFailurePayload = { type?: string; }; +export type ProjectedStreamFailureEvent = { + internalFailure: StreamFailurePayload; + publicMessage: string; + publicPayload: JsonRecord; +}; + +export type PreparedTranslatedStreamFailure = { + record: JsonRecord; + providerPayload: JsonRecord; + internalFailure: StreamFailurePayload; + publicMessage: string; +}; + +export function projectCompletedStreamError( + failure: StreamFailurePayload | null | undefined +): JsonRecord | null { + if (!failure) return null; + const status = Number.isInteger(failure.status) ? failure.status : 502; + return buildErrorBody(status, failure.message, undefined, { + type: failure.type ?? "server_error", + code: String(failure.status ?? 502), + }).error; +} + function asRecord(value: unknown): JsonRecord { return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : {}; } +const RESPONSES_FAILURE_SCALAR_FIELDS = [ + "id", + "object", + "created_at", + "completed_at", + "background", + "model", + "max_output_tokens", + "max_tool_calls", + "parallel_tool_calls", + "previous_response_id", + "service_tier", + "store", + "temperature", + "top_p", + "truncation", +] as const; + +const ABSOLUTE_PATH_SEGMENT = + /(?:^|[\\/])(?:Users|app|etc|home|opt|private|root|srv|tmp|usr|var|workspace)[\\/]/i; + +function projectResponsesFailureString(key: string, value: string): string { + const sanitized = sanitizeErrorMessage(value); + if (sanitized !== value || ABSOLUTE_PATH_SEGMENT.test(value)) return "[REDACTED]"; + if ( + (key === "id" || key === "previous_response_id") && + !/^[A-Za-z0-9][\w.:-]{0,511}$/.test(value) + ) { + return "[REDACTED]"; + } + if (key === "model" && !/^[A-Za-z0-9][A-Za-z0-9._:/-]{0,255}$/.test(value)) { + return "[REDACTED]"; + } + return sanitized; +} + +function projectResponsesFailureUsage(value: unknown): JsonRecord | null { + const usage = asRecord(value); + const projected: JsonRecord = {}; + for (const key of ["input_tokens", "output_tokens", "total_tokens"] as const) { + if (typeof usage[key] === "number" && Number.isFinite(usage[key])) { + projected[key] = usage[key]; + } + } + const allowedDetailFields = { + input_tokens_details: new Set(["cached_tokens"]), + output_tokens_details: new Set([ + "reasoning_tokens", + "accepted_prediction_tokens", + "rejected_prediction_tokens", + ]), + } as const; + for (const key of ["input_tokens_details", "output_tokens_details"] as const) { + const details = asRecord(usage[key]); + const projectedDetails = Object.fromEntries( + Object.entries(details).filter( + ([detailKey, detail]) => + allowedDetailFields[key].has(detailKey) && + typeof detail === "number" && + Number.isFinite(detail) + ) + ); + if (Object.keys(projectedDetails).length > 0) projected[key] = projectedDetails; + } + return Object.keys(projected).length > 0 ? projected : null; +} + +function projectResponsesFailureObject(response: JsonRecord, publicError: JsonRecord): JsonRecord { + const projected: JsonRecord = { status: "failed", error: publicError }; + + // A failed Responses event is an error boundary, so copy only documented protocol + // fields with their scalar shapes. Spreading the upstream object would also publish + // provider-only siblings such as diagnostics, settings, raw messages, or stack traces. + for (const key of RESPONSES_FAILURE_SCALAR_FIELDS) { + const value = response[key]; + if (typeof value === "string") projected[key] = projectResponsesFailureString(key, value); + else if (value === null || typeof value === "number" || typeof value === "boolean") + projected[key] = value; + } + if (Array.isArray(response.output)) { + projected.output = projectResponsesFailureOutput( + response.output, + projectResponsesFailureString + ); + } + const usage = projectResponsesFailureUsage(response.usage); + if (usage) projected.usage = usage; + if ("last_error" in response) projected.last_error = publicError; + return projected; +} + function toStreamFailureStatus(value: unknown): number | null { if (typeof value === "number" && Number.isInteger(value) && value >= 400 && value <= 599) { return value; @@ -48,19 +164,30 @@ function looksLikeStreamRateLimit(code: string, type: string, message: string): export function normalizeStreamFailurePayload(payload: unknown): StreamFailurePayload | null { const record = payload && typeof payload === "object" ? (payload as JsonRecord) : {}; const response = asRecord(record.response); - const error = Object.keys(asRecord(response.error)).length - ? asRecord(response.error) - : Object.keys(asRecord(record.error)).length - ? asRecord(record.error) - : record; + const responseError = response.error; + const responseLastError = response.last_error; + const rootError = record.error; + const error = Object.keys(asRecord(responseError)).length + ? asRecord(responseError) + : Object.keys(asRecord(responseLastError)).length + ? asRecord(responseLastError) + : Object.keys(asRecord(rootError)).length + ? asRecord(rootError) + : record; const code = typeof error.code === "string" ? error.code : "upstream_error"; const type = typeof error.type === "string" ? error.type : undefined; const message = typeof error.message === "string" && error.message.trim() ? error.message - : typeof record.message === "string" && record.message.trim() - ? record.message - : "Upstream failure"; + : typeof responseError === "string" && responseError.trim() + ? responseError + : typeof responseLastError === "string" && responseLastError.trim() + ? responseLastError + : typeof rootError === "string" && rootError.trim() + ? rootError + : typeof record.message === "string" && record.message.trim() + ? record.message + : "Upstream failure"; const status = toStreamFailureStatus(error.status_code) ?? toStreamFailureStatus(error.status) ?? @@ -78,6 +205,80 @@ export function normalizeStreamFailurePayload(payload: unknown): StreamFailurePa }; } +export function prepareTranslatedStreamFailure( + payload: unknown +): PreparedTranslatedStreamFailure | null { + const record = asRecord(payload); + const projected = projectStreamFailureEvent(record); + if (!projected && !record.error) return null; + return { + record, + providerPayload: projected?.publicPayload ?? record, + internalFailure: projected?.internalFailure ?? + normalizeStreamFailurePayload(record) ?? { + status: 502, + message: "Upstream failure", + code: "stream_error", + type: "server_error", + }, + publicMessage: projected?.publicMessage || "Upstream failure", + }; +} + +/** + * Project same-format upstream failure events before they cross the client/log boundary. + * + * `internalFailure` intentionally retains the raw provider wording: account fallback uses it + * to classify quota/reset hints before the persistence seam sanitizes the stored message. + * `publicPayload` is a separate protocol-preserving object whose failure subtrees are rebuilt by + * the canonical public boundary. Callers must never forward the raw payload for these events. + */ +export function projectStreamFailureEvent(payload: unknown): ProjectedStreamFailureEvent | null { + const record = asRecord(payload); + const response = asRecord(record.response); + const hasRootError = + Object.keys(asRecord(record.error)).length > 0 || + (typeof record.error === "string" && record.error.trim().length > 0); + const isResponsesFailure = + record.type === "response.failed" || + (record.type === "response.completed" && response.status === "failed"); + const isClaudeFailure = record.type === "error"; + if (!isResponsesFailure && !isClaudeFailure && !hasRootError) return null; + + const internalFailure = normalizeStreamFailurePayload(record); + if (!internalFailure) return null; + + const publicError = buildErrorBody(internalFailure.status, internalFailure.message, undefined, { + type: internalFailure.type ?? "server_error", + code: internalFailure.code ?? "stream_error", + }).error; + let publicPayload: JsonRecord; + if (isResponsesFailure) { + // Preserve protocol metadata and partial `output[].content[]` without passing output + // through a bounded-depth details sanitizer, while excluding arbitrary diagnostic siblings. + const publicResponse = projectResponsesFailureObject(response, publicError); + publicPayload = { + type: record.type, + response: publicResponse, + ...(typeof record.sequence_number === "number" + ? { sequence_number: record.sequence_number } + : {}), + }; + } else if (isClaudeFailure) { + publicPayload = { type: "error", error: publicError }; + } else { + // OpenAI-compatible HTTP-200 streams commonly emit a bare `{ error: ... }` frame. + // Rebuild the complete public envelope so provider-only fields cannot cross the wire. + publicPayload = { error: publicError }; + } + + return { + internalFailure, + publicMessage: publicError.message, + publicPayload, + }; +} + export function formatTranslatedStreamError(payload: unknown, sourceFormat?: string): string { const failure = normalizeStreamFailurePayload(payload) ?? { status: 502, diff --git a/open-sse/utils/streamFailureBoundary.ts b/open-sse/utils/streamFailureBoundary.ts new file mode 100644 index 0000000000..bb3da7c850 --- /dev/null +++ b/open-sse/utils/streamFailureBoundary.ts @@ -0,0 +1,76 @@ +import { buildErrorBody } from "./error.ts"; +import type { StreamFailurePayload } from "./streamErrorFormat.ts"; +import type { StreamTiming } from "./streamTiming.ts"; + +type CompletePayload = { + status: number; + usage: unknown; + responseBody: unknown; + providerPayload: unknown; + clientPayload: unknown; + error: string; + errorCode?: string; + ttft: number | null; + itlMs: number | null; + interrupted: boolean; +}; + +type AborterContext = { + onFailure?: ((payload: StreamFailurePayload) => boolean | void | Promise) | null; + onComplete?: ((payload: CompletePayload) => void) | null; + getUsage: () => unknown; + timing: StreamTiming; + buildProviderPayload: () => unknown; + buildClientPayload: (body: unknown) => unknown; + clearIdleTimer: () => void; + clearPendingRequest: () => void; + markPendingRequestCleared: (error: Error) => Error; + model?: string | null; +}; + +export function createStreamFailureAborter(context: AborterContext) { + return ( + controller: TransformStreamDefaultController, + failure: StreamFailurePayload, + publicMessage: string, + options: { notifyComplete?: boolean } = {} + ): void => { + let handled = false; + context.timing.markInterrupted(); + if (context.onFailure) { + try { + handled = context.onFailure(failure) === true; + } catch (error) { + console.debug("[STREAM] onFailure callback error:", error); + } + } + let safeMessage = publicMessage || "Upstream failure"; + if (options.notifyComplete && context.onComplete) { + const body = buildErrorBody(failure.status, failure.message); + safeMessage = body.error.message; + try { + context.onComplete({ + status: failure.status, + usage: context.getUsage(), + responseBody: body, + ttft: context.timing.ttftMs(), + itlMs: context.timing.avgItlMs(), + interrupted: context.timing.interrupted, + error: safeMessage, + errorCode: failure.code, + providerPayload: context.buildProviderPayload(), + clientPayload: context.buildClientPayload(body), + }); + handled = true; + } catch (error) { + console.debug( + `[STREAM] onComplete callback error in error path (${context.model || "unknown"}):`, + error + ); + } + } + context.clearIdleTimer(); + if (!handled) context.clearPendingRequest(); + controller.error(context.markPendingRequestCleared(new Error(safeMessage))); + }; +} diff --git a/open-sse/utils/streamFailureFinalization.ts b/open-sse/utils/streamFailureFinalization.ts index 7d4e57ffba..38a740d1fb 100644 --- a/open-sse/utils/streamFailureFinalization.ts +++ b/open-sse/utils/streamFailureFinalization.ts @@ -5,6 +5,7 @@ import { import { HTTP_STATUS } from "../config/constants.ts"; import { buildErrorBody } from "./error.ts"; +import { sanitizeErrorMessage } from "./errorSanitization.ts"; export type StreamCompletionPayload = { status: number; @@ -129,9 +130,7 @@ export function finalizeStreamRequestLog({ } else { console.warn( "finalizeMostRecentPendingRequest failed:", - error && typeof error === "object" && "message" in error - ? (error as { message?: unknown }).message - : error + sanitizeErrorMessage(error) || "Stream request finalization failed" ); } } catch {} @@ -158,12 +157,12 @@ export function createStreamFailureFinalizers({ const status = failure.status || HTTP_STATUS.BAD_GATEWAY; const message = failure.message || "Upstream stream error"; - const code = failure.code || failure.type || String(status); const classification = failure.code || failure.type ? { code: failure.code, type: failure.type } : undefined; + const errorBody = buildErrorBody(status, message, undefined, classification); + const projectedCode = errorBody.error.code || String(status); if (!isFailureCompletionRecorded()) { - const errorBody = buildErrorBody(status, message, undefined, classification); onStreamComplete({ status, usage: null, @@ -171,12 +170,12 @@ export function createStreamFailureFinalizers({ providerPayload: errorBody, clientPayload: errorBody, error: message, - errorCode: code, + errorCode: projectedCode, ttft: 0, }); } - persistFailureUsage(status, code); + persistFailureUsage(status, projectedCode); try { onStreamFailure?.(failure); } catch { diff --git a/open-sse/utils/upstreamErrorPassthrough.ts b/open-sse/utils/upstreamErrorPassthrough.ts index b62fff2adf..30fa4ffff2 100644 --- a/open-sse/utils/upstreamErrorPassthrough.ts +++ b/open-sse/utils/upstreamErrorPassthrough.ts @@ -1,13 +1,16 @@ -import { RAW_CREDENTIAL_PATTERNS } from "./error.ts"; +import { + containsSensitiveErrorCredential, + sanitizePassthroughUpstreamDetails, +} from "./errorSanitization.ts"; + /** * Selective upstream 4xx error passthrough (Claude Code auto-recover contract). * - * Claude Code matches the upstream error WORDING to auto-disable capabilities - * (thinking / output_config) for the rest of the conversation. Wrapping the body - * via buildErrorBody() truncates the message and breaks that recovery. For - * upstream-originated 4xx errors the body is the provider's public API message — - * not our internals — so it is safe and required to relay it verbatim. - * OmniRoute-generated errors MUST keep using buildErrorBody() (Hard Rule #12). + * Claude Code matches upstream error wording to auto-disable capabilities + * (thinking / output_config) for the rest of the conversation. This path keeps + * the wording and JSON shape required for that recovery after applying the + * canonical recursive sanitizer. OmniRoute-generated errors MUST keep using + * buildErrorBody() (Hard Rule #12). */ const PASSTHROUGH_MIN = 400; const PASSTHROUGH_MAX = 499; @@ -18,39 +21,28 @@ const EXCLUDED_STATUSES = new Set([401, 403, 407]); const INTERNAL_LEAK_RE = /\sat\s\/|node_modules|omniroute\//i; // #10898-sec / secret-in-error hardening: some providers echo the offending // request (including an Authorization header or api key) inside a 400/422/429 -// validation body. Passthrough relays the body VERBATIM (the Claude Code -// capability-recovery contract needs the exact wording), so we cannot key-drop -// via sanitizeUpstreamDetails without breaking that contract. Instead, if the -// body actually carries a credential pattern, REFUSE passthrough and let the -// caller fall back to the sanitized buildErrorBody path. Bodies without a -// secret (the overwhelming majority, carrying capability/quota wording) still -// relay verbatim. Mirrors the vocabulary of redactSensitiveErrorText in error.ts. -const LABELLED_CREDENTIAL_RE = - /\b(?:Bearer|Basic)\s+[A-Za-z0-9._~+/=-]{8,}|(?:api[_-]?key|access[_-]?token|refresh[_-]?token|authorization|cookie|secret)\\?["']?\s*[:=]\s*\\?["']?[^"'\\,\s}]{6,}/i; - -/** - * The raw-token shapes (sk-…, AIza…, JWT) come from error.ts's - * RAW_CREDENTIAL_PATTERNS rather than a second local copy. The previous local - * copy carried `sk-` while the sanitizer this file falls back to did NOT, so a - * body recognized as leaky here was returned unredacted there - * (GHSA-qv45-56jc-4wmj). One source, no drift. - */ -function containsCredential(text: string): boolean { - if (LABELLED_CREDENTIAL_RE.test(text)) return true; - return RAW_CREDENTIAL_PATTERNS.some((pattern) => { - pattern.lastIndex = 0; // the shared patterns are /g — reset before .test() - return pattern.test(text); - }); -} +// validation body. If the body carries a credential pattern, REFUSE passthrough +// before the recursive sanitizer so the caller falls back to buildErrorBody. +// Eligible JSON retains its safe shape and capability/quota wording after the +// recursive projection. Mirrors redactSensitiveErrorText in errorSanitization.ts. +const CREDENTIAL_LEAK_RE = + /\b(?:Bearer|Basic)\s+[A-Za-z0-9._~+/=-]{8,}|\bsk-[A-Za-z0-9._-]{8,}|(?:api[_-]?key|access[_-]?token|refresh[_-]?token|authorization|cookie|secret)\\?["']?\s*[:=]\s*\\?["']?[^"'\\,\s}]{6,}/i; export function shouldPassthroughUpstreamError(statusCode: number, upstreamBody: unknown): boolean { if (statusCode < PASSTHROUGH_MIN || statusCode > PASSTHROUGH_MAX) return false; if (EXCLUDED_STATUSES.has(statusCode)) return false; if (!upstreamBody || typeof upstreamBody !== "object") return false; - const text = JSON.stringify(upstreamBody); + let text: string | undefined; + try { + text = JSON.stringify(upstreamBody); + } catch { + // Relay only JSON-stable objects; cyclic/BigInt/hostile toJSON bodies fail closed. + return false; + } + if (typeof text !== "string") return false; if (INTERNAL_LEAK_RE.test(text)) return false; // Refuse passthrough when the provider echoed a credential back to us. - if (containsCredential(text)) return false; + if (CREDENTIAL_LEAK_RE.test(text) || containsSensitiveErrorCredential(text)) return false; return true; } @@ -60,8 +52,18 @@ export function buildPassthroughErrorResponse( headers?: Record ): Response | null { if (!shouldPassthroughUpstreamError(statusCode, upstreamBody)) return null; - return new Response(JSON.stringify(upstreamBody), { - status: statusCode, - headers: { "Content-Type": "application/json", ...(headers || {}) }, - }); + try { + const sanitizedBody = sanitizePassthroughUpstreamDetails(upstreamBody); + const publicBody = + sanitizedBody && typeof sanitizedBody === "object" + ? sanitizedBody + : { error: { message: "Upstream error" } }; + return new Response(JSON.stringify(publicBody), { + status: statusCode, + headers: { "Content-Type": "application/json", ...(headers || {}) }, + }); + } catch { + // A proxy/getter may behave differently between eligibility and projection. + return null; + } } diff --git a/open-sse/utils/upstreamErrorResponse.ts b/open-sse/utils/upstreamErrorResponse.ts new file mode 100644 index 0000000000..1581160900 --- /dev/null +++ b/open-sse/utils/upstreamErrorResponse.ts @@ -0,0 +1,46 @@ +import { buildErrorBody, sanitizeUpstreamDetails } from "./error.ts"; + +interface SanitizedUpstreamErrorResponseOptions { + status: number; + rawBody: string; + fallbackMessage: string; + headers?: Record; +} + +/** + * Preserve a provider's JSON error shape while applying the canonical recursive sanitizer. + * Providers sometimes label plain text as JSON; those bodies use OmniRoute's canonical error + * envelope so the advertised content type always matches the response bytes. + */ +export function buildSanitizedUpstreamErrorResponse({ + status, + rawBody, + fallbackMessage, + headers, +}: SanitizedUpstreamErrorResponseOptions): Response { + const trimmedBody = rawBody.trim(); + + if (trimmedBody) { + try { + const parsedBody: unknown = JSON.parse(trimmedBody); + const serializedBody = JSON.stringify(sanitizeUpstreamDetails(parsedBody)); + if (serializedBody !== undefined) { + return new Response(serializedBody, { + status, + headers: { ...headers, "Content-Type": "application/json" }, + }); + } + } catch { + // Upstreams commonly return text or HTML despite an application/json response header. + // Treat it as an opaque message and use the canonical JSON envelope below. + } + } + + // Non-JSON is an opaque upstream body. Do not echo even sanitized fragments: + // provider HTML/plaintext can contain credentials or implementation details + // outside the patterns the canonical sanitizer knows about. + return new Response(JSON.stringify(buildErrorBody(status, fallbackMessage)), { + status, + headers: { ...headers, "Content-Type": "application/json" }, + }); +} diff --git a/src/app/api/logs/[id]/route.ts b/src/app/api/logs/[id]/route.ts index afdb7d2432..4fad3932a9 100644 --- a/src/app/api/logs/[id]/route.ts +++ b/src/app/api/logs/[id]/route.ts @@ -1,5 +1,7 @@ import { NextResponse } from "next/server"; +import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/errorSanitization.ts"; import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; +import { sanitizeErrorFramesFromLogChunks } from "@/lib/logPayloads"; import { getCallLogById } from "@/lib/usageDb"; import { getCompletedDetails, getPendingById } from "@/lib/usage/usageHistory"; import { @@ -18,6 +20,29 @@ import { // before it's parsed. const CHUNK_LOG_TIMESTAMP_PREFIX = /^\[\d{2}:\d{2}:\d{2}\.\d{3}\]\s*/; +type ManagementStreamChunks = { + provider?: string[]; + openai?: string[]; + client?: string[]; +}; + +function projectManagementStreamChunks( + streamChunks: ManagementStreamChunks | null | undefined +): ManagementStreamChunks | null { + if (!streamChunks) return null; + return { + ...(streamChunks.provider + ? { provider: sanitizeErrorFramesFromLogChunks(streamChunks.provider) } + : {}), + ...(streamChunks.openai + ? { openai: sanitizeErrorFramesFromLogChunks(streamChunks.openai) } + : {}), + ...(streamChunks.client + ? { client: sanitizeErrorFramesFromLogChunks(streamChunks.client) } + : {}), + }; +} + // Best-effort parse of the accumulated SSE `data:` lines captured live for an // in-flight request (open-sse/utils/requestLogger.ts's appendConvertedChunk // mutates these arrays in place as chunks arrive, so this reflects "the reply @@ -77,12 +102,13 @@ export async function GET( try { const pendingRequestDetail = getPendingById().get(id); if (pendingRequestDetail) { + const safeStreamChunks = projectManagementStreamChunks(pendingRequestDetail.streamChunks); const pipelinePayloads: any = { clientRequest: pendingRequestDetail.clientRequest ?? null, providerRequest: pendingRequestDetail.providerRequest ?? null, providerResponse: pendingRequestDetail.providerResponse ?? null, clientResponse: pendingRequestDetail.clientResponse ?? null, - streamChunks: pendingRequestDetail.streamChunks ?? null, + streamChunks: safeStreamChunks, }; const activeEntry = { @@ -102,7 +128,7 @@ export async function GET( // The still-generating reply so far — the request's own context // panel renders this alongside its (already-complete) requestBody // instead of waiting for the stream to finish. - partialAssistantText: extractPartialAssistantText(pendingRequestDetail.streamChunks), + partialAssistantText: extractPartialAssistantText(safeStreamChunks), }; return NextResponse.json(activeEntry); @@ -123,12 +149,13 @@ export async function GET( const completed = getCompletedDetails(); const inMem = completed.get(id); if (inMem) { + const safeStreamChunks = projectManagementStreamChunks(inMem.streamChunks); const pipelinePayloads: any = { clientRequest: inMem.clientRequest ?? null, providerRequest: inMem.providerRequest ?? null, providerResponse: inMem.providerResponse ?? null, clientResponse: inMem.clientResponse ?? null, - streamChunks: inMem.streamChunks ?? null, + streamChunks: safeStreamChunks, }; const minimal = { @@ -142,7 +169,7 @@ export async function GET( duration: Date.now() - inMem.startedAt, detailState: "in-memory", active: false, - error: inMem.error || null, + error: sanitizeErrorMessage(inMem.error) || null, pipelinePayloads, hasPipelineDetails: true, }; diff --git a/src/app/api/providers/[id]/models/staleEncryptionGuard.ts b/src/app/api/providers/[id]/models/staleEncryptionGuard.ts index fc410b1a37..5f2384e921 100644 --- a/src/app/api/providers/[id]/models/staleEncryptionGuard.ts +++ b/src/app/api/providers/[id]/models/staleEncryptionGuard.ts @@ -40,9 +40,7 @@ export function buildStaleEncryptionKeyResponse( `(STORAGE_ENCRYPTION_KEY changed or unset). Re-authenticate this account, or verify ` + `STORAGE_ENCRYPTION_KEY matches the key used to store it.`; - // buildErrorBody sanitizes the message (Rule #12); override the type so the - // client can key off the specific stale-encryption cause. - const body = buildErrorBody(424, message); - body.error.type = "storage_encryption_stale"; + // buildErrorBody sanitizes the message and projects the client-visible classification. + const body = buildErrorBody(424, message, undefined, { type: "storage_encryption_stale" }); return NextResponse.json(body, { status: 424 }); } diff --git a/src/app/api/providers/[id]/test/publicErrorBoundary.ts b/src/app/api/providers/[id]/test/publicErrorBoundary.ts new file mode 100644 index 0000000000..30addfeffb --- /dev/null +++ b/src/app/api/providers/[id]/test/publicErrorBoundary.ts @@ -0,0 +1,155 @@ +import { projectProviderValidationResultForPublicResponse } from "@/lib/providers/validation/transport"; +import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/errorSanitization.ts"; +import { makeDiagnosis } from "./codexAppServerHealth"; +import { classifyAmbiguousOrAuthError, type ClassifyFailureArgs } from "./mistralAmbiguousAuth"; + +export function toSafeMessage(value: unknown, fallback = "Unknown error"): string { + const safeMessage = sanitizeErrorMessage(value).trim(); + return safeMessage || fallback; +} + +/** + * A provider/account that the upstream has deactivated (vs. a revoked/expired token). + * #1444: a Codex account can have a perfectly healthy OAuth refresh while its ChatGPT + * account is deactivated, in which case the API returns 401 — mislabeling that as + * "Token invalid or revoked" hides the real cause. Mirrors the deactivation phrases the + * account-fallback classifier already trusts. + */ +export function isAccountDeactivatedMessage(text: string): boolean { + const normalized = (text || "").toLowerCase(); + return ( + normalized.includes("account_deactivated") || + (normalized.includes("deactivat") && normalized.includes("account")) + ); +} + +export function classifyFailure({ + error, + statusCode = null, + refreshFailed = false, + unsupported = false, + provider, +}: ClassifyFailureArgs) { + const message = toSafeMessage(error, "Connection test failed"); + const normalized = message.toLowerCase(); + const numericStatus = Number.isFinite(statusCode) ? Number(statusCode) : null; + + if (unsupported) { + return makeDiagnosis("unsupported", "validation", message, "unsupported"); + } + + if (refreshFailed || normalized.includes("refresh failed")) { + return makeDiagnosis("token_refresh_failed", "oauth", message, "refresh_failed"); + } + + // #1444: a deactivated account is distinct from a revoked/expired token — surface it + // as account_deactivated (which the dashboard renders as "Account Deactivated") before + // the generic 401/403 branch below would mark it "upstream_auth_error". + if (isAccountDeactivatedMessage(normalized)) { + return makeDiagnosis("account_deactivated", "account", message, "account_deactivated"); + } + + if (numericStatus === 401 || numericStatus === 403) { + return classifyAmbiguousOrAuthError(provider, normalized, message, numericStatus); + } + + if (numericStatus === 429) { + return makeDiagnosis("upstream_rate_limited", "upstream", message, "429"); + } + + if (numericStatus && numericStatus >= 500) { + return makeDiagnosis("upstream_unavailable", "upstream", message, String(numericStatus)); + } + + if (normalized.includes("token expired") || normalized.includes("expired")) { + return makeDiagnosis("token_expired", "oauth", message, "token_expired"); + } + + if ( + normalized.includes("invalid api key") || + normalized.includes("token invalid") || + normalized.includes("revoked") || + normalized.includes("access denied") || + normalized.includes("unauthorized") || + normalized.includes("forbidden") + ) { + return makeDiagnosis( + "upstream_auth_error", + "upstream", + message, + numericStatus ? String(numericStatus) : "auth_failed" + ); + } + + if ( + normalized.includes("rate limit") || + normalized.includes("quota") || + normalized.includes("too many requests") + ) { + return makeDiagnosis( + "upstream_rate_limited", + "upstream", + message, + numericStatus ? String(numericStatus) : "rate_limited" + ); + } + + if ( + normalized.includes("fetch failed") || + normalized.includes("network") || + normalized.includes("timeout") || + normalized.includes("timed out") || + normalized.includes("econn") || + normalized.includes("enotfound") || + normalized.includes("socket") + ) { + return makeDiagnosis("network_error", "upstream", message, "network_error"); + } + + return makeDiagnosis( + "upstream_error", + "upstream", + message, + numericStatus ? String(numericStatus) : "upstream_error" + ); +} + +/** Allowlist the CLI health fields safe to expose outside the local runtime boundary. */ +export function projectProviderRuntimeForPublicResponse( + runtime: unknown +): Record | null { + if (!runtime || typeof runtime !== "object" || Array.isArray(runtime)) return null; + const record = runtime as Record; + const projected: Record = {}; + + for (const field of ["installed", "runnable", "requiresBinary"] as const) { + if (typeof record[field] === "boolean") projected[field] = record[field]; + } + for (const field of ["reason", "runtimeMode", "version", "command"] as const) { + if (typeof record[field] !== "string") continue; + const safeValue = sanitizeErrorMessage(record[field]).trim(); + if (safeValue) projected[field] = safeValue.slice(0, 512); + } + + return projected; +} + +/** Sanitize every connection-test result before health writes, logs, and HTTP responses. */ +export function projectConnectionTestResultForPublicResponse< + T extends { error?: unknown; warning?: unknown; diagnosis?: unknown }, +>(result: T) { + const projected = projectProviderValidationResultForPublicResponse(result); + if (!projected.diagnosis || typeof projected.diagnosis !== "object") return projected; + + const diagnosis = projected.diagnosis as Record; + return { + ...projected, + diagnosis: { + ...diagnosis, + message: + diagnosis.message === null || diagnosis.message === undefined + ? null + : toSafeMessage(diagnosis.message, "Connection test failed"), + }, + }; +} diff --git a/src/app/api/providers/[id]/test/route.ts b/src/app/api/providers/[id]/test/route.ts index cc663a95c8..1a81180358 100644 --- a/src/app/api/providers/[id]/test/route.ts +++ b/src/app/api/providers/[id]/test/route.ts @@ -7,6 +7,7 @@ import { isCloudEnabled, resolveProxyForConnection } from "@/lib/db/settings"; import { getConsistentMachineId } from "@/shared/utils/machineId"; import { syncToCloud } from "@/lib/cloudSync"; import { validateProviderApiKey } from "@/lib/providers/validation"; +import { projectProviderValidationResultForPublicResponse } from "@/lib/providers/validation/transport"; import { getCliRuntimeStatus } from "@/shared/services/cliRuntime"; import { buildQoderCliNotFoundHint } from "@omniroute/open-sse/services/qoderCliResolve.ts"; // Use the shared open-sse token refresh with built-in dedup/race-condition cache @@ -29,11 +30,19 @@ import { testCodexAppServerConnection, makeDiagnosis } from "./codexAppServerHea import { recoverKeyHealth } from "@omniroute/open-sse/services/apiKeyRotator.ts"; import { shouldClearErrorStateOnValidProbe } from "@/lib/usage/providerLimits"; import { isConnectionUnavailableToAuxiliaryActivity } from "@/lib/exclusiveLeaseIsolation"; -import { classifyAmbiguousOrAuthError, type ClassifyFailureArgs } from "./mistralAmbiguousAuth"; import { buildApiKeyConnectionTestResult } from "./apiKeyTestResult"; import { classifyOAuthProbeInconclusive, OAUTH_TEST_CONFIG } from "./oauthTestConfig"; import { isGeoBlockedError } from "@omniroute/open-sse/services/errorClassifier.ts"; import * as retirement from "@/lib/providers/chatgptWebRetirementResponse"; +import { + classifyFailure, + isAccountDeactivatedMessage, + projectConnectionTestResultForPublicResponse, + projectProviderRuntimeForPublicResponse, + toSafeMessage, +} from "./publicErrorBoundary"; + +export { classifyFailure, projectProviderRuntimeForPublicResponse } from "./publicErrorBoundary"; // Match the API-key path's 30s timeout so a hung OAuth upstream cannot block the test queue. const OAUTH_TEST_TIMEOUT_MS = 30_000; @@ -45,115 +54,6 @@ const providerConnectionTestBodySchema = z.object({ validationModelId: z.string().max(500).optional(), }); -function toSafeMessage(value: any, fallback = "Unknown error"): string { - if (typeof value !== "string") return fallback; - const trimmed = value.trim(); - return trimmed || fallback; -} - -/** - * A provider/account that the upstream has deactivated (vs. a revoked/expired token). - * #1444: a Codex account can have a perfectly healthy OAuth refresh while its ChatGPT - * account is deactivated, in which case the API returns 401 — mislabeling that as - * "Token invalid or revoked" hides the real cause. Mirrors the deactivation phrases the - * account-fallback classifier already trusts. - */ -function isAccountDeactivatedMessage(text: string): boolean { - const n = (text || "").toLowerCase(); - return n.includes("account_deactivated") || (n.includes("deactivat") && n.includes("account")); -} - -export function classifyFailure({ - error, - statusCode = null, - refreshFailed = false, - unsupported = false, - provider, -}: ClassifyFailureArgs) { - const message = toSafeMessage(error, "Connection test failed"); - const normalized = message.toLowerCase(); - const numericStatus = Number.isFinite(statusCode) ? Number(statusCode) : null; - - if (unsupported) { - return makeDiagnosis("unsupported", "validation", message, "unsupported"); - } - - if (refreshFailed || normalized.includes("refresh failed")) { - return makeDiagnosis("token_refresh_failed", "oauth", message, "refresh_failed"); - } - - // #1444: a deactivated account is distinct from a revoked/expired token — surface it - // as account_deactivated (which the dashboard renders as "Account Deactivated") before - // the generic 401/403 branch below would mark it "upstream_auth_error". - if (isAccountDeactivatedMessage(normalized)) { - return makeDiagnosis("account_deactivated", "account", message, "account_deactivated"); - } - - if (numericStatus === 401 || numericStatus === 403) { - return classifyAmbiguousOrAuthError(provider, normalized, message, numericStatus); - } - - if (numericStatus === 429) { - return makeDiagnosis("upstream_rate_limited", "upstream", message, "429"); - } - - if (numericStatus && numericStatus >= 500) { - return makeDiagnosis("upstream_unavailable", "upstream", message, String(numericStatus)); - } - - if (normalized.includes("token expired") || normalized.includes("expired")) { - return makeDiagnosis("token_expired", "oauth", message, "token_expired"); - } - - if ( - normalized.includes("invalid api key") || - normalized.includes("token invalid") || - normalized.includes("revoked") || - normalized.includes("access denied") || - normalized.includes("unauthorized") || - normalized.includes("forbidden") - ) { - return makeDiagnosis( - "upstream_auth_error", - "upstream", - message, - numericStatus ? String(numericStatus) : "auth_failed" - ); - } - - if ( - normalized.includes("rate limit") || - normalized.includes("quota") || - normalized.includes("too many requests") - ) { - return makeDiagnosis( - "upstream_rate_limited", - "upstream", - message, - numericStatus ? String(numericStatus) : "rate_limited" - ); - } - - if ( - normalized.includes("fetch failed") || - normalized.includes("network") || - normalized.includes("timeout") || - normalized.includes("timed out") || - normalized.includes("econn") || - normalized.includes("enotfound") || - normalized.includes("socket") - ) { - return makeDiagnosis("network_error", "upstream", message, "network_error"); - } - - return makeDiagnosis( - "upstream_error", - "upstream", - message, - numericStatus ? String(numericStatus) : "upstream_error" - ); -} - function hasQoderToken(connection: any): boolean { if (typeof connection?.apiKey === "string" && connection.apiKey.trim().length > 0) return true; const psd = connection?.providerSpecificData; @@ -218,7 +118,10 @@ async function getProviderRuntimeStatus(connection: any) { error: runtimeMessage, }; } catch (error) { - const runtimeMessage = `Failed to check local CLI runtime: ${(error as any)?.message || "runtime_check_failed"}`; + const runtimeMessage = `Failed to check local CLI runtime: ${toSafeMessage( + error, + "runtime_check_failed" + )}`; return { installed: false, runnable: false, @@ -302,7 +205,10 @@ async function refreshOAuthToken(connection: any) { }); return result; // { accessToken, expiresIn, refreshToken } or null } catch (err) { - console.error(`Error refreshing ${provider} token:`, (err as any).message); + console.error( + `Error refreshing ${provider} token:`, + toSafeMessage(err, "Token refresh failed") + ); return null; } } @@ -376,7 +282,10 @@ async function syncToCloudIfEnabled() { const machineId = await getConsistentMachineId(); await syncToCloud(machineId); } catch (error) { - console.log("Error syncing to cloud after token refresh:", error); + console.log( + "Error syncing to cloud after token refresh:", + toSafeMessage(error, "Cloud sync failed") + ); } } @@ -934,11 +843,13 @@ async function testApiKeyConnection(connection: any) { }; } - const result = await validateProviderApiKey({ - provider: connection.provider, - apiKey: connection.apiKey, - providerSpecificData: connection.providerSpecificData, - }); + const result = projectProviderValidationResultForPublicResponse( + await validateProviderApiKey({ + provider: connection.provider, + apiKey: connection.apiKey, + providerSpecificData: connection.providerSpecificData, + }) + ); if (result.unsupported) { const error = "Provider test not supported"; @@ -1001,8 +912,11 @@ export async function testSingleConnection(connectionId: string, validationModel let proxyInfo: any = null; try { proxyInfo = await resolveProxyForConnection(connectionId); - } catch (proxyErr: any) { - console.log(`[ConnectionTest] Failed to resolve proxy for ${connectionId}:`, proxyErr?.message); + } catch (proxyErr: unknown) { + console.log( + `[ConnectionTest] Failed to resolve proxy for ${connectionId}:`, + toSafeMessage(proxyErr, "Proxy resolution failed") + ); } let result; @@ -1046,6 +960,12 @@ export async function testSingleConnection(connectionId: string, validationModel ); } + // Every runtime path converges here before any health-state write, diagnosis, + // persistent log, or public response. API-key validation is projected at its + // own seam above as well so future refactors cannot move it past this boundary. + result = projectConnectionTestResultForPublicResponse(result); + const publicRuntime = projectProviderRuntimeForPublicResponse(runtime); + const latencyMs = Date.now() - startTime; // Unsupported validation capability is neutral: the probe established that @@ -1063,14 +983,14 @@ export async function testSingleConnection(connectionId: string, validationModel } catch (activateError) { console.log( `[ConnectionTest] Failed to activate unverifiable connection ${connectionId}:`, - (activateError as any)?.message || activateError + toSafeMessage(activateError, "Connection activation failed") ); } } return { ...result, latencyMs, - runtime: runtime || null, + runtime: publicRuntime, testedAt: null, }; } @@ -1214,7 +1134,7 @@ export async function testSingleConnection(connectionId: string, validationModel diagnosis, latencyMs, statusCode: result.statusCode || null, - runtime: runtime || null, + runtime: publicRuntime, testedAt: now, }; } @@ -1245,7 +1165,7 @@ export async function POST(request: Request, { params }: { params: Promise<{ id: } catch (error) { const retired = retirement.responseForError(error); if (retired) return retired; - console.log("Error testing connection:", error); + console.log("Error testing connection:", toSafeMessage(error, "Connection test failed")); return NextResponse.json({ error: "Test failed" }, { status: 500 }); } } diff --git a/src/app/api/providers/validate/route.ts b/src/app/api/providers/validate/route.ts index 7d92d4ac92..0992acde94 100644 --- a/src/app/api/providers/validate/route.ts +++ b/src/app/api/providers/validate/route.ts @@ -1,4 +1,5 @@ import { NextResponse } from "next/server"; +import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/errorSanitization.ts"; import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; import { getAuditRequestContext, logAuditEvent } from "@/lib/compliance/index"; import { getProviderNodeById } from "@/models"; @@ -8,6 +9,7 @@ import { isAnthropicCompatibleProvider, } from "@/shared/constants/providers"; import { validateProviderApiKey } from "@/lib/providers/validation"; +import { projectProviderValidationResultForPublicResponse } from "@/lib/providers/validation/transport"; import { getProxyForLevel } from "@/lib/db/settings"; import { resolveProxyForProvider } from "@/lib/db/proxies"; import { validateProviderApiKeySchema } from "@/shared/validation/schemas"; @@ -123,12 +125,14 @@ export async function POST(request) { proxyToUse = providerProxy || globalProxy || null; } - const result = await runWithProxyContextOrDirect(proxyToUse || null, () => - validateProviderApiKey({ - provider, - apiKey, - providerSpecificData, - }) + const result = projectProviderValidationResultForPublicResponse( + await runWithProxyContextOrDirect(proxyToUse || null, () => + validateProviderApiKey({ + provider, + apiKey, + providerSpecificData, + }) + ) ); if (result.unsupported) { @@ -174,7 +178,7 @@ export async function POST(request) { providerSpecificData: result.providerSpecificData || null, }); } catch (error) { - console.log("Error validating API key:", error); + console.log("Error validating API key:", sanitizeErrorMessage(error) || "Validation failed"); return NextResponse.json({ error: "Validation failed" }, { status: 500 }); } } diff --git a/src/lib/guardrails/credentialMasker.ts b/src/lib/guardrails/credentialMasker.ts index d5529f84f6..6ac88f8fb3 100644 --- a/src/lib/guardrails/credentialMasker.ts +++ b/src/lib/guardrails/credentialMasker.ts @@ -1,5 +1,9 @@ -import { BaseGuardrail, type GuardrailContext, type GuardrailResult } from "./base"; +import { CREDENTIAL_PATTERNS } from "@omniroute/open-sse/utils/credentialPatterns.ts"; import { getSettings } from "@/lib/db/settings"; +import { BaseGuardrail, type GuardrailContext, type GuardrailResult } from "./base"; + +export { CREDENTIAL_PATTERNS }; +export type { CredentialPattern } from "@omniroute/open-sse/utils/credentialPatterns.ts"; /** * CredentialMaskerGuardrail — redacts well-known API-key / secret-token patterns @@ -11,88 +15,6 @@ import { getSettings } from "@/lib/db/settings"; * Future: per-pipeline / per-provider scoping via GuardrailContext. */ -export interface CredentialPattern { - name: string; - regex: RegExp; - replacement: string; -} - -export const CREDENTIAL_PATTERNS: CredentialPattern[] = [ - // ── LLM provider keys ────────────────────────────────────────────────── - { name: "openai_proj", regex: /sk-proj-[A-Za-z0-9_-]{20,}/g, replacement: "[REDACTED:openai]" }, - { name: "openai", regex: /\bsk-[A-Za-z0-9]{48}\b/g, replacement: "[REDACTED:openai]" }, - { - name: "anthropic", - regex: /sk-ant-api[0-9]?-[A-Za-z0-9_-]{20,}/g, - replacement: "[REDACTED:anthropic]", - }, - { - name: "anthropic_alt", - regex: /sk-ant-[A-Za-z0-9_-]{20,}/g, - replacement: "[REDACTED:anthropic]", - }, - { name: "google", regex: /AIza[0-9A-Za-z_-]{35}/g, replacement: "[REDACTED:google]" }, - { name: "huggingface", regex: /hf_[A-Za-z0-9]{34}/g, replacement: "[REDACTED:hf]" }, - { name: "replicate", regex: /r8_[A-Za-z0-9]{37}/g, replacement: "[REDACTED:replicate]" }, - // ── VCS / SaaS tokens ────────────────────────────────────────────────── - { name: "github", regex: /gh[pousr]_[A-Za-z0-9]{36,}/g, replacement: "[REDACTED:github]" }, - { name: "slack", regex: /xox[bpoa]-[A-Za-z0-9-]{10,}/g, replacement: "[REDACTED:slack]" }, - { name: "linear", regex: /lin_api_[A-Za-z0-9]{40}/g, replacement: "[REDACTED:linear]" }, - { name: "notion", regex: /secret_[A-Za-z0-9]{43}/g, replacement: "[REDACTED:notion]" }, - { name: "npm", regex: /npm_[A-Za-z0-9]{36}/g, replacement: "[REDACTED:npm]" }, - { name: "postman", regex: /PMAK-[a-f0-9]{8}-[a-f0-9]{32}/g, replacement: "[REDACTED:postman]" }, - { - name: "discord", - regex: /\b[MN][A-Za-z0-9]{23}\.[A-Za-z0-9]{6}\.[A-Za-z0-9]{27}\b/g, - replacement: "[REDACTED:discord]", - }, - // ── Payments ─────────────────────────────────────────────────────────── - { - name: "stripe", - regex: /(?:sk|rk)_(?:live|test)_[0-9a-zA-Z]{24,}/g, - replacement: "[REDACTED:stripe]", - }, - { - name: "square", - regex: /sq0(?:atp-[0-9A-Za-z_-]{22}|csp-[0-9A-Za-z_-]{43})/g, - replacement: "[REDACTED:square]", - }, - // ── Cloud / infra ────────────────────────────────────────────────────── - { name: "aws_access_key", regex: /AKIA[0-9A-Z]{16}/g, replacement: "[REDACTED:aws]" }, - { name: "twilio", regex: /\bSK[0-9a-fA-F]{32}\b/g, replacement: "[REDACTED:twilio]" }, - { - name: "sendgrid", - regex: /SG\.[A-Za-z0-9_-]{22}\.[A-Za-z0-9_-]{43}/g, - replacement: "[REDACTED:sendgrid]", - }, - { name: "mailgun", regex: /key-[a-f0-9]{32}/g, replacement: "[REDACTED:mailgun]" }, - // ── Crypto / identity ────────────────────────────────────────────────── - { - name: "private_key", - regex: - /-----BEGIN (?:RSA |EC |DSA |OPENSSH |PGP )?PRIVATE KEY-----[\s\S]*?-----END (?:RSA |EC |DSA |OPENSSH |PGP )?PRIVATE KEY-----/g, - replacement: "[REDACTED:private_key]", - }, - { - name: "jwt", - regex: /\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\b/g, - replacement: "[REDACTED:jwt]", - }, - // ── Connection strings (creds embedded in URI) ───────────────────────── - { - name: "connection_string", - regex: /(?:mongodb(?:\+srv)?|postgres(?:ql)?|mysql|redis|amqp):\/\/[^:/@\s"']+:[^:/@\s"']+@/g, - replacement: "[REDACTED:connection_string]", - }, - // ── Header-style secrets ─────────────────────────────────────────────── - { - name: "auth_header", - regex: - /((?:["\x27]?(?:Authorization|x-api-key|api-key|apikey)["\x27]?\s*[:=]\s*["\x27]?)(?:(?:Bearer|Basic|Token)\s+)?)[A-Za-z0-9._~+/=-]{10,}/gi, - replacement: "$1[REDACTED:auth_header]", - }, -]; - export interface CredentialRedactionResult { text: string; detections: Array<{ type: string; count: number }>; diff --git a/src/lib/logPayloads.ts b/src/lib/logPayloads.ts index 5aa2f9eb9c..5fdef8c675 100644 --- a/src/lib/logPayloads.ts +++ b/src/lib/logPayloads.ts @@ -1,3 +1,8 @@ +import { + sanitizeErrorMessage, + sanitizeUpstreamDetails, +} from "@omniroute/open-sse/utils/errorSanitization.ts"; +import { projectResponsesFailureOutput } from "@omniroute/open-sse/utils/responsesFailureOutput.ts"; import { sanitizePII } from "./piiSanitizer"; const SENSITIVE_KEYS = new Set([ @@ -35,6 +40,21 @@ const SENSITIVE_KEYS = new Set([ "runtimeKey", ]); +const SENSITIVE_CHALLENGE_KEYS = new Set([ + "recaptchav3token", + "recaptchatoken", + "turnstiletoken", + "prooftoken", + "resumetoken", + "preparetoken", +]); + +function isSensitivePayloadKey(key: string): boolean { + if (SENSITIVE_KEYS.has(key)) return true; + const normalizedKey = key.replace(/[-_]/g, "").toLowerCase(); + return SENSITIVE_CHALLENGE_KEYS.has(normalizedKey); +} + type JsonRecord = Record; const ENCRYPTED_REASONING_KEY = "encrypted_content"; @@ -60,6 +80,283 @@ export function omitEncryptedReasoningFromLogChunks(chunks: string[]): string[] return found ? [omitted] : chunks; } +const ERROR_SUBTREE_KEYS = new Set([ + "error", + "errors", + "warning", + "warnings", + "errormessage", + "warningmessage", + "errordescription", + "warningdescription", + "lasterror", +]); + +function isErrorSubtreeKey(key: string): boolean { + return ERROR_SUBTREE_KEYS.has(key.replace(/[-_]/g, "").toLowerCase()); +} + +function sanitizeErrorSubtreeValue(value: unknown): unknown { + if (typeof value === "string") return sanitizeErrorMessage(value); + try { + if (value instanceof Error) { + return { + name: sanitizeErrorMessage(value.name) || "Error", + message: sanitizeErrorMessage(value.message), + }; + } + return sanitizeUpstreamDetails(value); + } catch { + return "[REDACTED]"; + } +} + +type ErrorSubtreeProjection = { value: unknown; found: boolean }; + +function projectErrorSubtreesForLog( + value: unknown, + seen = new WeakSet(), + forceResponsesFailure = false, + protocolResponseObject = false +): ErrorSubtreeProjection { + if (forceResponsesFailure && typeof value === "string") { + return { value: sanitizeErrorMessage(value) || "[REDACTED]", found: true }; + } + if (typeof value === "string") { + const trimmed = value.trim(); + if ( + (trimmed.startsWith("{") || trimmed.startsWith("[")) && + STREAM_ERROR_ENVELOPE_RE.test(trimmed) + ) { + try { + const parsed: unknown = JSON.parse(trimmed); + const projected = isDiscriminatedStreamError(parsed) + ? { value: sanitizeErrorSubtreeValue(parsed), found: true } + : projectErrorSubtreesForLog(parsed, seen); + if (projected.found) { + const serialized = JSON.stringify(projected.value); + if (typeof serialized === "string") return { value: serialized, found: true }; + } + } catch { + return { value: sanitizeErrorMessage(value) || "[REDACTED]", found: true }; + } + } + return { value, found: false }; + } + if (value === null || value === undefined || typeof value !== "object") { + return { value, found: false }; + } + if (isOpaqueBinary(value)) return { value, found: false }; + if (isDiscriminatedStreamError(value)) { + return { value: sanitizeErrorSubtreeValue(value), found: true }; + } + const declaresResponsesFailure = isResponsesFailureEvent(value); + const responsesFailure = forceResponsesFailure || declaresResponsesFailure; + if (seen.has(value)) return { value: "[circular]", found: false }; + seen.add(value); + + if (Array.isArray(value)) { + try { + let found = false; + const projected = value.map((entry) => { + const result = projectErrorSubtreesForLog(entry, seen, responsesFailure, false); + found ||= result.found; + return result.value; + }); + return { value: projected, found }; + } finally { + seen.delete(value); + } + } + + try { + let found = responsesFailure; + const projected: JsonRecord = {}; + for (const [key, entryValue] of Object.entries(value)) { + if (isErrorSubtreeKey(key) || (responsesFailure && isResponseFailureMessageKey(key))) { + projected[key] = sanitizeErrorSubtreeValue(entryValue); + found = true; + continue; + } + // Responses failures may attach diagnostics under neutral key names. Keep + // projecting through that envelope, while preserving partial model output + // as content rather than treating it as an error message. + const normalizedKey = key.replace(/[-_]/g, "").toLowerCase(); + const preservePartialOutput = + responsesFailure && + normalizedKey === "output" && + (protocolResponseObject || declaresResponsesFailure); + if (preservePartialOutput) { + projected[key] = projectResponsesFailureOutput( + entryValue, + (_field, stringValue) => sanitizeErrorMessage(stringValue) || "[REDACTED]" + ); + found = true; + continue; + } + const childIsProtocolResponse = + normalizedKey === "response" && + (declaresResponsesFailure || (forceResponsesFailure && !protocolResponseObject)); + const result = projectErrorSubtreesForLog( + entryValue, + seen, + responsesFailure, + childIsProtocolResponse + ); + projected[key] = result.value; + found ||= result.found; + } + return { value: projected, found }; + } catch { + return { value: "[REDACTED]", found: false }; + } finally { + seen.delete(value); + } +} + +const STREAM_ERROR_DISCRIMINATOR_KEYS = ["type", "event", "kind", "status"] as const; +const STREAM_ERROR_DISCRIMINATORS = new Set(["error", "warning"]); +const RESPONSES_FAILURE_DISCRIMINATORS = new Set(["response.failed"]); +const RESPONSE_FAILURE_MESSAGE_KEYS = new Set(["message", "detail", "details", "description"]); +const STREAM_ERROR_ENVELOPE_RE = + /["'](?:error|errors|warning|warnings|last_error|lastError|errorMessage|warningMessage)["']\s*:|["'](?:type|event|kind)["']\s*:\s*["'](?:error|warning|response\.(?:failed|completed))["']|["']status["']\s*:\s*["']failed["']/i; + +function isResponseFailureMessageKey(key: string): boolean { + return RESPONSE_FAILURE_MESSAGE_KEYS.has(key.replace(/[-_]/g, "").toLowerCase()); +} + +function isResponsesFailureEvent(value: unknown): boolean { + if (!value || typeof value !== "object" || Array.isArray(value)) return false; + try { + const record = value as JsonRecord; + const directFailure = STREAM_ERROR_DISCRIMINATOR_KEYS.some((key) => { + const discriminator = record[key]; + return ( + typeof discriminator === "string" && + RESPONSES_FAILURE_DISCRIMINATORS.has(discriminator.trim().toLowerCase()) + ); + }); + if (directFailure) return true; + + const status = record.status; + if (typeof status === "string" && status.trim().toLowerCase() === "failed") return true; + + const nestedResponse = record.response; + if (!nestedResponse || typeof nestedResponse !== "object" || Array.isArray(nestedResponse)) { + return false; + } + const nestedStatus = (nestedResponse as JsonRecord).status; + return typeof nestedStatus === "string" && nestedStatus.trim().toLowerCase() === "failed"; + } catch { + return true; + } +} + +function isDiscriminatedStreamError(value: unknown): boolean { + if (!value || typeof value !== "object" || Array.isArray(value)) return false; + try { + const record = value as JsonRecord; + return STREAM_ERROR_DISCRIMINATOR_KEYS.some((key) => { + const discriminator = record[key]; + return ( + typeof discriminator === "string" && + STREAM_ERROR_DISCRIMINATORS.has(discriminator.trim().toLowerCase()) + ); + }); + } catch { + return true; + } +} + +function sanitizeStreamErrorPayload( + rawPayload: string, + forceError: boolean, + forceResponsesFailure = false +): { found: boolean; value: string } { + try { + const parsed: unknown = JSON.parse(rawPayload); + if (forceError || isDiscriminatedStreamError(parsed)) { + const projected = sanitizeErrorSubtreeValue(parsed); + const serialized = JSON.stringify(projected); + return { + found: true, + value: typeof serialized === "string" ? serialized : "[REDACTED]", + }; + } + + const projected = projectErrorSubtreesForLog( + parsed, + new WeakSet(), + forceResponsesFailure + ); + if (!projected.found) return { found: false, value: rawPayload }; + return { found: true, value: JSON.stringify(projected.value) }; + } catch { + if (!forceError && !forceResponsesFailure && !STREAM_ERROR_ENVELOPE_RE.test(rawPayload)) { + return { found: false, value: rawPayload }; + } + return { + found: true, + value: sanitizeErrorMessage(rawPayload) || "[REDACTED]", + }; + } +} + +/** + * Sanitize error/warning records captured as fragmented SSE or NDJSON text. + * Prefixes are matched at the start of a line so unrelated `metadata:` fields + * cannot be mistaken for SSE `data:` frames. + */ +export function sanitizeErrorFramesFromLogChunks(chunks: string[]): string[] { + const combined = chunks.map((chunk) => chunk.replace(STREAM_CHUNK_TIMESTAMP_RE, "")).join(""); + let found = false; + let errorEventActive = false; + let responsesFailureEventActive = false; + const projectedLines = combined.split("\n").map((line) => { + if (line.trim().length === 0) { + errorEventActive = false; + responsesFailureEventActive = false; + return line; + } + + const eventMatch = line.match(/^\s*event:\s*([^\s]+)\s*$/i); + if (eventMatch) { + const eventName = eventMatch[1].toLowerCase(); + errorEventActive = STREAM_ERROR_DISCRIMINATORS.has(eventName); + responsesFailureEventActive = RESPONSES_FAILURE_DISCRIMINATORS.has(eventName); + return line; + } + + const dataMatch = line.match(/^(\s*data:)([ \t]?)(.*)$/); + if (dataMatch) { + const rawPayload = dataMatch[3].trim(); + if (!rawPayload || rawPayload === "[DONE]") return line; + const projected = sanitizeStreamErrorPayload( + rawPayload, + errorEventActive, + responsesFailureEventActive + ); + if (!projected.found) return line; + found = true; + return `${dataMatch[1]}${dataMatch[2]}${projected.value}`; + } + + if (errorEventActive || responsesFailureEventActive) { + found = true; + return sanitizeErrorMessage(line) || "[REDACTED]"; + } + + const trimmed = line.trim(); + if (!trimmed.startsWith("{") && !trimmed.startsWith("[")) return line; + const projected = sanitizeStreamErrorPayload(trimmed, false); + if (!projected.found) return line; + found = true; + return `${line.slice(0, line.length - line.trimStart().length)}${projected.value}`; + }); + + return found ? [projectedLines.join("\n")] : chunks; +} + /** * True for any binary/opaque byte view (Uint8Array, Buffer, DataView, other * typed arrays). `Array.isArray()` returns false for these, so callers that @@ -125,7 +422,7 @@ export function redactPayload(payload: unknown): unknown { const redacted: JsonRecord = {}; for (const [key, value] of Object.entries(payload)) { - if (SENSITIVE_KEYS.has(key)) { + if (isSensitivePayloadKey(key)) { redacted[key] = "[REDACTED]"; } else if (typeof value === "string" && value.startsWith("Bearer ")) { redacted[key] = "Bearer [REDACTED]"; @@ -162,7 +459,19 @@ export function sanitizePayloadPII(payload: unknown): unknown { export function protectPayloadForLog(payload: unknown): unknown { if (payload === null || payload === undefined) return null; const normalized = normalizePayloadForLog(payload); - const reasoningOmitted = omitEncryptedReasoningForLog(normalized); + const errorProjected = projectErrorSubtreesForLog(normalized).value; + const reasoningOmitted = omitEncryptedReasoningForLog(errorProjected); + const piiSanitized = sanitizePayloadPII(reasoningOmitted); + return redactPayload(piiSanitized); +} + +/** Project every string leaf because the payload is known to represent a failed response. */ +export function protectErrorPayloadForLog(payload: unknown): unknown { + if (payload === null || payload === undefined) return null; + const normalized = normalizePayloadForLog(payload); + if (isOpaqueBinary(normalized)) return describeOpaqueBinary(normalized); + const errorProjected = sanitizeErrorSubtreeValue(normalized); + const reasoningOmitted = omitEncryptedReasoningForLog(errorProjected); const piiSanitized = sanitizePayloadPII(reasoningOmitted); return redactPayload(piiSanitized); } diff --git a/src/lib/providers/validation/transport.ts b/src/lib/providers/validation/transport.ts index cbf6686aa9..c9472feb5c 100644 --- a/src/lib/providers/validation/transport.ts +++ b/src/lib/providers/validation/transport.ts @@ -1,6 +1,7 @@ // Outbound fetch wrappers for provider validation: proxy-fallback, SSRF-aware proxy targeting, and -// error→result mapping. Extracted from validation.ts (god-file decomposition). Behavior is -// byte-identical to the original inline defs. +// error→result mapping. Extracted from validation.ts (god-file decomposition) and kept as the +// common boundary for sanitizing validation failures. +import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/errorSanitization.ts"; import { SAFE_OUTBOUND_FETCH_PRESETS, SafeOutboundFetchError, @@ -11,6 +12,28 @@ import { isPrivateHost } from "@/shared/network/outboundUrlGuard"; import { getProviderValidationGuard } from "@/shared/network/outboundUrlGuardPolicy"; import { selectProxyForValidation } from "@omniroute/open-sse/services/proxyAutoSelector.ts"; +export type ProjectedProviderValidationResult = { + [K in keyof T]: K extends "error" | "warning" ? string | null : T[K]; +} & { + error?: string | null; + warning?: string | null; +}; + +export function projectProviderValidationResultForPublicResponse< + T extends { error?: unknown; warning?: unknown }, +>(result: T): ProjectedProviderValidationResult; +export function projectProviderValidationResultForPublicResponse( + result: Record +): Record { + const projected: Record = { ...result }; + for (const field of ["error", "warning"] as const) { + if (!Object.prototype.hasOwnProperty.call(result, field)) continue; + const value = result[field]; + projected[field] = value === null || value === undefined ? null : sanitizeErrorMessage(value); + } + return projected; +} + /** * Wrapped fetch call that auto-retries with a proxy when the direct connection * fails. This happens transparently so individual validators don't need to @@ -156,17 +179,30 @@ export function toWebCookieValidationErrorResult(provider: string, error: unknow } export function toValidationErrorResult(error: unknown) { - const message = error instanceof Error ? error.message : String(error || "Validation failed"); - const statusCode = getSafeOutboundFetchErrorStatus(error); + let rawMessage: unknown = error || "Validation failed"; + try { + if (error instanceof Error) rawMessage = error.message; + } catch { + rawMessage = "Validation failed"; + } + const message = sanitizeErrorMessage(rawMessage); + let statusCode: number | null = null; + let timeout = false; + let securityBlocked = false; + try { + statusCode = getSafeOutboundFetchErrorStatus(error); + timeout = error instanceof SafeOutboundFetchError && error.code === "TIMEOUT"; + securityBlocked = isSecurityBlockError(error); + } catch { + // Classification is advisory; hostile accessors must not escape the safe error boundary. + } return { valid: false, error: message || "Validation failed", unsupported: false as const, ...(statusCode ? { statusCode } : {}), - ...(error instanceof SafeOutboundFetchError && error.code === "TIMEOUT" - ? { timeout: true } - : {}), - ...(isSecurityBlockError(error) ? { securityBlocked: true } : {}), + ...(timeout ? { timeout: true } : {}), + ...(securityBlocked ? { securityBlocked: true } : {}), }; } diff --git a/src/lib/proxyLogger.ts b/src/lib/proxyLogger.ts index 8665e20c75..0bb278aa0f 100644 --- a/src/lib/proxyLogger.ts +++ b/src/lib/proxyLogger.ts @@ -7,6 +7,7 @@ * Pattern follows callLogs.js (T-15 decomposition). */ import { v4 as uuidv4 } from "uuid"; +import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/errorSanitization.ts"; import { getDbInstance, isCloud, isBuildPhase } from "./db/core"; import { ensureProxyLogsColumns } from "./db/schemaColumns"; @@ -99,7 +100,10 @@ function loadFromDb() { console.log(`[proxyLogger] Loaded ${proxyLogs.length} proxy logs from SQLite`); } } catch (err: any) { - console.warn("[proxyLogger] Failed to load from DB:", err.message); + console.warn( + "[proxyLogger] Failed to load from DB:", + sanitizeErrorMessage(err) || "Proxy log hydration failed" + ); } } @@ -113,10 +117,7 @@ loadFromDb(); /** Read at call time so tests can toggle it between imports. */ export function isProxyLogIncludeIps(): boolean { - return ( - process.env.PROXY_LOG_INCLUDE_IPS === "true" || - process.env.PROXY_LOG_INCLUDE_IPS === "1" - ); + return process.env.PROXY_LOG_INCLUDE_IPS === "true" || process.env.PROXY_LOG_INCLUDE_IPS === "1"; } /** @@ -152,6 +153,10 @@ export function formatProxyEgressConsoleLine(params: { // ──────────────── Log a proxy event ──────────────── export function logProxyEvent(entry: ProxyLogInput) { + const safeError = + entry.error === null || entry.error === undefined || entry.error === "" + ? null + : sanitizeErrorMessage(entry.error) || "Proxy request failed"; const log: ProxyLogEntry = { id: uuidv4(), timestamp: new Date().toISOString(), @@ -164,7 +169,7 @@ export function logProxyEvent(entry: ProxyLogInput) { clientIp: entry.clientIp ?? entry.publicIp ?? null, egressIp: entry.egressIp ?? null, latencyMs: entry.latencyMs || 0, - error: entry.error || null, + error: safeError, connectionId: entry.connectionId || null, comboId: entry.comboId || null, account: entry.account || null, @@ -236,15 +241,17 @@ export function flushProxyLogsSync() { // 1. If Redis driver is active, asynchronously publish batch to Redis Stream/Channel if (process.env.QUOTA_STORE_DRIVER === "redis" || process.env.QUOTA_STORE_REDIS_URL) { try { - import("@/lib/quota/redisQuotaStore").then(({ getRedisQuotaStore }) => { - const store = getRedisQuotaStore(process.env.QUOTA_STORE_REDIS_URL || ""); - const client = (store as any)?.client; - if (client && typeof client.publish === "function") { - for (const entry of batch) { - client.publish("omniroute:proxy_logs", JSON.stringify(entry)).catch(() => {}); + import("@/lib/quota/redisQuotaStore") + .then(({ getRedisQuotaStore }) => { + const store = getRedisQuotaStore(process.env.QUOTA_STORE_REDIS_URL || ""); + const client = (store as any)?.client; + if (client && typeof client.publish === "function") { + for (const entry of batch) { + client.publish("omniroute:proxy_logs", JSON.stringify(entry)).catch(() => {}); + } } - } - }).catch(() => {}); + }) + .catch(() => {}); } catch { /* ignore redis pub errors */ } @@ -289,7 +296,10 @@ export function flushProxyLogsSync() { transaction(batch); } catch (err: any) { - console.warn("[proxyLogger] Failed to write proxy log batch to disk:", err?.message || err); + console.warn( + "[proxyLogger] Failed to write proxy log batch to disk:", + sanitizeErrorMessage(err) || "Proxy log persistence failed" + ); } } @@ -351,7 +361,10 @@ export function clearProxyLogs() { const db = getDbInstance(); db.prepare("DELETE FROM proxy_logs").run(); } catch (err: any) { - console.warn("[proxyLogger] Failed to clear DB:", err.message); + console.warn( + "[proxyLogger] Failed to clear DB:", + sanitizeErrorMessage(err) || "Proxy log cleanup failed" + ); } } } diff --git a/src/lib/skills/executor.ts b/src/lib/skills/executor.ts index 692716d485..ac958f1a54 100644 --- a/src/lib/skills/executor.ts +++ b/src/lib/skills/executor.ts @@ -1,3 +1,8 @@ +import { + sanitizeErrorMessage, + sanitizeUpstreamDetails, +} from "@omniroute/open-sse/utils/errorSanitization.ts"; + import { skillRegistry } from "./registry"; import { SkillExecution, SkillStatus, SkillHandler } from "./types"; import { builtinSkills } from "./builtins"; @@ -8,6 +13,169 @@ import { logger } from "../../../open-sse/utils/logger.ts"; const log = logger("SKILLS_EXECUTOR"); +function toSafeSkillErrorMessage(value: unknown): string { + try { + const raw = value instanceof Error ? value.message : value; + return sanitizeErrorMessage(raw) || "Skill execution failed"; + } catch { + return "Skill execution failed"; + } +} + +const SKILL_FAILURE_DISCRIMINATORS = new Set(["error", "failed", "failure"]); + +function isSkillErrorKey(key: string): boolean { + const normalizedKey = key.replace(/[-_]/g, "").toLowerCase(); + return ( + normalizedKey === "error" || + normalizedKey === "errors" || + normalizedKey === "warning" || + normalizedKey === "warnings" + ); +} + +function isFailureDiscriminator(value: unknown): boolean { + return typeof value === "string" && SKILL_FAILURE_DISCRIMINATORS.has(value.trim().toLowerCase()); +} + +function isSkillFailureOutput(output: Record): boolean { + try { + const status = output.status; + return ( + output.success === false || + (typeof status === "number" && Number.isFinite(status) && status >= 400) || + isFailureDiscriminator(status) || + isFailureDiscriminator(output.type) || + isFailureDiscriminator(output.event) || + isFailureDiscriminator(output.kind) + ); + } catch { + return true; + } +} + +type SensitiveSkillReferences = { + objects: WeakSet; + strings: Set; +}; + +function markSensitiveSkillReference(value: unknown, sensitive: SensitiveSkillReferences): void { + if (typeof value === "string") { + sensitive.strings.add(value); + return; + } + if (!value || typeof value !== "object" || sensitive.objects.has(value)) return; + + sensitive.objects.add(value); + try { + for (const entry of Object.values(value as Record)) { + markSensitiveSkillReference(entry, sensitive); + } + } catch { + // A revoked proxy or throwing getter is unsafe to expose at the boundary. + } +} + +function collectSensitiveSkillReferences( + value: unknown, + sensitive: SensitiveSkillReferences, + visited: WeakSet +): void { + if (!value || typeof value !== "object" || visited.has(value)) return; + visited.add(value); + + try { + for (const [key, entry] of Object.entries(value as Record)) { + if (isSkillErrorKey(key)) { + markSensitiveSkillReference(entry, sensitive); + } else { + collectSensitiveSkillReferences(entry, sensitive, visited); + } + } + } catch { + markSensitiveSkillReference(value, sensitive); + } +} + +type SkillProjectionContext = { + active: WeakSet; + projected: WeakMap; + sensitive: SensitiveSkillReferences; +}; + +function projectNestedSkillErrorSubtrees(value: unknown, context: SkillProjectionContext): unknown { + if (typeof value === "string") { + return context.sensitive.strings.has(value) ? sanitizeErrorMessage(value) : value; + } + if (!value || typeof value !== "object") return value; + if (context.active.has(value)) return "[circular]"; + if (context.projected.has(value)) return context.projected.get(value); + + if (context.sensitive.objects.has(value)) { + const safeValue = sanitizeUpstreamDetails(value); + context.projected.set(value, safeValue); + return safeValue; + } + + context.active.add(value); + if (Array.isArray(value)) { + const projected: unknown[] = []; + context.projected.set(value, projected); + for (const entry of value) projected.push(projectNestedSkillErrorSubtrees(entry, context)); + context.active.delete(value); + return projected; + } + + const projected: Record = {}; + context.projected.set(value, projected); + for (const [key, entry] of Object.entries(value as Record)) { + projected[key] = isSkillErrorKey(key) + ? sanitizeUpstreamDetails(entry) + : projectNestedSkillErrorSubtrees(entry, context); + } + context.active.delete(value); + return projected; +} + +function skillFailureMessage(output: Record): string { + try { + for (const candidate of [output.message, output.reason, output.statusText, output.error]) { + if (typeof candidate === "string" || candidate instanceof Error) { + return toSafeSkillErrorMessage(candidate); + } + } + } catch { + // Fall through to the stable public message. + } + return "Skill execution failed"; +} + +export function projectSkillOutputForBoundary( + output: Record +): Record { + try { + if (isSkillFailureOutput(output)) { + const projected = sanitizeUpstreamDetails(output); + return projected && typeof projected === "object" && !Array.isArray(projected) + ? (projected as Record) + : { success: false, error: "Skill execution failed" }; + } + + const sensitive: SensitiveSkillReferences = { + objects: new WeakSet(), + strings: new Set(), + }; + collectSensitiveSkillReferences(output, sensitive, new WeakSet()); + return projectNestedSkillErrorSubtrees(output, { + active: new WeakSet(), + projected: new WeakMap(), + sensitive, + }) as Record; + } catch { + return { success: false, error: "Skill execution failed" }; + } +} + class SkillExecutor { private static instance: SkillExecutor; private handlers: Map = new Map(); @@ -99,9 +267,14 @@ class SkillExecutor { const result = await this.executeWithTimeout( handler(input, { apiKeyId: context.apiKeyId, sessionId: context.sessionId || "" }) ); - output = result; + const resultIsFailure = isSkillFailureOutput(result); + output = projectSkillOutputForBoundary(result); + if (resultIsFailure) { + errorMessage = skillFailureMessage(result); + status = SkillStatus.ERROR; + } } catch (err) { - errorMessage = err instanceof Error ? err.message : String(err); + errorMessage = toSafeSkillErrorMessage(err); status = SkillStatus.ERROR; } @@ -131,7 +304,7 @@ class SkillExecutor { }; } catch (err) { const durationMs = Date.now() - startTime; - const errorMessage = err instanceof Error ? err.message : String(err); + const errorMessage = toSafeSkillErrorMessage(err); db.prepare( `UPDATE skill_executions SET status = ?, error_message = ?, duration_ms = ? WHERE id = ?` diff --git a/src/lib/skills/interception.ts b/src/lib/skills/interception.ts index 16b0146728..43c83c2d25 100644 --- a/src/lib/skills/interception.ts +++ b/src/lib/skills/interception.ts @@ -1,14 +1,29 @@ -import { skillExecutor } from "./executor"; +import { projectSkillOutputForBoundary, skillExecutor } from "./executor"; import { skillRegistry } from "./registry"; import { builtinSkills } from "./builtins"; import { memoryBuiltinHandlers, MEMORY_BUILTIN_TOOL_NAMES } from "./memoryBuiltins"; import { detectProvider, decodeSkillToolName } from "./injection"; import { OMNIROUTE_WEB_SEARCH_FALLBACK_TOOL_NAME } from "@omniroute/open-sse/services/webSearchFallback.ts"; import { OMNIROUTE_WEB_FETCH_FALLBACK_TOOL_NAME } from "@omniroute/open-sse/services/webFetchInterception.ts"; +import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/errorSanitization.ts"; import { logger } from "../../../open-sse/utils/logger.ts"; const log = logger("SKILLS_INTERCEPTION"); +function toSafeSkillErrorMessage(value: unknown): string { + try { + const raw = value instanceof Error ? value.message : value; + return sanitizeErrorMessage(raw) || "Skill execution failed"; + } catch { + return "Skill execution failed"; + } +} + +function projectSkillResultForPublicResponse(result: unknown): unknown { + if (!result || typeof result !== "object" || Array.isArray(result)) return result; + return projectSkillOutputForBoundary(result as Record); +} + interface ToolCall { id: string; name: string; @@ -130,7 +145,7 @@ export async function interceptToolCalls( return { id: call.id, - result, + result: projectSkillResultForPublicResponse(result), }; } @@ -151,11 +166,12 @@ export async function interceptToolCalls( sessionId: context.sessionId, }); - const result = + const result = projectSkillResultForPublicResponse( execution.output ?? - (execution.errorMessage - ? { error: execution.errorMessage } - : { error: "Skill execution returned no output" }); + (execution.errorMessage + ? { error: toSafeSkillErrorMessage(execution.errorMessage) } + : { error: "Skill execution returned no output" }) + ); log.info("skills.interception.execution_complete", { toolName: call.name, @@ -167,14 +183,15 @@ export async function interceptToolCalls( result, }; } catch (err) { + const safeError = toSafeSkillErrorMessage(err); log.error("skills.interception.execution_failed", { toolName: call.name, callId: call.id, - err: err instanceof Error ? err.message : String(err), + err: safeError, }); return { id: call.id, - result: { error: err instanceof Error ? err.message : String(err) }, + result: { error: safeError }, }; } }) diff --git a/src/lib/usage/callLogs.ts b/src/lib/usage/callLogs.ts index 43a5028b9d..5f0e3a03fc 100644 --- a/src/lib/usage/callLogs.ts +++ b/src/lib/usage/callLogs.ts @@ -8,6 +8,7 @@ import fs from "node:fs"; import path from "node:path"; import type { RequestPipelinePayloads } from "@omniroute/open-sse/utils/requestLogger.ts"; +import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/errorSanitization.ts"; import { getDbInstance } from "../db/core"; import { getRequestDetailLogByCallLogId } from "../db/detailedLogs"; import { shouldPersistToDisk } from "./migrations"; @@ -21,7 +22,11 @@ import { getObservedReasoning, } from "./tokenAccounting"; import { isNoLog } from "../compliance/noLog"; -import { protectPayloadForLog, parseStoredPayload } from "../logPayloads"; +import { + parseStoredPayload, + protectErrorPayloadForLog, + protectPayloadForLog, +} from "../logPayloads"; import { pickDisplayValue } from "@/shared/utils/maskEmail"; import { CALL_LOGS_DIR, @@ -335,7 +340,10 @@ function readLegacyLogFromDisk(entry: { return JSON.parse(fs.readFileSync(path.join(dir, files[0]), "utf8")); } } catch (error) { - console.error("[callLogs] Failed to read legacy disk log:", (error as Error).message); + console.error( + "[callLogs] Failed to read legacy disk log:", + sanitizeErrorMessage(error) || "Legacy call log read failed" + ); } return null; @@ -447,10 +455,19 @@ async function saveCallLogOperation(entry: any): Promise { const noLogEnabled = Boolean(entry.noLog) || (apiKeyId ? isNoLog(apiKeyId) : false); const protectedRequestBody = noLogEnabled ? null : protectPayloadForLog(entry.requestBody); - const protectedResponseBody = noLogEnabled ? null : protectPayloadForLog(entry.responseBody); + const responseStatus = Number(entry.status); + const failedResponse = Number.isFinite(responseStatus) && responseStatus >= 400; + const protectedResponseBody = noLogEnabled + ? null + : failedResponse + ? protectErrorPayloadForLog(entry.responseBody) + : protectPayloadForLog(entry.responseBody); const protectedPipelinePayloads = noLogEnabled ? null - : protectPipelinePayloads(entry.pipelinePayloads ?? entry.pipeline ?? null); + : protectPipelinePayloads( + entry.pipelinePayloads ?? entry.pipeline ?? null, + failedResponse ? responseStatus : undefined + ); const protectedError = sanitizeErrorForLog(entry.error); const account = await resolveAccountName(entry.connectionId || null); @@ -582,7 +599,10 @@ async function saveCallLogOperation(entry: any): Promise { scheduleCallLogRotation(); } catch (error) { - console.error("[callLogs] Failed to save call log:", (error as Error).message); + console.error( + "[callLogs] Failed to save call log:", + sanitizeErrorMessage(error) || "Call log persistence failed" + ); } } diff --git a/src/lib/usage/callLogs/format.ts b/src/lib/usage/callLogs/format.ts index 40054c3a37..63068e73bc 100644 --- a/src/lib/usage/callLogs/format.ts +++ b/src/lib/usage/callLogs/format.ts @@ -1,7 +1,16 @@ import type { RequestPipelinePayloads } from "@omniroute/open-sse/utils/requestLogger.ts"; import { classifyProviderError } from "@omniroute/open-sse/services/errorClassifier.ts"; +import { + sanitizeErrorMessage, + sanitizeUpstreamDetails, +} from "@omniroute/open-sse/utils/errorSanitization.ts"; import { sanitizePII } from "../../piiSanitizer"; -import { omitEncryptedReasoningFromLogChunks, protectPayloadForLog } from "../../logPayloads"; +import { + omitEncryptedReasoningFromLogChunks, + protectErrorPayloadForLog, + protectPayloadForLog, + sanitizeErrorFramesFromLogChunks, +} from "../../logPayloads"; import type { CallLogDetailState } from "../callLogArtifacts"; // #7879: re-export the canonical helper so existing consumers of this module // keep importing `toNumber` from here unchanged. @@ -44,15 +53,24 @@ export function normalizeDetailState(value: unknown): CallLogDetailState { export function sanitizeErrorForLog(error: unknown): unknown { if (error === null || error === undefined) return null; - if (typeof error === "string") return sanitizePII(error).text; - if (error instanceof Error) { - return { - message: sanitizePII(error.message).text, - stack: sanitizePII(error.stack || "").text || undefined, - name: error.name, - }; + if (typeof error === "string") { + return sanitizePII(sanitizeErrorMessage(error)).text; + } + try { + if (error instanceof Error) { + const message = sanitizePII(sanitizeErrorMessage(error.message)).text; + const stack = sanitizePII(sanitizeErrorMessage(error.stack || "")).text; + const name = sanitizeErrorMessage(error.name) || "Error"; + return { + message, + ...(stack ? { stack } : {}), + name, + }; + } + return protectPayloadForLog(sanitizeUpstreamDetails(error)); + } catch { + return "[REDACTED]"; } - return protectPayloadForLog(error); } export function toStoredErrorSummary(error: unknown): string | null { @@ -70,7 +88,10 @@ export function toStoredErrorSummary(error: unknown): string | null { } } -export function protectPipelinePayloads(payloads: unknown): RequestPipelinePayloads | null { +export function protectPipelinePayloads( + payloads: unknown, + responseStatus?: unknown +): RequestPipelinePayloads | null { if (!payloads || typeof payloads !== "object") return null; const protectedPayloads: RequestPipelinePayloads = {}; @@ -84,7 +105,9 @@ export function protectPipelinePayloads(payloads: unknown): RequestPipelinePaylo .filter(([, chunkValue]) => Array.isArray(chunkValue) && chunkValue.length > 0) .map(([stage, chunkValue]) => [ stage, - omitEncryptedReasoningFromLogChunks(chunkValue as string[]), + sanitizeErrorFramesFromLogChunks( + omitEncryptedReasoningFromLogChunks(chunkValue as string[]) + ), ]) ); if (Object.keys(compacted).length > 0) { @@ -95,6 +118,21 @@ export function protectPipelinePayloads(payloads: unknown): RequestPipelinePaylo continue; } + if (key === "providerResponse" || key === "clientResponse") { + const response = asRecord(value); + const status = Number(response.status ?? responseStatus); + if (Number.isFinite(status) && status >= 400 && status <= 599) { + const projectedResponse = + "body" in response + ? { ...response, body: protectErrorPayloadForLog(response.body) } + : protectErrorPayloadForLog(value); + protectedPayloads[key as "providerResponse" | "clientResponse"] = protectPayloadForLog( + projectedResponse + ) as RequestPipelinePayloads["providerResponse"]; + continue; + } + } + protectedPayloads[key as keyof RequestPipelinePayloads] = protectPayloadForLog(value) as never; } diff --git a/src/lib/usage/usageHistory.ts b/src/lib/usage/usageHistory.ts index 3d9b0dfa68..4a1a9f9216 100644 --- a/src/lib/usage/usageHistory.ts +++ b/src/lib/usage/usageHistory.ts @@ -9,6 +9,7 @@ import { getDbInstance } from "../db/core"; import { protectPayloadForLog } from "../logPayloads"; +import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/errorSanitization.ts"; import { resolveOrphanedUsageAccountIdentity, resolveUsageAccountIdentity, @@ -128,7 +129,7 @@ function normalizePendingMetadata(metadata?: PendingRequestMetadata): PendingReq normalized.status = Number.isFinite(status) ? status : null; } if (metadata.error !== undefined) { - normalized.error = toStringOrNull(metadata.error) || null; + normalized.error = sanitizeErrorMessage(toStringOrNull(metadata.error)) || null; } if (metadata.errorCode !== undefined) { normalized.errorCode = toStringOrNull(metadata.errorCode) || null; diff --git a/src/lib/usage/usageStats.ts b/src/lib/usage/usageStats.ts index 2bc459fef7..833eb6ef31 100644 --- a/src/lib/usage/usageStats.ts +++ b/src/lib/usage/usageStats.ts @@ -318,6 +318,10 @@ export async function getUsageStats() { } const pendingRequests = getPendingRequests(); + const publicPendingRequests = { + byModel: pendingRequests.byModel, + byAccount: pendingRequests.byAccount, + }; const stats: { totalRequests: number; @@ -329,7 +333,7 @@ export async function getUsageStats() { byAccount: Record; byApiKey: Record; last10Minutes: UsageBucket[]; - pending: ReturnType; + pending: Pick, "byModel" | "byAccount">; activeRequests: ActiveRequest[]; } = { totalRequests: 0, @@ -341,7 +345,7 @@ export async function getUsageStats() { byAccount: {}, byApiKey: {}, last10Minutes: [], - pending: pendingRequests, + pending: publicPendingRequests, activeRequests: [], }; diff --git a/src/shared/utils/apiKeyPolicy.ts b/src/shared/utils/apiKeyPolicy.ts index 49cb628bb0..67d9b4b70e 100644 --- a/src/shared/utils/apiKeyPolicy.ts +++ b/src/shared/utils/apiKeyPolicy.ts @@ -254,8 +254,7 @@ async function isComboAllowedForKey( } function quotaPolicyResponse(message: string, code: string): Response { - const body = buildErrorBody(HTTP_STATUS.FORBIDDEN, message); - body.error.code = code; + const body = buildErrorBody(HTTP_STATUS.FORBIDDEN, message, undefined, { code }); return new Response(JSON.stringify(body), { status: HTTP_STATUS.FORBIDDEN, headers: { "Content-Type": "application/json" }, diff --git a/src/shared/utils/terminalStatus.ts b/src/shared/utils/terminalStatus.ts index 1b74768b9a..b2e46ed614 100644 --- a/src/shared/utils/terminalStatus.ts +++ b/src/shared/utils/terminalStatus.ts @@ -1,17 +1,33 @@ import { updateProviderConnection } from "@/lib/db/providers"; import { shouldIsolateProbeFailures } from "@/shared/utils/probeOrigin"; +import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/errorSanitization.ts"; -type Patch = { testStatus: string; isActive?: boolean; lastError?: string | null; errorCode?: string | null; lastErrorType?: string | null; lastErrorAt?: string | null }; -const TERMINAL = new Set(["banned","expired","deactivated","credits_exhausted"]); +type Patch = { + testStatus: string; + isActive?: boolean; + lastError?: string | null; + errorCode?: string | null; + lastErrorType?: string | null; + lastErrorAt?: string | null; +}; +const TERMINAL = new Set(["banned", "expired", "deactivated", "credits_exhausted"]); -export async function writeTerminalStatus(connectionId: string, patch: Patch, origin: "probe" | "production"): Promise { +export async function writeTerminalStatus( + connectionId: string, + patch: Patch, + origin: "probe" | "production" +): Promise { const isTerminal = TERMINAL.has(patch.testStatus.toLowerCase()); + const persistedLastError = + patch.lastError == null + ? null + : sanitizeErrorMessage(patch.lastError) || "Provider request failed"; // Double gate: AsyncLocalStorage probe + explicit origin "probe" — fail-safe ON const probeIsolated = await shouldIsolateProbeFailures(); if ((origin === "probe" || probeIsolated) && isTerminal) { // record-only: never remove from pool await updateProviderConnection(connectionId, { - lastError: patch.lastError ?? null, + lastError: persistedLastError, lastErrorAt: new Date().toISOString(), lastErrorType: patch.lastErrorType ?? null, errorCode: patch.errorCode ?? null, @@ -21,7 +37,7 @@ export async function writeTerminalStatus(connectionId: string, patch: Patch, or await updateProviderConnection(connectionId, { isActive: patch.isActive ?? (isTerminal ? false : undefined), testStatus: patch.testStatus, - lastError: patch.lastError ?? null, + lastError: persistedLastError, lastErrorAt: new Date().toISOString(), lastErrorType: patch.lastErrorType ?? null, errorCode: patch.errorCode ?? null, diff --git a/src/sse/services/auth.ts b/src/sse/services/auth.ts index b121876376..37029593af 100644 --- a/src/sse/services/auth.ts +++ b/src/sse/services/auth.ts @@ -73,6 +73,7 @@ import { } from "@omniroute/open-sse/services/accountFallback.ts"; import { isLocalProvider } from "@omniroute/open-sse/config/providerRegistry.ts"; import { COOLDOWN_MS, RateLimitReason } from "@omniroute/open-sse/config/constants.ts"; +import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/errorSanitization.ts"; import { honorsRuleLockScope, isEgressBucketedLockScope, @@ -2717,14 +2718,13 @@ export async function markAccountUnavailable( // the opt-in setting probeCanDisable restores the historical behavior. if (await shouldIsolateProbeFailures()) { await updateProviderConnection(connectionId, { - // lastError kept RAW (full text) — maximal probe visibility; the - // divergence vs the normal path's slice(0,100) is intentional. + // Persist safe wording only after classification has consumed the raw provider text. // backoffLevel is deliberately NOT written: a positive backoff // triggers the selection-time auto-decay (resetConnectionBackoff, // auth.ts getProviderCredentials) which wipes lastError back to // NULL on the next attempt — silently destroying the probe record. // The backoff is also routing state a probe must not touch (#9817). - lastError: errorText, + lastError: sanitizeErrorMessage(errorText) || "Provider request failed", lastErrorType: fallbackResult.reason || null, errorCode: status, lastErrorAt: new Date().toISOString(), @@ -3140,8 +3140,8 @@ export async function markAccountUnavailable( ); return { shouldFallback: true, cooldownMs: lockout.cooldownMs }; } - - const errorMsg = describeUpstreamFailure(errorText); + const errorMsg = + sanitizeErrorMessage(describeUpstreamFailure(errorText)) || "Provider request failed"; // T09: Codex per-scope lockout (do not block the whole account globally). if ( diff --git a/tests/unit/calllogs-format-split.test.ts b/tests/unit/calllogs-format-split.test.ts index a88f86c960..4bb267bfab 100644 --- a/tests/unit/calllogs-format-split.test.ts +++ b/tests/unit/calllogs-format-split.test.ts @@ -17,7 +17,6 @@ import { describe, it } from "node:test"; import assert from "node:assert/strict"; - import { asRecord, toNumber, @@ -96,6 +95,15 @@ describe("callLogs/format — toStoredErrorSummary", () => { assert.ok(out.includes("kaboom")); assert.ok(out.includes("message")); }); + it("removes credentials, filesystem paths, and stack frames before persistence", () => { + const out = toStoredErrorSummary( + "Provider failed access_token=persisted-secret at /srv/private/provider.json\n" + + " at dispatch (/srv/private/dispatcher.ts:42:7)" + ); + + assert.equal(typeof out, "string"); + assert.doesNotMatch(out, /persisted-secret|srv\/private|dispatcher\.ts|\bat dispatch\b/i); + }); }); describe("callLogs/format — buildRequestSummary", () => { diff --git a/tests/unit/chatcore-stream-error-result.test.ts b/tests/unit/chatcore-stream-error-result.test.ts index 352877046a..3ebe5821e7 100644 --- a/tests/unit/chatcore-stream-error-result.test.ts +++ b/tests/unit/chatcore-stream-error-result.test.ts @@ -43,6 +43,23 @@ test("createStreamingErrorResult attaches optional code and type", async () => { assert.equal(json.error.type, "rate_limit_error"); }); +test("createStreamingErrorResult sanitizes code and type at the SSE boundary", async () => { + const result = createStreamingErrorResult( + 502, + "upstream failed", + "sk-live-secret-value", + "server_error\nX-Leak: yes" + ); + const body = await result.response.text(); + const json = JSON.parse(body.slice("data: ".length, body.indexOf("\n\n"))) as { + error: { code: string; type: string }; + }; + + assert.equal(json.error.code, "bad_gateway"); + assert.equal(json.error.type, "server_error"); + assert.doesNotMatch(body, /sk-live-secret-value|X-Leak/); +}); + test("getUpstreamErrorIdentifier returns a non-empty string code or undefined", () => { assert.equal(getUpstreamErrorIdentifier({ code: "ECONNRESET" }), "ECONNRESET"); assert.equal(getUpstreamErrorIdentifier({ code: "" }), undefined); diff --git a/tests/unit/chatcore-translation-paths.test.ts b/tests/unit/chatcore-translation-paths.test.ts index 861f716251..c0125fff83 100644 --- a/tests/unit/chatcore-translation-paths.test.ts +++ b/tests/unit/chatcore-translation-paths.test.ts @@ -4,8 +4,15 @@ import assert from "node:assert/strict"; import fs from "node:fs"; import os from "node:os"; import path from "node:path"; -const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-chatcore-translation-")); +const TEST_ROOT = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-chatcore-translation-")); +const TEST_DATA_DIR = path.join(TEST_ROOT, "data"); +const TEST_PLUGINS_DIR = path.join(TEST_ROOT, "plugins"); +const ORIGINAL_DATA_DIR = process.env.DATA_DIR; +const ORIGINAL_PLUGINS_DIR = process.env.OMNIROUTE_PLUGINS_DIR; +fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +fs.mkdirSync(TEST_PLUGINS_DIR, { recursive: true }); process.env.DATA_DIR = TEST_DATA_DIR; +process.env.OMNIROUTE_PLUGINS_DIR = TEST_PLUGINS_DIR; const core = await import("../../src/lib/db/core.ts"); const providersDb = await import("../../src/lib/db/providers.ts"); const settingsDb = await import("../../src/lib/db/settings.ts"); @@ -448,7 +455,11 @@ test.after(async () => { resetAccountSemaphores(); await flushAsyncSideEffects(); await resetStorage(); - fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + if (ORIGINAL_DATA_DIR === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = ORIGINAL_DATA_DIR; + if (ORIGINAL_PLUGINS_DIR === undefined) delete process.env.OMNIROUTE_PLUGINS_DIR; + else process.env.OMNIROUTE_PLUGINS_DIR = ORIGINAL_PLUGINS_DIR; + fs.rmSync(TEST_ROOT, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); }); test("chatCore times out upstream execution before provider response headers", async () => { // This test asserts pendingDetail.providerRequest — only attached when the @@ -1938,35 +1949,12 @@ test("chatCore surfaces translation errors with explicit status codes", async () FORMATS.OPENAI_RESPONSES, FORMATS.OPENAI, () => { - const error = new Error("responses translator rejected the payload"); - error.statusCode = 409; - throw error; - }, - null - ); - - const { result } = await invokeChatCore({ - provider: "openai", - model: "gpt-4o-mini", - endpoint: "/v1/responses", - body: { - model: "gpt-4o-mini", - input: "hello", - }, - }); - - assert.equal(result.success, false); - assert.equal(result.status, 409); - assert.equal(result.error, "responses translator rejected the payload"); -}); -test("chatCore surfaces typed translation errors with the declared error type", async () => { - register( - FORMATS.OPENAI_RESPONSES, - FORMATS.OPENAI, - () => { - const error = new Error("typed translator failure"); + const error = new Error( + "translator rejected access_token=translation-secret at /srv/private/translator.ts\n" + + " at translate (/srv/private/translator.ts:41:8)" + ); error.statusCode = 422; - error.errorType = "unsupported_feature"; + error.errorType = "unsupported_feature access_token=type-secret /srv/private/type.ts"; throw error; }, null @@ -1984,10 +1972,16 @@ test("chatCore surfaces typed translation errors with the declared error type", assert.equal(result.success, false); assert.equal(result.status, 422); - - const payload = (await result.response.json()) as any; - assert.equal(payload.error.type, "unsupported_feature"); - assert.equal(payload.error.code, "unsupported_feature"); + const payload = (await result.response.json()) as { + error: { message: string; type: string; code: string }; + }; + assert.equal(payload.error.type, "invalid_request_error"); + assert.equal(payload.error.code, ""); + assert.match(payload.error.message, /translator rejected/); + assert.doesNotMatch( + JSON.stringify({ payload, internalError: result.error }), + /translation-secret|type-secret|srv\/private|translator\.ts|type\.ts|\bat translate\b/i + ); }); test("chatCore returns 500 when translation throws a generic error", async () => { register( diff --git a/tests/unit/combo-diagnostics-trace.test.ts b/tests/unit/combo-diagnostics-trace.test.ts index fe8bb546a4..dc1354f7ca 100644 --- a/tests/unit/combo-diagnostics-trace.test.ts +++ b/tests/unit/combo-diagnostics-trace.test.ts @@ -9,9 +9,9 @@ import test from "node:test"; import assert from "node:assert/strict"; -const { errorResponseWithComboDiagnostics, sanitizeComboDiagnostics } = await import( - "../../open-sse/utils/error.ts" -); +const { errorResponseWithComboDiagnostics, sanitizeComboDiagnostics } = + await import("../../open-sse/utils/error.ts"); +const { buildRecoveryHint } = await import("../../open-sse/services/combo/pinRecovery.ts"); test("combo diagnostics: headers + body carry the sanitized trace (code override preserved)", async () => { const res = errorResponseWithComboDiagnostics( @@ -89,7 +89,9 @@ test("combo diagnostics: terminalReason with a non-Latin1 char (em dash) must no { poolSize: 4, attempted: 1, - excluded: [{ provider: "deepseek", model: "deepseek-v4-flash-free", reason: "quality — bad" }], + excluded: [ + { provider: "deepseek", model: "deepseek-v4-flash-free", reason: "quality — bad" }, + ], attemptOrder: [{ provider: "deepseek", model: "deepseek-v4-flash-free" }], terminalReason, } @@ -112,8 +114,43 @@ test("combo diagnostics: JSON body keeps the original non-Latin1 text even thoug } ); // Header value must be a valid Latin1 ByteString — em dash (U+2014) replaced. - assert.equal(res.headers.get("x-omniroute-combo-terminal-reason"), terminalReason.replace("—", "?")); + assert.equal( + res.headers.get("x-omniroute-combo-terminal-reason"), + terminalReason.replace("—", "?") + ); const body = await res.json(); // JSON body keeps the original, readable (unsanitized) em dash. assert.equal(body.diagnostics.terminalReason, terminalReason); }); + +test("combo diagnostics preserve every canonical recovery hint up to the existing cap", async () => { + const reasons = [ + "reasoning_budget_exhausted", + "max_attempts_exceeded", + "all_accounts_inactive", + "quota_exhausted", + "all_models_failed", + "no_executable_targets", + "context_requirements_exhausted", + "all_targets_skipped", + "unknown_reason", + ]; + + for (const reason of reasons) { + const recovery = buildRecoveryHint(reason, 30); + const response = errorResponseWithComboDiagnostics(503, "combo failed", { + poolSize: 1, + attempted: 1, + excluded: [], + attemptOrder: [], + terminalReason: reason, + recovery, + }); + const body = (await response.json()) as { + recovery_hint?: { action: string; next_step: string }; + }; + + assert.equal(body.recovery_hint?.action, recovery.action, reason); + assert.equal(body.recovery_hint?.next_step, recovery.next_step.slice(0, 200), reason); + } +}); diff --git a/tests/unit/error-message-sanitization.test.ts b/tests/unit/error-message-sanitization.test.ts index f33a74a591..8813e7ac71 100644 --- a/tests/unit/error-message-sanitization.test.ts +++ b/tests/unit/error-message-sanitization.test.ts @@ -8,8 +8,16 @@ import fs from "node:fs"; import os from "node:os"; import path from "node:path"; -const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-err-sanitize-")); +const TEST_ROOT = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-err-sanitize-")); +const TEST_DATA_DIR = path.join(TEST_ROOT, "data"); +const TEST_PLUGINS_DIR = path.join(TEST_ROOT, "plugins"); +const ORIGINAL_DATA_DIR = process.env.DATA_DIR; +const ORIGINAL_PLUGINS_DIR = process.env.OMNIROUTE_PLUGINS_DIR; +const ORIGINAL_API_KEY_SECRET = process.env.API_KEY_SECRET; +fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +fs.mkdirSync(TEST_PLUGINS_DIR, { recursive: true }); process.env.DATA_DIR = TEST_DATA_DIR; +process.env.OMNIROUTE_PLUGINS_DIR = TEST_PLUGINS_DIR; process.env.API_KEY_SECRET = "test-api-key-secret-32chars-long!!"; const core = await import("../../src/lib/db/core.ts"); @@ -42,7 +50,13 @@ test.beforeEach(async () => { test.after(() => { core.resetDbInstance(); - fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + if (ORIGINAL_DATA_DIR === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = ORIGINAL_DATA_DIR; + if (ORIGINAL_PLUGINS_DIR === undefined) delete process.env.OMNIROUTE_PLUGINS_DIR; + else process.env.OMNIROUTE_PLUGINS_DIR = ORIGINAL_PLUGINS_DIR; + if (ORIGINAL_API_KEY_SECRET === undefined) delete process.env.API_KEY_SECRET; + else process.env.API_KEY_SECRET = ORIGINAL_API_KEY_SECRET; + fs.rmSync(TEST_ROOT, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); }); async function createCombo(name: string, model: string) { @@ -338,7 +352,8 @@ test("buildErrorBody — upstream details with stack key are stripped", async () !("stack" in (body.upstream_details as any)), "stack must be stripped from upstream_details" ); - assert.equal((body.upstream_details as any).code, "internal"); + assert.equal((body.upstream_details as any).code, ""); + assert.doesNotMatch(JSON.stringify(body.upstream_details), /internal/); }); // ── createErrorResult with upstreamDetails ─────────────────────────────────── diff --git a/tests/unit/error-public-boundaries-hardening.test.ts b/tests/unit/error-public-boundaries-hardening.test.ts new file mode 100644 index 0000000000..8e0e202b5b --- /dev/null +++ b/tests/unit/error-public-boundaries-hardening.test.ts @@ -0,0 +1,11 @@ +import test from "node:test"; + +import { runIsolatedBoundaryFixture } from "./helpers/runIsolatedBoundaryFixture.ts"; + +test("public error boundaries pass in an isolated child process", () => { + runIsolatedBoundaryFixture({ + fixtureUrl: new URL("./fixtures/error-public-boundaries-hardening.fixture.ts", import.meta.url), + expectedTests: 23, + label: "public error boundaries", + }); +}); diff --git a/tests/unit/error-sensitive-redaction.test.ts b/tests/unit/error-sensitive-redaction.test.ts index 1c7b5129fb..eb2e7e2392 100644 --- a/tests/unit/error-sensitive-redaction.test.ts +++ b/tests/unit/error-sensitive-redaction.test.ts @@ -1,8 +1,11 @@ import test from "node:test"; import assert from "node:assert/strict"; -import { sanitizeErrorMessage, sanitizeUpstreamDetails } from "../../open-sse/utils/error.ts"; +import { + sanitizeErrorMessage, + sanitizeUpstreamDetails, +} from "../../open-sse/utils/errorSanitization.ts"; -test("sanitizeErrorMessage redacts bearer credentials and image data URLs", () => { +test("sanitizeErrorMessage removes bearer credentials and image data URLs", () => { const raw = "upstream echoed Authorization: Bearer eyJ.secret.token and data:image/png;charset=utf-8;base64,iVBORw0KGgoAAAANSUhEUgAAAAE="; const safe = sanitizeErrorMessage(raw); @@ -10,7 +13,10 @@ test("sanitizeErrorMessage redacts bearer credentials and image data URLs", () = assert.doesNotMatch(safe, /eyJ\.secret\.token/); assert.doesNotMatch(safe, /iVBORw0KGgo/); assert.match(safe, /\[REDACTED\]/); - assert.match(safe, /\[REDACTED_DATA_URL\]/); + // Authorization labels are fail-closed: once a credential label is seen, + // the sanitizer may discard the remaining untrusted tail instead of + // preserving a marker for each later secret. + assert.equal(safe, "upstream echoed Authorization: [REDACTED]"); }); test("sanitizeErrorMessage redacts common JSON credential fields", () => { @@ -25,6 +31,122 @@ test("sanitizeErrorMessage redacts common JSON credential fields", () => { assert.match(safe, /\[REDACTED\]/); }); +test("sanitizeErrorMessage redacts URL credentials while preserving safe URLs", () => { + const safeUrl = "https://example.com/docs/error?lang=en#recovery"; + const projected = sanitizeErrorMessage( + "proxy failed https://svc-user:p4ss-opaque-9382@internal.example/v1 " + + "then https://storage.example/blob?X-Amz-Credential=AKIAOPAQUE%2Fscope&" + + "X-Amz-Signature=signature-secret&X-Amz-Expires=60 " + + "and https://account.blob.core.windows.net/c?sv=2025-01-05&sig=sas-secret&se=soon " + + "then https://vertex.example/predict?key=vertex-key-secret&mode=express " + + "plus https://gateway.example/v1?api_key=query-api-secret&token=query-token-secret " + + `see ${safeUrl}` + ); + + assert.doesNotMatch( + projected, + /svc-user|p4ss-opaque|AKIAOPAQUE|signature-secret|sas-secret|vertex-key-secret|query-api-secret|query-token-secret/i + ); + assert.match(projected, /\[REDACTED\]/); + assert.match(projected, /X-Amz-Expires=60/); + assert.match(projected, /sv=2025-01-05/); + assert.match(projected, /se=soon/); + assert.match(projected, new RegExp(safeUrl.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"))); +}); + +test("sanitizeErrorMessage redacts credentials hidden behind serialized whitespace", () => { + const inputs = [ + String.raw`api_key\t=opaque-tab-secret-9382746`, + String.raw`api_key\u0009=opaque-unicode-tab-9382746`, + String.raw`Bearer\topaque-bearer-secret-9382746`, + String.raw`api_key\\t=opaque-double-tab-secret-9382746`, + ]; + + for (const input of inputs) { + const projected = sanitizeErrorMessage(input); + assert.doesNotMatch(projected, /opaque-(?:tab|unicode-tab|bearer|double-tab)-secret/i); + assert.match(projected, /\[REDACTED\]/); + } +}); + +test("sanitizeErrorMessage redacts CLI credential flag values", () => { + const inputs = [ + "spawn failed: helper --api-key opaque-cli-key-9382746 --mode check", + 'spawn failed: helper --token "opaque cli token 9382746" --mode check', + "spawn failed: helper --password 'opaque-cli-password-9382746' --mode check", + ]; + + for (const input of inputs) { + const projected = sanitizeErrorMessage(input); + assert.doesNotMatch(projected, /opaque(?: cli|-cli)/i); + assert.match(projected, /\[REDACTED\]/); + } +}); + +test("sanitizeErrorMessage covers the canonical credential pattern catalog", () => { + const credentials = [ + `AIza${"A".repeat(35)}`, + `hf_${"A".repeat(34)}`, + `r8_${"A".repeat(37)}`, + `gho_${"A".repeat(36)}`, + `ghu_${"A".repeat(36)}`, + `ghs_${"A".repeat(36)}`, + `ghr_${"A".repeat(36)}`, + `lin_api_${"A".repeat(40)}`, + `secret_${"A".repeat(43)}`, + `npm_${"A".repeat(36)}`, + `PMAK-1234abcd-${"a".repeat(32)}`, + `rk_live_${"A".repeat(24)}`, + `sq0atp-${"A".repeat(22)}`, + `SK${"a".repeat(32)}`, + `SG.${"A".repeat(22)}.${"B".repeat(43)}`, + `key-${"a".repeat(32)}`, + `M${"A".repeat(23)}.${"B".repeat(6)}.${"C".repeat(27)}`, + "postgresql://db-user:db-password@db.internal.example/app", + ]; + + for (const credential of credentials) { + const projected = sanitizeErrorMessage(`upstream echoed ${credential}`); + assert.equal(projected.includes(credential), false, credential.slice(0, 16)); + assert.match(projected, /\[REDACTED(?::[^\]]+)?\]/); + } +}); + +test("sanitizeErrorMessage redacts credentials that cross the public length boundary", () => { + const credential = `hf_${"A".repeat(34)}`; + const projected = sanitizeErrorMessage(`${"x".repeat(4088)}${credential}`); + const escapedPrefixProjected = sanitizeErrorMessage( + `${String.raw`\t`}${"x".repeat(4088)}${credential}` + ); + + for (const output of [projected, escapedPrefixProjected]) { + assert.equal(output.includes("hf_"), false); + assert.equal(output.includes(credential), false); + assert.match(output, /\[REDACTED(?::[^\]]+)?\]$/); + assert.ok(output.length <= 4096); + } +}); + +test("sanitizeErrorMessage redacts closed and unterminated PGP private-key armor", () => { + const closed = sanitizeErrorMessage( + "provider returned -----BEGIN PGP PRIVATE KEY BLOCK-----\n" + + "Version: test\n\npgp-private-material\n" + + "-----END PGP PRIVATE KEY BLOCK----- after" + ); + const unterminated = sanitizeErrorMessage( + "provider returned -----BEGIN PGP PRIVATE KEY BLOCK-----\npgp-unterminated-material" + ); + + // Public exception messages fail closed at the first physical line; the + // post-block suffix is intentionally not recovered from a multiline secret. + assert.equal(closed, "provider returned [REDACTED]"); + assert.equal(unterminated, "provider returned [REDACTED]"); + assert.doesNotMatch( + `${closed} ${unterminated}`, + /pgp-private-material|pgp-unterminated-material/ + ); +}); + test("sanitizeUpstreamDetails drops credential headers and redacts data URLs", () => { const safe = sanitizeUpstreamDetails({ authorization: "Bearer sensitive", diff --git a/tests/unit/fixtures/error-public-boundaries-hardening.fixture.ts b/tests/unit/fixtures/error-public-boundaries-hardening.fixture.ts new file mode 100644 index 0000000000..2b953ebd9e --- /dev/null +++ b/tests/unit/fixtures/error-public-boundaries-hardening.fixture.ts @@ -0,0 +1,608 @@ +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; +import { fileURLToPath } from "node:url"; + +const TEST_ROOT = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-public-errors-")); +const TEST_DATA_DIR = path.join(TEST_ROOT, "data"); +const TEST_PLUGINS_DIR = path.join(TEST_ROOT, "plugins"); +const ORIGINAL_DATA_DIR = process.env.DATA_DIR; +const ORIGINAL_PLUGINS_DIR = process.env.OMNIROUTE_PLUGINS_DIR; +const REPO_ROOT = fileURLToPath(new URL("../../..", import.meta.url)); + +fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +fs.mkdirSync(TEST_PLUGINS_DIR, { recursive: true }); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.OMNIROUTE_PLUGINS_DIR = TEST_PLUGINS_DIR; + +const core = await import("../../../src/lib/db/core.ts"); +const { + buildErrorBody, + buildModelCooldownBody, + createErrorResult, + parseUpstreamError, + projectPublicErrorIdentifier, + providerCircuitOpenResponse, + sanitizeErrorMessage, + sanitizeUpstreamDetails, + unavailableResponse, +} = await import("../../../open-sse/utils/error.ts"); +const { buildPassthroughErrorResponse, shouldPassthroughUpstreamError } = + await import("../../../open-sse/utils/upstreamErrorPassthrough.ts"); + +test.after(() => { + core.resetDbInstance(); + if (ORIGINAL_DATA_DIR === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = ORIGINAL_DATA_DIR; + if (ORIGINAL_PLUGINS_DIR === undefined) delete process.env.OMNIROUTE_PLUGINS_DIR; + else process.env.OMNIROUTE_PLUGINS_DIR = ORIGINAL_PLUGINS_DIR; + fs.rmSync(TEST_ROOT, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("sanitizeErrorMessage removes non-source paths, credentials, and serialized stacks", () => { + const raw = String.raw`Provider failed at /srv/private/provider-key.json access_token=provider-secret\n at validate (C:\Users\admin\private\validator.ts:42:7)`; + const safe = sanitizeErrorMessage(raw); + + assert.match(safe, /Provider failed/i); + assert.doesNotMatch(safe, /srv\/private|provider-secret|C:\\Users|validator\.ts/i); + assert.doesNotMatch(safe, /\\n\s*at validate/i); +}); + +test("sanitizeErrorMessage redacts Windows drive-root-relative filesystem paths", () => { + const plain = sanitizeErrorMessage( + String.raw`Provider failed at \Users\admin\private\secret.txt` + ); + const quoted = sanitizeErrorMessage( + String.raw`Provider failed opening "\Windows\Temp\native.dll"` + ); + const singleSegment = sanitizeErrorMessage(String.raw`Provider failed opening \private.db`); + const prose = sanitizeErrorMessage(String.raw`Provider reported \offline without a path`); + const escapedInitialPaths = [ + String.raw`Provider failed at \bin\private.db`, + String.raw`Provider failed at \folder\private.db`, + String.raw`Provider failed at \new\private.db`, + String.raw`Provider failed at \root\private.db`, + String.raw`Provider failed at \temp\private.db`, + String.raw`Provider failed at C:\temp\private.db`, + ].map((message) => sanitizeErrorMessage(message)); + + assert.equal(plain, "Provider failed at "); + assert.equal(quoted, 'Provider failed opening ""'); + assert.equal(singleSegment, "Provider failed opening "); + assert.equal(prose, String.raw`Provider reported \offline without a path`); + for (const projected of escapedInitialPaths) { + assert.equal(projected, "Provider failed at "); + } +}); + +test("sanitizeErrorMessage redacts extensionless POSIX paths without hiding explicit routes", () => { + const compact = sanitizeErrorMessage("Provider failed at /custom/internal/secret"); + const spaced = sanitizeErrorMessage("Provider failed at /custom/internal secret directory"); + const route = sanitizeErrorMessage("Route /dashboard/providers is unavailable"); + const singleSegment = sanitizeErrorMessage("Provider failed opening /vault"); + const singleSegmentRoute = sanitizeErrorMessage("Route /vault is unavailable"); + const compoundPathAndRoute = sanitizeErrorMessage( + "Failed /vault then GET /home/profile returned 404" + ); + const knownRootRoutes = [ + sanitizeErrorMessage("GET /home returned 404"), + sanitizeErrorMessage("Route /run is unavailable"), + sanitizeErrorMessage("POST /data returned 409"), + sanitizeErrorMessage("Route /var is unavailable"), + ]; + const body = buildErrorBody(500, "Provider failed at /custom/internal/secret"); + + assert.doesNotMatch(compact, /custom\/internal\/secret/); + assert.doesNotMatch(spaced, /custom\/internal|secret directory/); + assert.doesNotMatch(body.error.message, /custom\/internal\/secret/); + assert.match(compact, //); + assert.equal(route, "Route /dashboard/providers is unavailable"); + assert.equal(singleSegment, "Provider failed opening "); + assert.equal(singleSegmentRoute, "Route /vault is unavailable"); + assert.equal(compoundPathAndRoute, "Failed then GET /home/profile returned 404"); + assert.deepEqual(knownRootRoutes, [ + "GET /home returned 404", + "Route /run is unavailable", + "POST /data returned 409", + "Route /var is unavailable", + ]); +}); + +test("sanitizeErrorMessage fails closed when string coercion is hostile", () => { + const hostile = { + toString(): never { + throw new Error("access_token=hostile-secret at /srv/private/hostile.ts:1:2"); + }, + }; + + assert.equal(sanitizeErrorMessage(hostile), ""); +}); + +test("buildErrorBody projects untrusted error classifications onto safe identifiers", () => { + const body = buildErrorBody(502, "upstream failed", undefined, { + type: "server_error\nX-Leak: yes", + code: "sk-live-secret-value", + reason: "access_token=reason-secret", + }); + + assert.equal(body.error.type, "server_error"); + assert.equal(body.error.code, "bad_gateway"); + assert.equal(body.error.reason, undefined); +}); + +test("createErrorResult rejects opaque upstream identifiers that could be echoed credentials", async () => { + const opaqueCredential = "AbC9xY7pQ2mN8vR4kL6z"; + const result = createErrorResult( + 502, + "upstream failed", + null, + opaqueCredential, + opaqueCredential + ); + const body = (await result.response.json()) as { + error: { code: string; type: string }; + }; + + assert.equal(body.error.code, "bad_gateway"); + assert.equal(body.error.type, "server_error"); + assert.doesNotMatch(JSON.stringify(body), new RegExp(opaqueCredential)); +}); + +test("parseUpstreamError never stringifies an untrusted error object into the public message", async () => { + const opaqueIdentifier = "AbC9xY7pQ2mN8vR4kL6z"; + const parsed = await parseUpstreamError( + Response.json( + { + error: { + code: opaqueIdentifier, + type: opaqueIdentifier, + reason: opaqueIdentifier, + }, + }, + { status: 502 } + ), + "openai" + ); + const result = createErrorResult( + parsed.statusCode, + parsed.message, + parsed.retryAfterMs, + parsed.errorCode as string, + parsed.errorType as string, + parsed.responseBody + ); + const bodyText = await result.response.text(); + + assert.equal(parsed.message, "Upstream error: 502"); + assert.doesNotMatch(bodyText, new RegExp(opaqueIdentifier)); +}); + +test("buildErrorBody preserves the configured empty code for unmapped client statuses", () => { + const body = buildErrorBody(424, "Dependency failed"); + + assert.equal(body.error.type, "invalid_request_error"); + assert.equal(body.error.code, ""); +}); + +test("public identifier vocabulary preserves current internal machine-readable contracts", () => { + const identifiers = [ + "context_length_exceeded", + "tool_calling_not_supported", + "vision", + "tools", + "structured_output", + "context_window", + "unsupported_endpoint", + "unverified_codex_client", + "invalid_previous_response_binding", + "incompatible_reasoning_effort", + "STREAM_READINESS_TIMEOUT", + "stream_timeout", + "STREAM_EARLY_EOF", + "stream_early_eof", + "LEASE_NO_ELIGIBLE_CONNECTION", + "LEASE_ELIGIBILITY_UNAVAILABLE", + "LEASE_UNSUPPORTED_ROUTE", + "LEASE_UNSUPPORTED_TRANSPORT", + "DIRECT_RESPONSE_START_TIMEOUT", + "PROXY_FAMILY_UNAVAILABLE", + "RELAY_TIMEOUT", + "TLS_FINGERPRINT_FAILED", + "PROXY_REQUEST_FAILED", + "TLS_SESSION_CAPACITY", + "TLS_CIRCUIT_OPEN", + "PROVIDER_RETIRED", + "upstream_empty_response", + "upstream_response_error", + "upstream_response_failed", + "stream_pipeline_error", + "stream_terminated", + "rate_limited", + "usage_limit_reached", + "timeout", + "semaphore_timeout", + "semaphore_queue_full", + "RATE_LIMIT_EXECUTION_TIMEOUT", + "RATE_LIMIT_QUEUE_FULL", + "RATE_LIMIT_QUEUE_WEDGED", + "RATE_LIMIT_QUEUE_TIMEOUT", + "rate_limit_queue_wedged", + "429", + "empty_response", + "stream_idle_timeout", + "empty_content", + "UNAVAILABLE", + "RESOURCE_EXHAUSTED", + "provider_unavailable", + "unsupported_feature", + "missing_project_id", + "oauth_missing_project_id", + "gcp_project_required", + "QUOTA_ONLY", + "QUOTA_NOT_ALLOCATED", + "cloudflare_challenge", + "cf_mitigated_challenge", + "upstream_protocol_error", + "claude_web_protocol_error", + "service_not_running", + "storage_encryption_stale", + "HTTP_429", + "BLACKBOX_SUBSCRIPTION_REQUIRED", + "BLACKBOX_AUTH_REQUIRED", + "BLACKBOX_RATE_LIMIT", + "abort", + "ABORTED", + "CHIPOTLE_ERROR", + "premium_model_requires_key", + "GROK_ERROR", + "TLS_CLIENT_UNAVAILABLE", + "upstream_access_denied", + "proxy_unavailable", + "EXECUTOR_ERROR", + "executor_contract_violation", + "orphan_tool_result", + "bedrock_stream_error", + "invalid_kiro_tool_call", + "devin_cli_error", + "upstream_websocket_error", + "upstream_websocket_connect_failed", + "codex_app_server_turn_failed", + "missing_credits", + "reached_limit", + "rate_limit_reached", + "rate_limit_longer_reached", + "client_cancelled", + "client_closed_request", + "compaction_control_unavailable", + "compaction_handoff_failed", + "connector_not_found", + "connector_error", + "prompt_attachment_integrity", + "chatgpt_session_expired", + "chatgpt_subscription_unavailable", + "upstream_server_error", + "multipart_protocol_violation", + "browser_stream_inconsistent", + "structured_output_validation_failed", + "chatgpt_submission_ambiguous", + "chatgpt_submitted_turn_failed", + "cli_not_found", + "upstream_auth_error", + "wreq_unavailable", + "api_error", + "connection_error", + "unsupported_runtime", + "VIDEO_ARTIFACT_URL_INVALID", + "VIDEO_ARTIFACT_URL_BLOCKED", + "VIDEO_ARTIFACT_DOWNLOAD_FAILED", + "VIDEO_ARTIFACT_TOO_LARGE", + "VIDEO_ARTIFACT_SIGNATURE_INVALID", + "VIDEO_ARTIFACT_NOT_READY", + "VIDEO_ARTIFACT_UNAVAILABLE", + "VIDEO_ARTIFACT_CONTENT_TYPE_INVALID", + "codex_app_server_unconfigured", + "meta_ai_warmup_failed", + "meta_ai_mode_switch_failed", + "meta_ai_ws_error", + "meta_ai_empty_response", + "PPLX_ERROR", + "cloudflare_or_bot", + "request_failed", + "lmarena_error", + "network_error", + ]; + + for (const identifier of identifiers) { + assert.equal(projectPublicErrorIdentifier(identifier, "bad_request"), identifier, identifier); + } +}); + +test("public numeric identifiers are limited to three-digit HTTP status codes", () => { + assert.equal(projectPublicErrorIdentifier("100", "bad_request"), "100"); + assert.equal(projectPublicErrorIdentifier("599", "bad_request"), "599"); + assert.equal(projectPublicErrorIdentifier("099", "bad_request"), "bad_request"); + assert.equal(projectPublicErrorIdentifier("600", "bad_request"), "bad_request"); + assert.equal(projectPublicErrorIdentifier("5000", "bad_request"), "bad_request"); + assert.equal(projectPublicErrorIdentifier("40002", "bad_request"), "bad_request"); + assert.equal(projectPublicErrorIdentifier("HTTP_600", "bad_request"), "bad_request"); + assert.equal(projectPublicErrorIdentifier("HTTP_40002", "bad_request"), "bad_request"); + assert.equal(projectPublicErrorIdentifier("weird_error", "bad_gateway"), "bad_gateway"); +}); + +test("buildErrorBody callers never overwrite a projected public classification", () => { + const productionFiles: string[] = []; + const collectTypeScriptFiles = (directory: string): void => { + for (const entry of fs.readdirSync(directory, { withFileTypes: true })) { + const entryPath = path.join(directory, entry.name); + if (entry.isDirectory()) { + if (entry.name === "__tests__") continue; + collectTypeScriptFiles(entryPath); + } else if (entry.isFile() && /\.tsx?$/.test(entry.name)) { + productionFiles.push(entryPath); + } + } + }; + + collectTypeScriptFiles(path.join(REPO_ROOT, "open-sse")); + collectTypeScriptFiles(path.join(REPO_ROOT, "src")); + + const mutationPattern = /\b[A-Za-z_$][A-Za-z0-9_$]*\.error\.(?:code|type|reason)\s*=(?!=)/g; + const violations: string[] = []; + for (const filePath of productionFiles) { + const source = fs.readFileSync(filePath, "utf8"); + if (!source.includes("buildErrorBody")) continue; + for (const match of source.matchAll(mutationPattern)) { + const line = source.slice(0, match.index).split("\n").length; + violations.push(`${path.relative(REPO_ROOT, filePath)}:${line}`); + } + } + + assert.deepEqual(violations, []); + + const chatCoreSource = fs.readFileSync( + path.join(REPO_ROOT, "open-sse/handlers/chatCore.ts"), + "utf8" + ); + assert.doesNotMatch( + chatCoreSource, + /JSON\.stringify\(\s*\{\s*error\s*:\s*\{/, + "chatCore must not bypass buildErrorBody with a manually assembled error envelope" + ); +}); + +test("operational log persistence catches use the canonical sanitizer", () => { + const callLogsSource = fs.readFileSync(path.join(REPO_ROOT, "src/lib/usage/callLogs.ts"), "utf8"); + const proxyLoggerSource = fs.readFileSync(path.join(REPO_ROOT, "src/lib/proxyLogger.ts"), "utf8"); + + assert.match(callLogsSource, /sanitizeErrorMessage\(error\)/); + assert.doesNotMatch(callLogsSource, /\(error as Error\)\.message/); + assert.match(proxyLoggerSource, /sanitizeErrorMessage\(err\)/); + assert.doesNotMatch(proxyLoggerSource, /err\?\.message\s*\|\|\s*err/); +}); + +test("stream request finalization never warns with a raw error object", () => { + const source = fs.readFileSync( + path.join(REPO_ROOT, "open-sse/utils/streamFailureFinalization.ts"), + "utf8" + ); + + assert.match(source, /sanitizeErrorMessage\(error\)/); + assert.doesNotMatch(source, /"message" in error[\s\S]{0,160}: error/); +}); + +test("chatCore provider-failure writes use the projected persistent message", () => { + const source = fs.readFileSync(path.join(REPO_ROOT, "open-sse/handlers/chatCore.ts"), "utf8"); + const failureStart = source.indexOf("providerFailure: if (!providerResponse.ok)"); + const failureEnd = source.indexOf("// Non-streaming response", failureStart); + assert.ok(failureStart >= 0 && failureEnd > failureStart, "providerFailure block must exist"); + const failureBlock = source.slice(failureStart, failureEnd); + + assert.doesNotMatch(failureBlock, /lastError:\s*message\b/); + assert.ok( + (failureBlock.match(/lastError:\s*persistentMessage\b/g) || []).length >= 11, + "every providerFailure persistence branch must use persistentMessage" + ); +}); + +test("public cooldown and circuit responses sanitize dynamic context", async () => { + const unavailable = unavailableResponse( + 503, + "Provider failed at /srv/private/state.sqlite access_token=unavailable-secret", + 5, + "retry after reading C:\\Users\\admin\\private\\state.json" + ); + const unavailableBody = (await unavailable.json()) as { error: { message: string } }; + assert.doesNotMatch(unavailableBody.error.message, /srv\/private|unavailable-secret|C:\\Users/i); + + const circuit = providerCircuitOpenResponse( + "provider access_token=circuit-secret /home/service/provider.json", + 5 + ); + const circuitBody = (await circuit.json()) as { + error: { message: string; provider: string }; + }; + assert.equal(circuitBody.error.provider, "unknown"); + assert.doesNotMatch(JSON.stringify(circuitBody), /circuit-secret|\/home\/service/i); + + const cooldown = buildModelCooldownBody({ + model: "model access_token=model-secret /opt/models/private.json", + retryAfterSec: Number.NaN, + retryAfterAt: "not-a-timestamp access_token=timestamp-secret", + }); + assert.equal(cooldown.error.model, undefined); + assert.equal(cooldown.error.retry_after, undefined); + assert.equal(cooldown.error.reset_seconds, 1); + assert.doesNotMatch(JSON.stringify(cooldown), /model-secret|timestamp-secret|\/opt\/models/i); +}); + +test("sanitizeUpstreamDetails drops credential aliases and prototype-control keys", () => { + const input = Object.create(null) as Record; + input.error = { + message: "quota metadata at /srv/provider/private.json", + credential: "credential-secret", + sessionId: "session-secret", + session_count: 2, + }; + input.__proto__ = { leaked: true }; + + const safe = sanitizeUpstreamDetails(input) as Record; + const serialized = JSON.stringify(safe); + + assert.doesNotMatch(serialized, /credential-secret|session-secret|srv\/provider|__proto__/i); + assert.match(serialized, /"session_count":2/); +}); + +test("buildErrorBody fails closed for hostile upstream detail accessors", () => { + const hostile = new Proxy( + {}, + { + ownKeys(): never { + throw new Error("access_token=hostile-detail at /srv/private/detail.ts:1:2"); + }, + } + ); + + let body: ReturnType | undefined; + assert.doesNotThrow(() => { + body = buildErrorBody(502, "upstream failed", hostile); + }); + assert.equal(body?.upstream_details, undefined); + assert.doesNotMatch(JSON.stringify(body), /hostile-detail|srv\/private|detail\.ts/i); +}); + +test("upstream passthrough preserves safe wording but recursively sanitizes the JSON body", async () => { + const opaqueIdentifier = "AbC9xY7pQ2mN8vR4kL6z"; + const upstream = { + type: "error", + error: { + type: "invalid_request_error", + code: opaqueIdentifier, + reason: opaqueIdentifier, + message: "quota metadata from /srv/provider/private.json", + credential: "credential-secret", + session_count: 2, + details: [{ type: "integer", reason: "must be positive" }], + }, + }; + + assert.equal(shouldPassthroughUpstreamError(422, upstream), true); + const response = buildPassthroughErrorResponse(422, upstream); + assert.ok(response); + const serialized = JSON.stringify(await response.json()); + + assert.match(serialized, /invalid_request_error/); + assert.match(serialized, /"session_count":2/); + assert.match(serialized, /"type":"integer","reason":"must be positive"/); + assert.doesNotMatch( + serialized, + new RegExp(`credential-secret|srv/provider|${opaqueIdentifier}`, "i") + ); +}); + +test("upstream classification projection preserves HTTP numbers and rejects opaque aliases", () => { + const opaqueIdentifier = "AbC9xY7pQ2mN8vR4kL6z"; + const projected = sanitizeUpstreamDetails({ + code: 400, + status: "UNAVAILABLE", + oversizedCode: 40002, + error: { + code: 40002, + error_code: opaqueIdentifier, + errorCode: opaqueIdentifier, + error_type: opaqueIdentifier, + errorType: opaqueIdentifier, + sub_type: opaqueIdentifier, + subType: opaqueIdentifier, + status: opaqueIdentifier, + status_code: opaqueIdentifier, + statusCode: opaqueIdentifier, + message: "safe provider wording", + }, + }) as { + code?: unknown; + status?: unknown; + oversizedCode?: unknown; + error?: Record; + }; + + assert.equal(projected.code, 400); + assert.equal(projected.status, "UNAVAILABLE"); + assert.equal(projected.oversizedCode, 40002); + assert.equal(projected.error?.code, undefined); + assert.equal(projected.error?.error_code, ""); + assert.equal(projected.error?.errorCode, ""); + assert.equal(projected.error?.error_type, "upstream_error"); + assert.equal(projected.error?.errorType, "upstream_error"); + assert.equal(projected.error?.sub_type, "upstream_error"); + assert.equal(projected.error?.subType, "upstream_error"); + assert.equal(projected.error?.status, undefined); + assert.equal(projected.error?.status_code, undefined); + assert.equal(projected.error?.statusCode, undefined); + assert.equal(projected.error?.message, "safe provider wording"); + assert.doesNotMatch(JSON.stringify(projected), new RegExp(opaqueIdentifier)); +}); + +test("upstream classification projection preserves only real gRPC numeric codes", () => { + const projected = sanitizeUpstreamDetails({ + error: { code: 7 }, + errors: [{ code: 16 }, { code: 17 }, { code: 40002 }], + status: 7, + warning: { code: "model_capacity", type: "unknown" }, + }) as { + error?: { code?: unknown }; + errors?: Array<{ code?: unknown }>; + status?: unknown; + warning?: { code?: unknown; type?: unknown }; + }; + + assert.equal(projected.error?.code, 7); + assert.equal(projected.errors?.[0]?.code, 16); + assert.equal(projected.errors?.[1]?.code, undefined); + assert.equal(projected.errors?.[2]?.code, undefined); + assert.equal(projected.status, undefined); + assert.equal(projected.warning?.code, ""); + assert.equal(projected.warning?.type, "upstream_error"); +}); + +test("sanitizeUpstreamDetails fails closed for hostile prototype access", () => { + const hostile = new Proxy( + {}, + { + getPrototypeOf(): never { + throw new Error("access_token=prototype-secret at /srv/private/prototype.ts"); + }, + } + ); + + let projected: unknown; + assert.doesNotThrow(() => { + projected = sanitizeUpstreamDetails(hostile); + }); + assert.doesNotMatch(JSON.stringify(projected), /prototype-secret|srv\/private|prototype\.ts/i); +}); + +test("upstream passthrough fails closed for non-serializable bodies", () => { + const cyclic: Record = { error: { message: "safe" } }; + cyclic.self = cyclic; + + assert.equal(shouldPassthroughUpstreamError(400, cyclic), false); + assert.equal(buildPassthroughErrorResponse(400, cyclic), null); +}); + +test("upstream passthrough fails closed when getters change after eligibility", () => { + let reads = 0; + const upstream = Object.create(null) as Record; + Object.defineProperty(upstream, "error", { + enumerable: true, + get(): unknown { + reads += 1; + if (reads === 1) return { message: "safe capability error" }; + throw new Error("access_token=second-read-secret at /srv/private/getter.ts:1:2"); + }, + }); + + assert.doesNotThrow(() => buildPassthroughErrorResponse(400, upstream)); + assert.equal(buildPassthroughErrorResponse(400, upstream), null); +}); diff --git a/tests/unit/fixtures/mcp-public-error-boundaries.fixture.ts b/tests/unit/fixtures/mcp-public-error-boundaries.fixture.ts new file mode 100644 index 0000000000..1eaad34050 --- /dev/null +++ b/tests/unit/fixtures/mcp-public-error-boundaries.fixture.ts @@ -0,0 +1,184 @@ +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; +import { fileURLToPath } from "node:url"; + +const testRoot = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-mcp-error-boundaries-")); +const repoRoot = fileURLToPath(new URL("../../..", import.meta.url)); +const originalDataDir = process.env.DATA_DIR; +const originalPluginsDir = process.env.OMNIROUTE_PLUGINS_DIR; +const originalApiKey = process.env.OMNIROUTE_API_KEY; +const originalApiKeyId = process.env.OMNIROUTE_API_KEY_ID; +const originalInternalToken = process.env.OMNIROUTE_INTERNAL_SERVICE_TOKEN; +const originalInternalTokenFile = process.env.OMNIROUTE_INTERNAL_SERVICE_TOKEN_FILE; +const originalBaseUrl = process.env.OMNIROUTE_BASE_URL; +process.env.DATA_DIR = path.join(testRoot, "data"); +process.env.OMNIROUTE_PLUGINS_DIR = path.join(testRoot, "plugins"); +process.env.OMNIROUTE_API_KEY = "mcp-boundary-test-key"; +process.env.OMNIROUTE_API_KEY_ID = "mcp-boundary-test-key-id"; +process.env.OMNIROUTE_INTERNAL_SERVICE_TOKEN = "mcp-boundary-internal-test-token"; +process.env.OMNIROUTE_BASE_URL = "http://localhost:20128"; +delete process.env.OMNIROUTE_INTERNAL_SERVICE_TOKEN_FILE; +fs.mkdirSync(process.env.DATA_DIR, { recursive: true }); +fs.mkdirSync(process.env.OMNIROUTE_PLUGINS_DIR, { recursive: true }); + +const { createMcpServer } = await import("../../../open-sse/mcp-server/server.ts"); +const { closeAuditDb, queryAuditEntries } = await import("../../../open-sse/mcp-server/audit.ts"); +const { obsidianTools } = await import("../../../open-sse/mcp-server/tools/obsidianTools.ts"); +const { skillTools } = await import("../../../open-sse/mcp-server/tools/skillTools.ts"); +const { skillRegistry } = await import("../../../src/lib/skills/registry.ts"); +const { skillExecutor } = await import("../../../src/lib/skills/executor.ts"); +const core = await import("../../../src/lib/db/core.ts"); + +type McpResult = { + content?: Array<{ type: string; text: string }>; + isError?: boolean; +}; + +type RegisteredTool = { + handler: (args: unknown, extra?: unknown) => Promise; +}; + +function getRegisteredHandler(server: unknown, toolName: string): RegisteredTool["handler"] { + const registry = (server as { _registeredTools?: Record }) + ._registeredTools; + assert.ok(registry, "McpServer should expose _registeredTools"); + const tool = registry[toolName]; + assert.ok(tool, `${toolName} must be registered`); + return tool.handler; +} + +function assertPublicMcpError(result: McpResult): void { + const text = result.content?.[0]?.text ?? ""; + assert.equal(result.isError, true); + assert.match(text, /Error:/); + assert.doesNotMatch(text, /mcp-boundary-secret|srv\/private|mcp-boundary\.ts|\bat execute\b/i); +} + +test.after(() => { + closeAuditDb(); + core.resetDbInstance(); + if (originalDataDir === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = originalDataDir; + if (originalPluginsDir === undefined) delete process.env.OMNIROUTE_PLUGINS_DIR; + else process.env.OMNIROUTE_PLUGINS_DIR = originalPluginsDir; + if (originalApiKey === undefined) delete process.env.OMNIROUTE_API_KEY; + else process.env.OMNIROUTE_API_KEY = originalApiKey; + if (originalApiKeyId === undefined) delete process.env.OMNIROUTE_API_KEY_ID; + else process.env.OMNIROUTE_API_KEY_ID = originalApiKeyId; + if (originalInternalToken === undefined) delete process.env.OMNIROUTE_INTERNAL_SERVICE_TOKEN; + else process.env.OMNIROUTE_INTERNAL_SERVICE_TOKEN = originalInternalToken; + if (originalInternalTokenFile === undefined) { + delete process.env.OMNIROUTE_INTERNAL_SERVICE_TOKEN_FILE; + } else { + process.env.OMNIROUTE_INTERNAL_SERVICE_TOKEN_FILE = originalInternalTokenFile; + } + if (originalBaseUrl === undefined) delete process.env.OMNIROUTE_BASE_URL; + else process.env.OMNIROUTE_BASE_URL = originalBaseUrl; + fs.rmSync(testRoot, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("core MCP handlers sanitize upstream bodies before public and audit boundaries", async () => { + const hostile = "Bearer mcp-fetch-boundary-secret at /srv/private/mcp-fetch-boundary.ts:9:3"; + const originalFetch = globalThis.fetch; + const calledUrls: string[] = []; + globalThis.fetch = async (input) => { + calledUrls.push(String(input)); + return new Response(hostile, { status: 500 }); + }; + + try { + const handler = getRegisteredHandler(createMcpServer(), "omniroute_list_combos"); + const result = await handler({ includeMetrics: false }); + const publicText = result.content?.[0]?.text ?? ""; + assert.equal(result.isError, true); + assert.doesNotMatch( + publicText, + /mcp-fetch-boundary-secret|srv\/private|mcp-fetch-boundary\.ts/i + ); + assert.deepEqual(calledUrls, ["http://localhost:20128/api/combos"]); + + const audit = await queryAuditEntries({ tool: "omniroute_list_combos", success: false }); + assert.ok(audit.entries.length >= 1); + assert.doesNotMatch( + JSON.stringify(audit.entries), + /mcp-fetch-boundary-secret|srv\/private|mcp-fetch-boundary\.ts/i + ); + } finally { + globalThis.fetch = originalFetch; + } +}); + +test("every MCP public catch uses the canonical fail-closed projector", () => { + const source = fs.readFileSync(path.join(repoRoot, "open-sse/mcp-server/server.ts"), "utf8"); + assert.doesNotMatch(source, /err instanceof Error \? err\.message : String\(err\)/); +}); + +test("Obsidian and dynamic-skill MCP wrappers sanitize thrown errors", async () => { + const hostile = new Error( + "MCP failed access_token=mcp-boundary-secret at /srv/private/mcp-boundary.ts\n" + + " at execute (/srv/private/mcp-boundary.ts:9:3)" + ); + const mutableObsidianTool = obsidianTools[0] as unknown as { + name: string; + handler: (args: unknown, extra?: unknown) => Promise; + }; + const originalObsidianHandler = mutableObsidianTool.handler; + try { + mutableObsidianTool.handler = async () => { + throw hostile; + }; + const obsidianHandler = getRegisteredHandler(createMcpServer(), mutableObsidianTool.name); + assertPublicMcpError(await obsidianHandler({}, { authInfo: { scopes: ["read:obsidian"] } })); + } finally { + mutableObsidianTool.handler = originalObsidianHandler; + } + + const mutableRegistry = skillRegistry as unknown as { + list: () => Array<{ name: string; description: string; enabled: boolean }>; + }; + const mutableExecutor = skillExecutor as unknown as { + execute: (...args: unknown[]) => Promise; + }; + const originalList = mutableRegistry.list; + const originalExecute = mutableExecutor.execute; + try { + mutableRegistry.list = () => [ + { name: "mcp_boundary_skill", description: "boundary test", enabled: true }, + ]; + const dynamicHandler = getRegisteredHandler(createMcpServer(), "skill_mcp_boundary_skill"); + mutableExecutor.execute = async () => { + throw hostile; + }; + assertPublicMcpError( + await dynamicHandler({}, { authInfo: { clientId: "test", scopes: ["execute:skills"] } }) + ); + } finally { + mutableRegistry.list = originalList; + mutableExecutor.execute = originalExecute; + } +}); + +test("skill-tool MCP wrapper uses its own fail-closed fallback for hostile thrown values", async () => { + const mutableSkillTool = Object.values(skillTools)[0] as unknown as { + name: string; + handler: (args: unknown, extra?: unknown) => Promise; + }; + const originalHandler = mutableSkillTool.handler; + const revocable = Proxy.revocable({}, {}); + revocable.revoke(); + + try { + mutableSkillTool.handler = async () => { + throw revocable.proxy; + }; + const handler = getRegisteredHandler(createMcpServer(), mutableSkillTool.name); + const result = await handler({}, { authInfo: { scopes: ["read:skills"] } }); + assert.equal(result.isError, true); + assert.equal(result.content?.[0]?.text, "Error: Skill tool execution failed"); + } finally { + mutableSkillTool.handler = originalHandler; + } +}); diff --git a/tests/unit/fixtures/provider-connection-test-error-boundaries.fixture.ts b/tests/unit/fixtures/provider-connection-test-error-boundaries.fixture.ts new file mode 100644 index 0000000000..3ab084243b --- /dev/null +++ b/tests/unit/fixtures/provider-connection-test-error-boundaries.fixture.ts @@ -0,0 +1,262 @@ +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; + +const testRoot = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-provider-errors-")); +const originalDataDir = process.env.DATA_DIR; +const originalPluginsDir = process.env.OMNIROUTE_PLUGINS_DIR; +const originalApiKeySecret = process.env.API_KEY_SECRET; +const originalDisableBackup = process.env.DISABLE_SQLITE_AUTO_BACKUP; +const originalDisableHealthCheck = process.env.OMNIROUTE_DISABLE_CREDENTIAL_HEALTH_CHECK; +const pluginsDir = path.join(testRoot, "plugins"); +const testDataDir = path.join(testRoot, "data"); +fs.mkdirSync(pluginsDir, { recursive: true }); +fs.mkdirSync(testDataDir, { recursive: true }); +process.env.OMNIROUTE_PLUGINS_DIR = pluginsDir; +process.env.DATA_DIR = testDataDir; +assert.notEqual(fs.realpathSync(testDataDir), "/home/diegosouzapw/.omniroute"); +assert.notEqual(fs.realpathSync(pluginsDir), "/home/diegosouzapw/.omniroute/plugins"); + +process.env.API_KEY_SECRET = "provider-error-boundary-test-secret"; +process.env.DISABLE_SQLITE_AUTO_BACKUP = "true"; +process.env.OMNIROUTE_DISABLE_CREDENTIAL_HEALTH_CHECK = "true"; + +// Connection tests suppress their call-log entry under node --test. This file +// exercises the real persistent boundary, so present a normal runtime identity +// before importing the route and its logging modules. +const originalArgv = process.argv; +const originalExecArgv = process.execArgv; +const originalNodeEnv = process.env.NODE_ENV; +const originalVitest = process.env.VITEST; +process.argv = [ + process.execPath, + path.join(process.cwd(), "scripts/ad-hoc/omniroute-boundary-harness.mjs"), +]; +process.execArgv = []; +process.env.NODE_ENV = "development"; +delete process.env.VITEST; + +const hostileValidationMessage = + "Jules failed access_token=jules-boundary-secret at /srv/private/validator.ts\n" + + " at probe (/srv/private/validator.ts:42:7)"; +const julesValidationUrl = "https://jules.googleapis.com/v1alpha/sources"; +const originalFetch = globalThis.fetch; +let validationFetchCalls = 0; +const boundaryFetch = (async (input: string | URL | Request) => { + const url = + typeof input === "string" ? input : input instanceof Request ? input.url : input.toString(); + assert.equal(url, julesValidationUrl, `unexpected outbound request: ${url}`); + validationFetchCalls += 1; + return new Response(hostileValidationMessage, { status: 500 }); +}) as typeof fetch; +globalThis.fetch = boundaryFetch; + +const core = await import("../../../src/lib/db/core.ts"); +const providersDb = await import("../../../src/lib/db/providers.ts"); +const { saveCallLog, waitForCallLogSaves, closeCallLogSaves } = + await import("../../../src/lib/usage/callLogs.ts"); +const { flushProxyLogsSync } = await import("../../../src/lib/proxyLogger.ts"); +const { projectProviderRuntimeForPublicResponse, testSingleConnection } = + await import("../../../src/app/api/providers/[id]/test/route.ts"); +// proxyFetch installs its global dispatcher while the imports above load. Put +// the deterministic stub back at the final fetch seam so this test can never +// reach Jules over the network. +globalThis.fetch = boundaryFetch; + +type ArtifactRow = { artifact_relpath: string | null; error_summary: string | null }; + +function isRecord(value: unknown): value is Record { + return value !== null && typeof value === "object" && !Array.isArray(value); +} + +function readArtifact(relativePath: string | null): Record { + assert.ok(relativePath, "call log must have a persisted detail artifact"); + const absolutePath = path.join(testDataDir, "call_logs", relativePath); + return JSON.parse(fs.readFileSync(absolutePath, "utf8")) as Record; +} + +test.after(async () => { + await closeCallLogSaves(2_000); + flushProxyLogsSync(); + globalThis.fetch = originalFetch; + process.argv = originalArgv; + process.execArgv = originalExecArgv; + if (originalNodeEnv === undefined) delete process.env.NODE_ENV; + else process.env.NODE_ENV = originalNodeEnv; + if (originalVitest === undefined) delete process.env.VITEST; + else process.env.VITEST = originalVitest; + if (originalPluginsDir === undefined) delete process.env.OMNIROUTE_PLUGINS_DIR; + else process.env.OMNIROUTE_PLUGINS_DIR = originalPluginsDir; + core.resetDbInstance(); + if (originalDataDir === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = originalDataDir; + if (originalApiKeySecret === undefined) delete process.env.API_KEY_SECRET; + else process.env.API_KEY_SECRET = originalApiKeySecret; + if (originalDisableBackup === undefined) delete process.env.DISABLE_SQLITE_AUTO_BACKUP; + else process.env.DISABLE_SQLITE_AUTO_BACKUP = originalDisableBackup; + if (originalDisableHealthCheck === undefined) { + delete process.env.OMNIROUTE_DISABLE_CREDENTIAL_HEALTH_CHECK; + } else { + process.env.OMNIROUTE_DISABLE_CREDENTIAL_HEALTH_CHECK = originalDisableHealthCheck; + } + fs.rmSync(testRoot, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("public runtime projection omits host paths and internal error envelopes", () => { + const projected = projectProviderRuntimeForPublicResponse({ + installed: true, + runnable: false, + requiresBinary: true, + reason: "not_executable", + runtimeMode: "local", + version: "v1 from /srv/private/bin/tool", + command: "/srv/private/bin/tool", + commandPath: "/srv/private/bin/tool", + settingsPath: "C:\\Users\\admin\\.config\\tool.json", + error: "access_token=runtime-secret at /srv/private/runtime.json", + diagnosis: { message: "runtime-secret at /srv/private/runtime.ts" }, + }); + const serialized = JSON.stringify(projected); + + assert.equal(projected?.installed, true); + assert.equal(projected?.runnable, false); + assert.equal("commandPath" in (projected || {}), false); + assert.equal("settingsPath" in (projected || {}), false); + assert.equal("error" in (projected || {}), false); + assert.equal("diagnosis" in (projected || {}), false); + assert.doesNotMatch(serialized, /runtime-secret|srv\/private|C:\\\\Users/i); +}); + +test("connection validation projects hostile errors before public and persistent boundaries", async () => { + const connection = await providersDb.createProviderConnection({ + provider: "jules", + authType: "apikey", + name: "Jules Error Boundary", + apiKey: "jules-test-key", + isActive: true, + testStatus: "active", + }); + assert.ok(connection?.id); + + const result = await testSingleConnection(connection.id); + assert.equal(result.valid, false); + assert.ok(validationFetchCalls > 0, "the deterministic Jules stub must handle the probe"); + assert.match(String(result.error), /Jules failed/i); + assert.equal(await waitForCallLogSaves(10_000), true, "call-log write must drain"); + flushProxyLogsSync(); + + const db = core.getDbInstance(); + const providerRow = db + .prepare("SELECT last_error FROM provider_connections WHERE id = ?") + .get(connection.id) as { last_error: string | null }; + const callLogRow = db + .prepare( + `SELECT error_summary, artifact_relpath + FROM call_logs + WHERE connection_id = ? AND model = 'connection-test' + ORDER BY rowid DESC LIMIT 1` + ) + .get(connection.id) as ArtifactRow; + const proxyLogRow = db + .prepare( + `SELECT error + FROM proxy_logs + WHERE connection_id = ? AND provider = 'jules' + AND target_url = 'jules/connection-test' + ORDER BY rowid DESC LIMIT 1` + ) + .get(connection.id) as { error: string | null }; + assert.ok(callLogRow, "connection test must write call_logs"); + assert.ok(proxyLogRow, "connection test must write proxy_logs"); + const artifact = readArtifact(callLogRow.artifact_relpath); + + const boundaries = { + publicResult: result, + providerLastError: providerRow.last_error, + callLogSummary: callLogRow.error_summary, + callLogArtifactError: artifact.error, + proxyLogError: proxyLogRow.error, + }; + const leakPattern = /jules-boundary-secret|srv\/private|validator\.ts|\bat probe\b/i; + const leakingBoundaries = Object.entries(boundaries) + .filter(([, value]) => leakPattern.test(JSON.stringify(value))) + .map(([name]) => name); + assert.deepEqual(leakingBoundaries, []); +}); + +test("failed call logs sanitize response-body copies while successful bodies stay unchanged", async () => { + const hostileBody = { + message: "access_token=call-body-secret at /srv/private/upstream.json", + detail: "Error: api_key=call-detail-secret\n at dispatch (/srv/private/rerank.ts:7:2)", + }; + const successBody = { + message: "Successful output mentions /tmp/public-example.ts and remains unchanged", + usage: { total_tokens: 4 }, + }; + + await saveCallLog({ + id: "error-body-json", + status: 502, + provider: "rerank-test", + model: "rerank-test", + responseBody: hostileBody, + pipelinePayloads: { + providerResponse: { body: hostileBody }, + clientResponse: { body: hostileBody }, + }, + }); + await saveCallLog({ + id: "error-body-text", + status: 503, + provider: "rerank-test", + model: "rerank-test", + responseBody: "Bearer plaintext-body-secret at C:\\Users\\admin\\upstream.txt", + }); + await saveCallLog({ + id: "success-body-control", + status: 200, + provider: "rerank-test", + model: "rerank-test", + responseBody: successBody, + pipelinePayloads: { + providerResponse: { body: successBody }, + clientResponse: { body: successBody }, + }, + }); + await saveCallLog({ + id: "error-body-binary", + status: 500, + provider: "rerank-test", + model: "rerank-test", + responseBody: Buffer.from([1, 2, 3, 4]), + }); + assert.equal(await waitForCallLogSaves(2_000), true, "call-log writes must drain"); + + const db = core.getDbInstance(); + const rows = db + .prepare( + `SELECT id, artifact_relpath FROM call_logs + WHERE id IN ( + 'error-body-json', 'error-body-text', 'success-body-control', 'error-body-binary' + )` + ) + .all() as Array<{ id: string; artifact_relpath: string | null }>; + const artifacts = Object.fromEntries( + rows.map((row) => [row.id, readArtifact(row.artifact_relpath)]) + ) as Record>; + + assert.doesNotMatch( + JSON.stringify({ json: artifacts["error-body-json"], text: artifacts["error-body-text"] }), + /call-body-secret|call-detail-secret|plaintext-body-secret|srv\/private|C:\\\\Users|\bat dispatch\b/i + ); + assert.deepEqual(artifacts["success-body-control"].responseBody, successBody); + assert.equal(artifacts["error-body-binary"].responseBody, "[binary 4 bytes]"); + const pipeline = artifacts["success-body-control"].pipeline; + assert.ok(isRecord(pipeline)); + assert.ok(isRecord(pipeline.providerResponse)); + assert.ok(isRecord(pipeline.clientResponse)); + assert.deepEqual(pipeline.providerResponse.body, successBody); + assert.deepEqual(pipeline.clientResponse.body, successBody); +}); diff --git a/tests/unit/fixtures/provider-last-error-sanitization.fixture.ts b/tests/unit/fixtures/provider-last-error-sanitization.fixture.ts new file mode 100644 index 0000000000..b131bdc67d --- /dev/null +++ b/tests/unit/fixtures/provider-last-error-sanitization.fixture.ts @@ -0,0 +1,109 @@ +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; + +const testRoot = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-provider-last-error-")); +const testDataDir = path.join(testRoot, "data"); +const testPluginsDir = path.join(testRoot, "plugins"); +const originalEnv = { + DATA_DIR: process.env.DATA_DIR, + OMNIROUTE_PLUGINS_DIR: process.env.OMNIROUTE_PLUGINS_DIR, + API_KEY_SECRET: process.env.API_KEY_SECRET, + DISABLE_SQLITE_AUTO_BACKUP: process.env.DISABLE_SQLITE_AUTO_BACKUP, +}; +fs.mkdirSync(testDataDir, { recursive: true }); +fs.mkdirSync(testPluginsDir, { recursive: true }); +process.env.DATA_DIR = testDataDir; +process.env.OMNIROUTE_PLUGINS_DIR = testPluginsDir; +process.env.API_KEY_SECRET = "provider-last-error-test-secret"; +process.env.DISABLE_SQLITE_AUTO_BACKUP = "true"; + +const core = await import("../../../src/lib/db/core.ts"); +const providersDb = await import("../../../src/lib/db/providers.ts"); +const loggerResource = await import("../../../src/shared/utils/loggerResource.ts"); +const { runAsProbe } = await import("../../../src/shared/utils/probeOrigin.ts"); +const { writeTerminalStatus } = await import("../../../src/shared/utils/terminalStatus.ts"); +const { markAccountUnavailable } = await import("../../../src/sse/services/auth.ts"); + +function restoreEnv(name: keyof typeof originalEnv): void { + const original = originalEnv[name]; + if (original === undefined) delete process.env[name]; + else process.env[name] = original; +} + +function readLastError(connectionId: string): string | null { + const row = core + .getDbInstance() + .prepare("SELECT last_error FROM provider_connections WHERE id = ?") + .get(connectionId) as { last_error: string | null } | undefined; + return row?.last_error ?? null; +} + +test.after(async () => { + core.resetDbInstance(); + await loggerResource.closeSharedLoggerResource(); + restoreEnv("DATA_DIR"); + restoreEnv("OMNIROUTE_PLUGINS_DIR"); + restoreEnv("API_KEY_SECRET"); + restoreEnv("DISABLE_SQLITE_AUTO_BACKUP"); + fs.rmSync(testRoot, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("normal and probe failures sanitize provider_connections.lastError at the write seam", async () => { + const hostile = + "provider failed access_token=provider-last-error-secret at /srv/private/provider.ts\n" + + " at dispatch (/srv/private/provider.ts:12:4)"; + const normal = await providersDb.createProviderConnection({ + provider: "openai", + authType: "apikey", + name: "normal last-error boundary", + apiKey: "normal-last-error-test-key", // pragma: allowlist secret + isActive: true, + testStatus: "active", + }); + const probe = await providersDb.createProviderConnection({ + provider: "openai", + authType: "apikey", + name: "probe last-error boundary", + apiKey: "probe-last-error-test-key", // pragma: allowlist secret + isActive: true, + testStatus: "active", + }); + const terminal = await providersDb.createProviderConnection({ + provider: "openai", + authType: "apikey", + name: "terminal last-error boundary", + apiKey: "terminal-last-error-test-key", // pragma: allowlist secret + isActive: true, + testStatus: "active", + }); + + await markAccountUnavailable(normal.id, 500, hostile, "openai"); + await runAsProbe(() => markAccountUnavailable(probe.id, 500, hostile, "openai")); + await writeTerminalStatus( + terminal.id, + { + testStatus: "banned", + isActive: false, + lastError: hostile, + lastErrorType: "forbidden", + errorCode: "403", + }, + "production" + ); + + const persisted = { + normal: readLastError(normal.id), + probe: readLastError(probe.id), + terminal: readLastError(terminal.id), + }; + assert.match(String(persisted.normal), /provider failed/i); + assert.match(String(persisted.probe), /provider failed/i); + assert.match(String(persisted.terminal), /provider failed/i); + assert.doesNotMatch( + JSON.stringify(persisted), + /provider-last-error-secret|srv\/private|provider\.ts|\bat dispatch\b/i + ); +}); diff --git a/tests/unit/fixtures/request-log-management-boundary.fixture.ts b/tests/unit/fixtures/request-log-management-boundary.fixture.ts new file mode 100644 index 0000000000..e3b75b22a9 --- /dev/null +++ b/tests/unit/fixtures/request-log-management-boundary.fixture.ts @@ -0,0 +1,108 @@ +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; + +const TEST_ROOT = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-log-management-boundary-")); +const TEST_DATA_DIR = path.join(TEST_ROOT, "data"); +const TEST_PLUGINS_DIR = path.join(TEST_ROOT, "plugins"); +const ORIGINAL_DATA_DIR = process.env.DATA_DIR; +const ORIGINAL_PLUGINS_DIR = process.env.OMNIROUTE_PLUGINS_DIR; +const ORIGINAL_DISABLE_BACKUP = process.env.DISABLE_SQLITE_AUTO_BACKUP; + +fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +fs.mkdirSync(TEST_PLUGINS_DIR, { recursive: true }); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.OMNIROUTE_PLUGINS_DIR = TEST_PLUGINS_DIR; +process.env.DISABLE_SQLITE_AUTO_BACKUP = "1"; + +const core = await import("../../../src/lib/db/core.ts"); +const usageHistory = await import("../../../src/lib/usage/usageHistory.ts"); +const logsRoute = await import("../../../src/app/api/logs/[id]/route.ts"); +const usageHistoryRoute = await import("../../../src/app/api/usage/history/route.ts"); + +test.afterEach(() => { + usageHistory.clearPendingRequests(); +}); + +test.after(() => { + usageHistory.clearPendingRequests(); + core.resetDbInstance(); + if (ORIGINAL_DATA_DIR === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = ORIGINAL_DATA_DIR; + if (ORIGINAL_PLUGINS_DIR === undefined) delete process.env.OMNIROUTE_PLUGINS_DIR; + else process.env.OMNIROUTE_PLUGINS_DIR = ORIGINAL_PLUGINS_DIR; + if (ORIGINAL_DISABLE_BACKUP === undefined) delete process.env.DISABLE_SQLITE_AUTO_BACKUP; + else process.env.DISABLE_SQLITE_AUTO_BACKUP = ORIGINAL_DISABLE_BACKUP; + fs.rmSync(TEST_ROOT, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +const HOSTILE = + "Bearer management-cache-secret at /srv/private/completed-request.ts:12:3\n" + + " at finalize (/srv/private/finalize.ts:4:2)"; + +async function readManagementDetail(id: string): Promise> { + const response = await logsRoute.GET(undefined as unknown as Request, { params: { id } }); + assert.equal(response.status, 200); + return (await response.json()) as Record; +} + +function serializedDetail(detail: Record): string { + return JSON.stringify(detail); +} + +test("management detail sanitizes in-flight failure chunks at the endpoint boundary", async () => { + const requestId = usageHistory.trackPendingRequest("model", "provider", "conn-inflight", true); + assert.ok(requestId); + usageHistory.updatePendingRequestStreamChunks("model", "provider", "conn-inflight", { + provider: [`event: error\ndata: ${HOSTILE}\n\n`], + openai: [], + client: [], + }); + + const detail = await readManagementDetail(requestId); + assert.doesNotMatch( + serializedDetail(detail), + /management-cache-secret|srv\/private|completed-request\.ts|\bat finalize\b/i + ); +}); + +test("management detail sanitizes completed error metadata and cached chunks", async () => { + const requestId = usageHistory.trackPendingRequest("model", "provider", "conn-completed", true); + assert.ok(requestId); + usageHistory.updatePendingRequestStreamChunks("model", "provider", "conn-completed", { + provider: [`data: ${JSON.stringify({ type: "error", message: HOSTILE })}\n\n`], + openai: [], + client: [], + }); + assert.equal( + usageHistory.finalizePendingRequestById(requestId, { status: 502, error: HOSTILE }), + true + ); + + const detail = await readManagementDetail(requestId); + assert.doesNotMatch( + serializedDetail(detail), + /management-cache-secret|srv\/private|completed-request\.ts|\bat finalize\b/i + ); +}); + +test("usage history endpoint exposes pending counters without raw request details", async () => { + const requestId = usageHistory.trackPendingRequest("model", "provider", "conn-usage", true); + assert.ok(requestId); + usageHistory.updatePendingRequestStreamChunks("model", "provider", "conn-usage", { + provider: [`event: error\ndata: ${HOSTILE}\n\n`], + openai: [], + client: [], + }); + + const response = await usageHistoryRoute.GET(undefined as unknown as Request); + assert.equal(response.status, 200); + const body = (await response.json()) as { + pending?: { byModel?: Record; details?: unknown }; + }; + assert.equal(body.pending?.byModel?.["model (provider)"], 1); + assert.equal("details" in (body.pending ?? {}), false); + assert.doesNotMatch(JSON.stringify(body), /management-cache-secret|srv\/private/i); +}); diff --git a/tests/unit/fixtures/stream-failure-persistent-classification.fixture.ts b/tests/unit/fixtures/stream-failure-persistent-classification.fixture.ts new file mode 100644 index 0000000000..c589a4bd02 --- /dev/null +++ b/tests/unit/fixtures/stream-failure-persistent-classification.fixture.ts @@ -0,0 +1,91 @@ +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; + +const TEST_ROOT = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-stream-failure-code-")); +const TEST_DATA_DIR = path.join(TEST_ROOT, "data"); +const TEST_PLUGINS_DIR = path.join(TEST_ROOT, "plugins"); +const ORIGINAL_DATA_DIR = process.env.DATA_DIR; +const ORIGINAL_PLUGINS_DIR = process.env.OMNIROUTE_PLUGINS_DIR; + +fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +fs.mkdirSync(TEST_PLUGINS_DIR, { recursive: true }); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.OMNIROUTE_PLUGINS_DIR = TEST_PLUGINS_DIR; + +const core = await import("../../../src/lib/db/core.ts"); +const failureUsage = await import("../../../open-sse/handlers/chatCore/failureUsage.ts"); +const usageHistory = await import("../../../src/lib/usage/usageHistory.ts"); +const { createStreamFailureFinalizers } = + await import("../../../open-sse/utils/streamFailureFinalization.ts"); + +test.after(() => { + core.resetDbInstance(); + if (ORIGINAL_DATA_DIR === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = ORIGINAL_DATA_DIR; + if (ORIGINAL_PLUGINS_DIR === undefined) delete process.env.OMNIROUTE_PLUGINS_DIR; + else process.env.OMNIROUTE_PLUGINS_DIR = ORIGINAL_PLUGINS_DIR; + fs.rmSync(TEST_ROOT, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("stream failure persists only the projected public classification", () => { + const opaqueCode = "opaque-stream-code-secret-9382746"; + let completionCode: string | null | undefined; + let persistedCode: string | undefined; + let classifierCode: string | undefined; + const { handleStreamFailure } = createStreamFailureFinalizers({ + isFailureCompletionRecorded: () => false, + onStreamComplete: (payload) => { + completionCode = payload.errorCode; + }, + persistFailureUsage: (_status, errorCode) => { + persistedCode = errorCode; + }, + onStreamFailure: (failure) => { + classifierCode = failure.code; + }, + }); + + assert.equal( + handleStreamFailure({ status: 502, message: "upstream failed", code: opaqueCode }), + true + ); + assert.equal(completionCode, "bad_gateway"); + assert.equal(persistedCode, "bad_gateway"); + assert.equal(classifierCode, opaqueCode); +}); + +test("pre-response failures persist only the projected public classification", async () => { + const opaqueCode = "opaque-pre-response-code-secret-6382951"; + const projectedCode = failureUsage.projectFailureUsageErrorCode({ + statusCode: 502, + message: "upstream request failed", + errorCode: opaqueCode, + errorType: "opaque-pre-response-type-secret-9472013", + }); + + assert.equal(projectedCode, "bad_gateway"); + + const provider = "persistent-error-code-boundary"; + await usageHistory.saveRequestUsage( + failureUsage.buildFailureUsageRecord({ + provider, + model: "model", + connectionId: null, + apiKeyInfo: null, + effectiveServiceTier: "standard", + isCombo: false, + comboStrategy: null, + statusCode: 502, + errorCode: projectedCode, + latencyMs: 1, + }) + ); + + const rows = await usageHistory.getUsageHistory({ provider }); + assert.equal(rows.length, 1); + assert.equal(rows[0]?.errorCode, "bad_gateway"); + assert.doesNotMatch(JSON.stringify(rows), /opaque-pre-response|6382951|9472013/); +}); diff --git a/tests/unit/gemini-responses-error-redaction.test.ts b/tests/unit/gemini-responses-error-redaction.test.ts new file mode 100644 index 0000000000..f8814b3e8c --- /dev/null +++ b/tests/unit/gemini-responses-error-redaction.test.ts @@ -0,0 +1,39 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { translateResponse, initState } from "../../open-sse/translator/index.ts"; +import { FORMATS } from "../../open-sse/translator/formats.ts"; + +test("Gemini keeps raw failure wording internal but projects response.completed.error", () => { + const state = initState(FORMATS.OPENAI_RESPONSES); + const hostileMessage = + "Gemini failed at /srv/omniroute/private-runtime.ts:71:3 token=sk-gemini-secret-123456"; + + const translated = translateResponse( + FORMATS.GEMINI, + FORMATS.OPENAI_RESPONSES, + { + response: { + error: { + code: 503, + status: "UNAVAILABLE", + message: hostileMessage, + api_key: "sk-gemini-secret-abcdef", + }, + }, + }, + state + ); + assert.equal(translated?.length ?? 0, 0); + assert.match(state.upstreamError?.message ?? "", /private-runtime\.ts/); + + const flushed = translateResponse(FORMATS.GEMINI, FORMATS.OPENAI_RESPONSES, null, state); + const completed = flushed.find((event) => event?.data?.type === "response.completed"); + assert.ok(completed); + assert.equal(completed.data.response.status, "failed"); + + const publicError = JSON.stringify(completed.data.response.error); + assert.doesNotMatch(publicError, /private-runtime\.ts/); + assert.doesNotMatch(publicError, /sk-gemini-secret/); + assert.doesNotMatch(publicError, /api_key/); + assert.equal(completed.data.response.error.code, "503"); +}); diff --git a/tests/unit/helpers/runIsolatedBoundaryFixture.ts b/tests/unit/helpers/runIsolatedBoundaryFixture.ts new file mode 100644 index 0000000000..07cc672e28 --- /dev/null +++ b/tests/unit/helpers/runIsolatedBoundaryFixture.ts @@ -0,0 +1,73 @@ +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import { mkdirSync, mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; + +const REPO_ROOT = fileURLToPath(new URL("../../..", import.meta.url)); +const CHILD_PATH = "/usr/local/bin:/usr/bin:/bin"; +const CHILD_MAX_BUFFER_BYTES = 10 * 1024 * 1024; + +type IsolatedBoundaryFixtureOptions = { + fixtureUrl: URL; + expectedTests: number; + label: string; + timeoutMs?: number; +}; + +export function runIsolatedBoundaryFixture({ + fixtureUrl, + expectedTests, + label, + timeoutMs = 180_000, +}: IsolatedBoundaryFixtureOptions): void { + const root = mkdtempSync(join(tmpdir(), "omniroute-public-error-child-")); + const dataDir = join(root, "data"); + const pluginsDir = join(root, "plugins"); + mkdirSync(dataDir, { recursive: true }); + mkdirSync(pluginsDir, { recursive: true }); + + try { + const result = spawnSync( + process.execPath, + ["--import", "tsx/esm", "--test", "--test-reporter=tap", fileURLToPath(fixtureUrl)], + { + cwd: REPO_ROOT, + encoding: "utf8", + env: { + APP_LOG_TO_FILE: "false", + API_KEY_SECRET: "public-error-boundary-fixture-secret", + DATA_DIR: dataDir, + DISABLE_SQLITE_AUTO_BACKUP: "true", + LANG: "C.UTF-8", + LC_ALL: "C.UTF-8", + NODE_ENV: "test", + OMNIROUTE_DISABLE_CREDENTIAL_HEALTH_CHECK: "true", + OMNIROUTE_PLUGINS_DIR: pluginsDir, + PATH: CHILD_PATH, + TZ: "UTC", + }, + maxBuffer: CHILD_MAX_BUFFER_BYTES, + timeout: timeoutMs, + } + ); + const diagnostics = [ + `${label} child status=${String(result.status)} signal=${String(result.signal)}`, + result.error ? `error=${String(result.error)}` : "", + `stdout:\n${result.stdout}`, + `stderr:\n${result.stderr}`, + ] + .filter(Boolean) + .join("\n"); + + assert.equal(result.error, undefined, diagnostics); + assert.equal(result.signal, null, diagnostics); + assert.equal(result.status, 0, diagnostics); + assert.match(result.stdout, new RegExp(`# tests ${expectedTests}(?:\\r?\\n|$)`), diagnostics); + assert.match(result.stdout, new RegExp(`# pass ${expectedTests}(?:\\r?\\n|$)`), diagnostics); + assert.match(result.stdout, /# fail 0(?:\r?\n|$)/, diagnostics); + } finally { + rmSync(root, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + } +} diff --git a/tests/unit/mcp-public-error-boundaries.test.ts b/tests/unit/mcp-public-error-boundaries.test.ts new file mode 100644 index 0000000000..92095c8f85 --- /dev/null +++ b/tests/unit/mcp-public-error-boundaries.test.ts @@ -0,0 +1,11 @@ +import test from "node:test"; + +import { runIsolatedBoundaryFixture } from "./helpers/runIsolatedBoundaryFixture.ts"; + +test("MCP public error boundaries pass in an isolated child process", () => { + runIsolatedBoundaryFixture({ + fixtureUrl: new URL("./fixtures/mcp-public-error-boundaries.fixture.ts", import.meta.url), + expectedTests: 4, + label: "MCP public error boundaries", + }); +}); diff --git a/tests/unit/moderations-handler.test.ts b/tests/unit/moderations-handler.test.ts index 68ec32847a..640d0a68bc 100644 --- a/tests/unit/moderations-handler.test.ts +++ b/tests/unit/moderations-handler.test.ts @@ -2,9 +2,8 @@ import test from "node:test"; import assert from "node:assert/strict"; const { handleModeration } = await import("../../open-sse/handlers/moderations.ts"); -const { MODERATION_PROVIDERS, getModerationProvider, parseModerationModel } = await import( - "../../open-sse/config/moderationRegistry.ts" -); +const { MODERATION_PROVIDERS, getModerationProvider, parseModerationModel } = + await import("../../open-sse/config/moderationRegistry.ts"); const originalFetch = globalThis.fetch; @@ -136,6 +135,76 @@ test("handleModeration returns upstream error payloads with CORS headers", async assert.match(response.headers.get("access-control-allow-methods") || "", /OPTIONS/); }); +test("handleModeration sanitizes structured upstream error bodies", async () => { + globalThis.fetch = async () => + Response.json( + { + error: { + message: "quota metadata at /srv/provider/private.json", + api_key: "credential-value-12345", + }, + }, + { status: 429 } + ); + + const response = await handleModeration({ + body: { model: "openai/text-moderation-latest", input: "check this" }, + credentials: { apiKey: "sk-test" }, + }); + const payload = (await response.json()) as { + error: { message: string; api_key?: string }; + }; + + assert.equal(response.status, 429); + assert.equal(payload.error.api_key, undefined); + assert.doesNotMatch(payload.error.message, /srv\/provider/i); + assert.doesNotMatch(JSON.stringify(payload), /credential-value-12345/i); +}); + +test("handleModeration canonicalizes blank, plaintext, and mislabeled upstream failures", async () => { + const scenarios = [ + { name: "blank", body: " ", contentType: "application/json" }, + { + name: "plaintext", + body: "access_token=moderation-plain-secret at /srv/private/moderation.txt", + contentType: "text/plain", + }, + { + name: "mislabeled", + body: "api_key=moderation-html-secret at /srv/private/error.html", + contentType: "application/json", + }, + ]; + + for (const scenario of scenarios) { + globalThis.fetch = async () => + new Response(scenario.body, { + status: 502, + headers: { "content-type": scenario.contentType }, + }); + const response = await handleModeration({ + body: { model: "openai/text-moderation-latest", input: "check this" }, + credentials: { apiKey: "sk-test" }, + }); + const text = await response.text(); + const payload = JSON.parse(text) as { error: { message: string } }; + + assert.equal(response.status, 502, scenario.name); + assert.match(response.headers.get("content-type") || "", /application\/json/i, scenario.name); + assert.match( + response.headers.get("access-control-allow-methods") || "", + /OPTIONS/, + scenario.name + ); + assert.equal(typeof payload.error.message, "string", scenario.name); + assert.doesNotMatch( + text, + /moderation-plain-secret|moderation-html-secret|srv\/private|/i, + scenario.name + ); + } +}); + test("handleModeration returns a 500 when the upstream request throws", async () => { globalThis.fetch = async () => { throw new Error("socket closed"); diff --git a/tests/unit/ocr-handler-dispatch.test.ts b/tests/unit/ocr-handler-dispatch.test.ts index 2474f6b0b2..df8495b095 100644 --- a/tests/unit/ocr-handler-dispatch.test.ts +++ b/tests/unit/ocr-handler-dispatch.test.ts @@ -38,6 +38,86 @@ test("mistral path posts once and returns the upstream body", async () => { assert.equal(data.pages[0].markdown, "ok"); }); +test("OCR sanitizes structured upstream error bodies", async () => { + const opaqueIdentifier = "AbC9xY7pQ2mN8vR4kL6z"; + const res = await handleOcr({ + body: { + model: "mistral/mistral-ocr-latest", + document: { type: "image_url", image_url: "https://x/y.png" }, + }, + credentials: { apiKey: "sk" }, + fetchImpl: async () => + Response.json( + { + error: { + message: "quota metadata at /srv/provider/private.json", + type: opaqueIdentifier, + code: opaqueIdentifier, + reason: opaqueIdentifier, + api_key: "credential-value-12345", + }, + }, + { status: 429 } + ), + sleepImpl: noSleep, + }); + const payload = (await res.json()) as { + error: { message: string; type?: string; code?: string; reason?: string; api_key?: string }; + }; + + assert.equal(res.status, 429); + assert.equal(payload.error.api_key, undefined); + assert.doesNotMatch(payload.error.message, /srv\/provider/i); + assert.doesNotMatch( + JSON.stringify(payload), + new RegExp(`credential-value-12345|${opaqueIdentifier}`, "i") + ); +}); + +test("OCR canonicalizes blank, plaintext, and mislabeled upstream failures", async () => { + const scenarios = [ + { name: "blank", body: " ", contentType: "application/json" }, + { + name: "plaintext", + body: "access_token=ocr-plain-secret at /srv/private/ocr.txt", + contentType: "text/plain", + }, + { + name: "mislabeled", + body: "api_key=ocr-html-secret at /srv/private/ocr.html", + contentType: "application/json", + }, + ]; + + for (const scenario of scenarios) { + const res = await handleOcr({ + body: { + model: "mistral/mistral-ocr-latest", + document: { type: "image_url", image_url: "https://x/y.png" }, + }, + credentials: { apiKey: "sk" }, + fetchImpl: async () => + new Response(scenario.body, { + status: 502, + headers: { "content-type": scenario.contentType }, + }), + sleepImpl: noSleep, + }); + const text = await res.text(); + const payload = JSON.parse(text) as { error: { message: string } }; + + assert.equal(res.status, 502, scenario.name); + assert.match(res.headers.get("content-type") || "", /application\/json/i, scenario.name); + assert.match(res.headers.get("access-control-allow-methods") || "", /OPTIONS/, scenario.name); + assert.equal(typeof payload.error.message, "string", scenario.name); + assert.doesNotMatch( + text, + /ocr-plain-secret|ocr-html-secret|srv\/private|/i, + scenario.name + ); + } +}); + test("azure DI path polls Operation-Location until succeeded", async () => { const { impl, calls } = fetchStub([ { status: 202, headers: { "Operation-Location": "https://poll/op/1" } }, diff --git a/tests/unit/provider-connection-test-error-boundaries.test.ts b/tests/unit/provider-connection-test-error-boundaries.test.ts new file mode 100644 index 0000000000..4db74062b0 --- /dev/null +++ b/tests/unit/provider-connection-test-error-boundaries.test.ts @@ -0,0 +1,14 @@ +import test from "node:test"; + +import { runIsolatedBoundaryFixture } from "./helpers/runIsolatedBoundaryFixture.ts"; + +test("provider connection error boundaries pass in an isolated child process", () => { + runIsolatedBoundaryFixture({ + fixtureUrl: new URL( + "./fixtures/provider-connection-test-error-boundaries.fixture.ts", + import.meta.url + ), + expectedTests: 3, + label: "provider connection error boundaries", + }); +}); diff --git a/tests/unit/provider-last-error-sanitization.test.ts b/tests/unit/provider-last-error-sanitization.test.ts new file mode 100644 index 0000000000..e7dee5071a --- /dev/null +++ b/tests/unit/provider-last-error-sanitization.test.ts @@ -0,0 +1,11 @@ +import test from "node:test"; + +import { runIsolatedBoundaryFixture } from "./helpers/runIsolatedBoundaryFixture.ts"; + +test("provider last-error persistence passes in an isolated child process", () => { + runIsolatedBoundaryFixture({ + fixtureUrl: new URL("./fixtures/provider-last-error-sanitization.fixture.ts", import.meta.url), + expectedTests: 1, + label: "provider last-error persistence", + }); +}); diff --git a/tests/unit/provider-validation-error-sanitization.test.ts b/tests/unit/provider-validation-error-sanitization.test.ts new file mode 100644 index 0000000000..4a725216b9 --- /dev/null +++ b/tests/unit/provider-validation-error-sanitization.test.ts @@ -0,0 +1,101 @@ +import assert from "node:assert/strict"; +import fs from "node:fs"; +import test from "node:test"; +import { + projectProviderValidationResultForPublicResponse, + toValidationErrorResult, +} from "../../src/lib/providers/validation/transport.ts"; + +test("provider validation sanitizes thrown error details", () => { + const result = toValidationErrorResult( + new Error( + "Provider probe failed at /srv/private/provider-key.json " + + "access_token=provider-secret\n at validate (/srv/private/validator.ts:42:7)" + ) + ); + + assert.equal(result.valid, false); + assert.match(result.error, /Provider probe failed/i); + assert.doesNotMatch(result.error, /srv\/private|provider-secret|validator\.ts|\bat validate\b/i); + assert.equal(result.unsupported, false); +}); + +test("provider validation fails closed for hostile thrown values", () => { + const hostile = new Proxy( + {}, + { + getPrototypeOf(): never { + throw new Error("access_token=prototype-secret at /srv/private/prototype.ts:1:2"); + }, + get(_target, property): unknown { + if (property === "code" || property === "isRetryable") { + throw new Error("access_token=metadata-secret at /srv/private/metadata.ts:1:2"); + } + if (property === "toString") { + return () => { + throw new Error("access_token=coercion-secret at /srv/private/coercion.ts:1:2"); + }; + } + return undefined; + }, + } + ); + + assert.deepEqual(toValidationErrorResult(hostile), { + valid: false, + error: "Validation failed", + unsupported: false, + }); +}); + +test("provider validation route sanitizes unexpected failures before persistent logging", () => { + const routeSource = fs.readFileSync( + new URL("../../src/app/api/providers/validate/route.ts", import.meta.url), + "utf8" + ); + + assert.match( + routeSource, + /console\.log\(\s*"Error validating API key:",\s*sanitizeErrorMessage\(error\) \|\| "Validation failed"\s*\)/ + ); + assert.doesNotMatch(routeSource, /console\.log\(\s*"Error validating API key:",\s*error\s*\)/); +}); + +test("provider validation final response projection sanitizes validator errors and warnings", () => { + const projected = projectProviderValidationResultForPublicResponse({ + valid: false, + error: + "Provider echoed access_token=response-secret at /srv/private/provider.json\n" + + " at validate (/srv/private/validator.ts:42:7)", + warning: "Retry after reading C:\\Users\\admin\\private\\warning.json", + method: "probe", + }); + const serialized = JSON.stringify(projected); + + assert.equal(projected.valid, false); + assert.equal(projected.method, "probe"); + assert.doesNotMatch( + serialized, + /response-secret|srv\/private|validator\.ts|C:\\Users|warning\.json/i + ); +}); + +test("provider validation projection preserves intentionally empty fields without synthetic text", () => { + const projected = projectProviderValidationResultForPublicResponse({ + valid: false, + error: "", + warning: "", + }); + + assert.equal(projected.error, ""); + assert.equal(projected.warning, ""); +}); + +test("provider validation route applies the final response projection", () => { + const routeSource = fs.readFileSync( + new URL("../../src/app/api/providers/validate/route.ts", import.meta.url), + "utf8" + ); + + assert.match(routeSource, /projectProviderValidationResultForPublicResponse\(/); +}); diff --git a/tests/unit/request-log-management-boundary.test.ts b/tests/unit/request-log-management-boundary.test.ts new file mode 100644 index 0000000000..2b619774db --- /dev/null +++ b/tests/unit/request-log-management-boundary.test.ts @@ -0,0 +1,11 @@ +import test from "node:test"; + +import { runIsolatedBoundaryFixture } from "./helpers/runIsolatedBoundaryFixture.ts"; + +test("request-log management boundaries pass in an isolated child process", () => { + runIsolatedBoundaryFixture({ + fixtureUrl: new URL("./fixtures/request-log-management-boundary.fixture.ts", import.meta.url), + expectedTests: 3, + label: "request-log management boundaries", + }); +}); diff --git a/tests/unit/request-log-payloads.test.ts b/tests/unit/request-log-payloads.test.ts index 46aa84792d..098eaf12e8 100644 --- a/tests/unit/request-log-payloads.test.ts +++ b/tests/unit/request-log-payloads.test.ts @@ -4,6 +4,7 @@ import assert from "node:assert/strict"; const { normalizePayloadForLog, + protectErrorPayloadForLog, protectPayloadForLog, serializePayloadForStorage, parseStoredPayload, @@ -65,6 +66,426 @@ test("redacts web-impersonation body credentials but preserves non-secret 'capab }); }); +test("redacts challenge and handoff credentials from persistent request logs", () => { + const protectedPayload = protectPipelinePayloads({ + providerRequest: { + model: "browser-session-model", + recaptchaV3Token: "recaptcha-secret", + nested: { + recaptchaToken: "recaptcha-alias-secret", + turnstileToken: "turnstile-secret", + proofToken: "proof-secret", + resumeToken: "resume-secret", + prepare_token: "prepare-secret", + }, + }, + }); + + assert.deepEqual(protectedPayload?.providerRequest, { + model: "browser-session-model", + recaptchaV3Token: "[REDACTED]", + nested: { + recaptchaToken: "[REDACTED]", + turnstileToken: "[REDACTED]", + proofToken: "[REDACTED]", + resumeToken: "[REDACTED]", + prepare_token: "[REDACTED]", + }, + }); +}); + +test("sanitizes pipeline error messages before persistent request logs", () => { + const protectedPayload = protectPipelinePayloads({ + error: { + timestamp: "2026-09-02T00:00:00.000Z", + error: + "Provider failed access_token=pipeline-secret at /srv/private/provider.json\n" + + " at dispatch (/srv/private/dispatcher.ts:42:7)", + requestBody: { + max_tokens: 512, + temperature: 0.2, + prompt: "Inspect /tmp/example.ts without changing it", + }, + }, + }); + const serialized = JSON.stringify(protectedPayload); + + assert.doesNotMatch(serialized, /pipeline-secret|srv\/private|dispatcher\.ts|\bat dispatch\b/i); + assert.deepEqual(protectedPayload?.error?.requestBody, { + max_tokens: 512, + temperature: 0.2, + prompt: "Inspect /tmp/example.ts without changing it", + }); +}); + +test("sanitizes only nested error and warning subtrees in persisted response bodies", () => { + const payload = { + content: "Normal output mentions /tmp/public-example.ts and must remain intact", + usage: { completion_tokens: 7 }, + error: { + message: "access_token=response-secret at /srv/private/provider.json", + stack: "Error: response-secret\n at dispatch (/srv/private/dispatcher.ts:42:7)", + }, + warning: "Retry after reading C:\\Users\\admin\\private\\warning.json", + }; + + const protectedLegacyPayload = protectPayloadForLog(payload) as typeof payload; + const protectedPipeline = protectPipelinePayloads({ + providerResponse: { body: payload }, + clientResponse: { body: payload }, + }); + const serialized = JSON.stringify({ protectedLegacyPayload, protectedPipeline }); + + assert.doesNotMatch( + serialized, + /response-secret|srv\/private|dispatcher\.ts|C:\\Users|warning\.json/i + ); + assert.equal(protectedLegacyPayload.content, payload.content); + assert.deepEqual(protectedLegacyPayload.usage, payload.usage); + assert.equal(protectedPipeline?.providerResponse?.body?.content, payload.content); + assert.equal(protectedPipeline?.clientResponse?.body?.content, payload.content); +}); + +test("sanitizes in-band error marker objects even when an upstream uses HTTP 200", () => { + const protectedPayload = protectPayloadForLog({ + events: [ + { + type: "error", + content: + "access_token=in-band-secret at /srv/private/in-band.json\n" + + " at dispatch (/srv/private/in-band.ts:3:2)", + }, + ], + content: "Normal sibling content stays available", + }) as { events: Array<{ type: string; content: string }>; content: string }; + + assert.doesNotMatch( + JSON.stringify(protectedPayload.events), + /in-band-secret|srv\/private|in-band\.ts|\bat dispatch\b/i + ); + assert.equal(protectedPayload.content, "Normal sibling content stays available"); +}); + +test("sanitizes serialized error JSON nested below a neutral payload key", () => { + const protectedPayload = protectPayloadForLog({ + payload: JSON.stringify({ + type: "error", + message: "access_token=serialized-secret at /srv/private/serialized.json", + }), + }) as { payload: string }; + + assert.doesNotMatch(protectedPayload.payload, /serialized-secret|srv\/private/i); + assert.equal((JSON.parse(protectedPayload.payload) as { type: string }).type, "error"); +}); + +test("preserves deep successful payloads and still sanitizes deep error leaves", () => { + const successLeaf = { content: "deep successful content", usage: { total_tokens: 2 } }; + const errorLeaf = { + error: { + message: "access_token=deep-error-secret at /srv/private/deep.json", + }, + }; + let deepSuccess: Record = successLeaf; + let deepError: Record = errorLeaf; + for (let depth = 0; depth < 18; depth += 1) { + deepSuccess = { [`level_${depth}`]: deepSuccess }; + deepError = { [`level_${depth}`]: deepError }; + } + + assert.deepEqual(protectPayloadForLog(deepSuccess), deepSuccess); + assert.doesNotMatch( + JSON.stringify(protectPayloadForLog(deepError)), + /deep-error-secret|srv\/private/i + ); +}); + +test("error-mode log protection summarizes opaque binary bodies without enumerating bytes", () => { + assert.equal(protectErrorPayloadForLog(new Uint8Array([1, 2, 3, 4])), "[binary 4 bytes]"); + assert.equal(protectErrorPayloadForLog(Buffer.from([5, 6, 7])), "[binary 3 bytes]"); +}); + +test("error-mode log protection summarizes nested binary bodies without enumerating bytes", () => { + assert.deepEqual( + protectErrorPayloadForLog({ + data: new Uint8Array([11, 22, 33, 44]), + nested: { + body: Buffer.from([55, 66, 77]), + raw: new Uint8Array([88, 99]).buffer, + }, + }), + { + data: "[binary 4 bytes]", + nested: { + body: "[binary 3 bytes]", + raw: "[binary 2 bytes]", + }, + } + ); +}); + +test("sanitizes error frames split across persisted SSE chunks", () => { + const protectedPipeline = protectPipelinePayloads({ + streamChunks: { + provider: [ + '[12:00:00.000] data: {"error":{"message":"access_token=stream-secret at /srv/private/', + 'provider.json","stack":"Error: stream-secret\\n at dispatch (/srv/private/dispatcher.ts:42:7)"}}\n\n', + ], + }, + }); + const storedChunks = protectedPipeline?.streamChunks?.provider ?? []; + const serialized = JSON.stringify(storedChunks); + + assert.doesNotMatch(serialized, /stream-secret|srv\/private|dispatcher\.ts|\bat dispatch\b/i); + assert.match(serialized, /error/); +}); + +test("sanitizes plaintext SSE error events without treating metadata as data frames", () => { + const metadata = 'metadata: {"error":{"message":"healthy diagnostic"}}'; + const protectedPipeline = protectPipelinePayloads({ + streamChunks: { + provider: [ + `${metadata}\nevent: error\ndata: access_token=plain-sse-secret at /srv/private/plain.txt\n\n`, + ], + }, + }); + const storedChunks = protectedPipeline?.streamChunks?.provider ?? []; + const serialized = JSON.stringify(storedChunks); + + assert.doesNotMatch(serialized, /plain-sse-secret|srv\/private|plain\.txt/i); + assert.match(serialized, /event: error/); + assert.equal(storedChunks[0].includes(metadata), true); +}); + +test("sanitizes discriminated SSE and raw NDJSON error records", () => { + const protectedPipeline = protectPipelinePayloads({ + streamChunks: { + provider: [ + 'data: {"type":"error","message":"access_token=sse-json-secret at /srv/private/sse.json"}\n\n', + '{"type":"error","subType":"upstream","message":"Bearer ndjson-secret at C:\\\\Users\\\\admin\\\\private.json"}\n', + '{"type":"error","content":"Error: api_key=lmarena-secret\\n at dispatch (/srv/private/lmarena.ts:8:2)"}\n', + ], + }, + }); + const storedChunks = protectedPipeline?.streamChunks?.provider ?? []; + const serialized = JSON.stringify(storedChunks); + + assert.doesNotMatch( + serialized, + /sse-json-secret|ndjson-secret|lmarena-secret|srv\/private|C:\\\\Users|\bat dispatch\b/i + ); + assert.equal(storedChunks[0].includes('"type":"error"'), true); +}); + +test("sanitizes response last_error aliases in objects, SSE, and NDJSON", () => { + const hostile = "access_token=last-error-secret at /srv/private/last-error.ts"; + const objectPayload = protectPayloadForLog({ + response: { status: "failed", last_error: { message: hostile } }, + }); + const protectedPipeline = protectPipelinePayloads({ + streamChunks: { + provider: [ + `data: ${JSON.stringify({ response: { status: "failed", last_error: { message: hostile } } })}\n\n`, + `${JSON.stringify({ response: { status: "failed", lastError: { message: hostile } } })}\n`, + ], + }, + }); + const serialized = JSON.stringify({ objectPayload, protectedPipeline }); + + assert.doesNotMatch(serialized, /last-error-secret|srv\/private|last-error\.ts/i); + assert.match(serialized, /last_error|lastError/); +}); + +test("sanitizes response.failed messages without rewriting unrelated deep diagnostics", () => { + const hostile = "Bearer response-failed-secret at /srv/private/response-failed.ts:8:2"; + const diagnostics = { + trace: hostile, + output: { trace: hostile }, + level1: { level2: { level3: { level4: { level5: { label: "legitimate diagnostic" } } } } }, + }; + const output = [ + { + type: "message", + role: "assistant", + content: [{ type: "output_text", text: "safe direct partial output" }], + }, + { type: "reasoning", reasoning_content: "private direct reasoning" }, + ]; + const objectPayload = protectPayloadForLog({ + type: "response.failed", + message: hostile, + diagnostics, + output, + }); + const protectedPipeline = protectPipelinePayloads({ + streamChunks: { + provider: [ + `event: response.failed\ndata: ${JSON.stringify({ message: hostile, diagnostics })}\n\n`, + `${JSON.stringify({ type: "response.failed", message: hostile, diagnostics })}\n`, + ], + }, + }); + const serialized = JSON.stringify({ objectPayload, protectedPipeline }); + + assert.doesNotMatch(serialized, /response-failed-secret|srv\/private|response-failed\.ts/i); + assert.deepEqual( + (objectPayload as { diagnostics: typeof diagnostics }).diagnostics.level1, + diagnostics.level1 + ); + assert.deepEqual((objectPayload as { output: unknown }).output, [ + { + type: "message", + role: "assistant", + content: [{ type: "output_text", text: "safe direct partial output", annotations: [] }], + }, + ]); + assert.doesNotMatch(serialized, /private direct reasoning/); + assert.match(serialized, /response\.failed/); +}); + +test("projects nested output when the SSE event alone marks response.failed", () => { + const protectedPipeline = protectPipelinePayloads({ + streamChunks: { + provider: [ + `event: response.failed\ndata: ${JSON.stringify({ + response: { + output: [ + { + type: "message", + role: "assistant", + content: [{ type: "output_text", text: "safe event partial output" }], + }, + { type: "reasoning", reasoning_content: "private event reasoning" }, + ], + }, + })}\n\n`, + ], + }, + }); + const serialized = JSON.stringify(protectedPipeline); + + assert.match(serialized, /safe event partial output/); + assert.doesNotMatch(serialized, /private event reasoning|"reasoning"/); +}); + +test("sanitizes response.completed failed siblings in objects, SSE, and NDJSON", () => { + const hostile = "Bearer completed-failed-secret at /srv/private/completed-failed.ts:8:2"; + const partialOutput = [ + { + id: "msg_partial", + type: "message", + role: "assistant", + status: "in_progress", + diagnostics: { trace: hostile }, + content: [ + { + type: "output_text", + text: "partial safe output", + annotations: [{ type: "url_citation", url: "file:///srv/private/citation" }], + }, + { type: "output_text", phase: "commentary", text: "private commentary" }, + { type: "refusal", refusal: "safe refusal" }, + ], + }, + { + id: "msg_roleless", + type: "message", + content: [{ type: "output_text", text: "private roleless output" }], + }, + { + type: "reasoning", + reasoning_content: "private chain of thought", + encrypted_content: "private encrypted reasoning", + }, + { + type: "function_call", + name: "read_private_file", + arguments: '{"api_key":"private tool argument"}', + }, + ]; + const projectedOutput = [ + { + id: "msg_partial", + type: "message", + role: "assistant", + status: "in_progress", + content: [ + { type: "output_text", text: "partial safe output", annotations: [] }, + { type: "refusal", refusal: "safe refusal" }, + ], + }, + ]; + const completedFailure = { + type: "response.completed", + message: hostile, + response: { + status: "failed", + detail: hostile, + description: hostile, + error: { message: "Upstream request failed" }, + output: partialOutput, + }, + }; + const objectPayload = protectPayloadForLog(completedFailure); + const protectedPipeline = protectPipelinePayloads({ + streamChunks: { + provider: [ + `event: response.completed\ndata: ${JSON.stringify({ message: hostile, response: completedFailure.response })}\n\n`, + `${JSON.stringify(completedFailure)}\n`, + ], + }, + }); + const serialized = JSON.stringify({ objectPayload, protectedPipeline }); + + assert.doesNotMatch( + serialized, + /completed-failed-secret|srv\/private|completed-failed\.ts|private commentary|private roleless|private chain|private encrypted|private tool/i + ); + assert.match(serialized, /"annotations":\[\]/); + assert.doesNotMatch(serialized, /"url_citation"|"diagnostics"|"function_call"|"reasoning"/); + assert.match(serialized, /partial safe output/); + assert.match(serialized, /safe refusal/); + assert.deepEqual( + (objectPayload as { response: { output: typeof projectedOutput } }).response.output, + projectedOutput + ); +}); + +test("sanitizes upstream error bodies by status while preserving successful response bodies", () => { + const successBody = { + message: "Normal response mentions /tmp/public-example.ts and remains diagnostic content", + usage: { total_tokens: 3 }, + }; + const protectedJsonError = protectPipelinePayloads({ + providerResponse: { + status: 502, + statusText: "Bad Gateway", + headers: { "content-type": "application/json" }, + body: { + message: "access_token=json-body-secret at /srv/private/upstream.json", + detail: "Error: api_key=body-stack-secret\n at dispatch (/srv/private/body.ts:4:2)", + }, + }, + }); + const protectedPlaintextError = protectPipelinePayloads({ + providerResponse: { + status: 503, + body: "Bearer plaintext-body-secret at C:\\Users\\admin\\upstream.txt", + }, + }); + const protectedSuccess = protectPipelinePayloads({ + providerResponse: { status: 200, body: successBody }, + }); + const serialized = JSON.stringify({ protectedJsonError, protectedPlaintextError }); + + assert.doesNotMatch( + serialized, + /json-body-secret|body-stack-secret|plaintext-body-secret|srv\/private|C:\\\\Users|\bat dispatch\b/i + ); + assert.equal(protectedJsonError?.providerResponse?.status, 502); + assert.equal(protectedPlaintextError?.providerResponse?.status, 503); + assert.deepEqual(protectedSuccess?.providerResponse?.body, successBody); +}); + test("omits encrypted reasoning values from structured log payloads", () => { const encryptedContent = "encrypted".repeat(128); const payload = { diff --git a/tests/unit/skills-executor.test.ts b/tests/unit/skills-executor.test.ts index 99975c06b1..17349e84b2 100644 --- a/tests/unit/skills-executor.test.ts +++ b/tests/unit/skills-executor.test.ts @@ -4,8 +4,15 @@ import fs from "node:fs"; import os from "node:os"; import path from "node:path"; -const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-skills-executor-")); +const TEST_ROOT = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-skills-executor-")); +const TEST_DATA_DIR = path.join(TEST_ROOT, "data"); +const TEST_PLUGINS_DIR = path.join(TEST_ROOT, "plugins"); +const ORIGINAL_DATA_DIR = process.env.DATA_DIR; +const ORIGINAL_PLUGINS_DIR = process.env.OMNIROUTE_PLUGINS_DIR; +fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +fs.mkdirSync(TEST_PLUGINS_DIR, { recursive: true }); process.env.DATA_DIR = TEST_DATA_DIR; +process.env.OMNIROUTE_PLUGINS_DIR = TEST_PLUGINS_DIR; const coreDb = await import("../../src/lib/db/core.ts"); const settingsDb = await import("../../src/lib/db/settings.ts"); @@ -47,7 +54,11 @@ test.beforeEach(async () => { test.after(() => { resetSkillsRuntime(); coreDb.resetDbInstance(); - fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + if (ORIGINAL_DATA_DIR === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = ORIGINAL_DATA_DIR; + if (ORIGINAL_PLUGINS_DIR === undefined) delete process.env.OMNIROUTE_PLUGINS_DIR; + else process.env.OMNIROUTE_PLUGINS_DIR = ORIGINAL_PLUGINS_DIR; + fs.rmSync(TEST_ROOT, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); }); test("skillExecutor executes a registered handler and persists execution history", async () => { @@ -78,6 +89,119 @@ test("skillExecutor executes a registered handler and persists execution history assert.equal(listed[0].id, execution.id); }); +test("skillExecutor sanitizes failed outputs and nested error subtrees before persistence", async () => { + await registerEchoSkill(); + const hostile = + "tool failed access_token=skill-output-secret at /srv/private/skill-output.ts\n" + + " at run (/srv/private/skill-output.ts:8:2)"; + + skillExecutor.registerHandler("echo-handler", async () => ({ + success: false, + status: 502, + statusText: hostile, + headers: { authorization: "Bearer skill-output-secret" }, + body: hostile, + stdout: hostile, + stderr: hostile, + })); + + const failedOutput = await skillExecutor.execute( + "echo@1.0.0", + { value: "failure" }, + { apiKeyId: "key-a", sessionId: "session-output" } + ); + const storedFailure = skillExecutor.getExecution(failedOutput.id); + const failureSerialized = JSON.stringify({ failedOutput, storedFailure }); + + assert.equal((failedOutput.output as Record)?.status, 502); + assert.doesNotMatch( + failureSerialized, + /skill-output-secret|srv\/private|skill-output\.ts|\bat run\b/i + ); + + skillExecutor.registerHandler("echo-handler", async () => ({ + success: true, + payload: { + value: "preserve me", + error: { message: hostile }, + }, + warning: hostile, + })); + const successfulOutput = await skillExecutor.execute( + "echo@1.0.0", + { value: "success" }, + { apiKeyId: "key-a", sessionId: "session-success" } + ); + const storedSuccess = skillExecutor.getExecution(successfulOutput.id); + const successSerialized = JSON.stringify({ successfulOutput, storedSuccess }); + + assert.equal( + ((successfulOutput.output as Record)?.payload as Record) + ?.value, + "preserve me" + ); + assert.doesNotMatch( + successSerialized, + /skill-output-secret|srv\/private|skill-output\.ts|\bat run\b/i + ); +}); + +test("skillExecutor treats failure discriminators and aliased error objects as boundary failures", async () => { + await registerEchoSkill(); + const hostile = "Bearer skill-discriminator-secret at /srv/private/skill-discriminator.ts:8:2"; + + for (const result of [ + { type: "error", message: hostile }, + { status: "failed", reason: hostile }, + ]) { + skillExecutor.registerHandler("echo-handler", async () => result); + const execution = await skillExecutor.execute( + "echo@1.0.0", + { value: "discriminated-failure" }, + { apiKeyId: "key-a", sessionId: "session-discriminated" } + ); + const stored = skillExecutor.getExecution(execution.id); + assert.equal(execution.status, "error"); + assert.equal(stored?.status, "error"); + assert.doesNotMatch( + JSON.stringify({ execution, stored }), + /skill-discriminator-secret|srv\/private|skill-discriminator\.ts/i + ); + } + + const shared = { message: hostile }; + skillExecutor.registerHandler("echo-handler", async () => ({ + success: true, + payload: { error: shared }, + alias: shared, + })); + const aliased = await skillExecutor.execute( + "echo@1.0.0", + { value: "alias" }, + { apiKeyId: "key-a", sessionId: "session-alias" } + ); + assert.equal(aliased.status, "success"); + assert.doesNotMatch( + JSON.stringify({ aliased, stored: skillExecutor.getExecution(aliased.id) }), + /skill-discriminator-secret|srv\/private|skill-discriminator\.ts/i + ); + + const cyclic: Record = { success: true, error: shared }; + cyclic.self = cyclic; + skillExecutor.registerHandler("echo-handler", async () => cyclic); + const cycleSafe = await skillExecutor.execute( + "echo@1.0.0", + { value: "cycle" }, + { apiKeyId: "key-a", sessionId: "session-cycle" } + ); + assert.equal(cycleSafe.status, "success"); + assert.doesNotThrow(() => JSON.stringify(cycleSafe.output)); + assert.doesNotMatch( + JSON.stringify({ cycleSafe, stored: skillExecutor.getExecution(cycleSafe.id) }), + /skill-discriminator-secret|srv\/private|skill-discriminator\.ts/i + ); +}); + test("skillExecutor blocks execution when Skills are disabled in settings", async () => { await registerEchoSkill(); await settingsDb.updateSettings({ skillsEnabled: false }); @@ -122,7 +246,10 @@ test("skillExecutor turns handler errors and timeouts into error executions", as await registerEchoSkill(); skillExecutor.registerHandler("echo-handler", async () => { - throw new Error("handler exploded"); + throw new Error( + "handler exploded access_token=skill-db-secret at /srv/private/skill-executor.ts\n" + + " at execute (/srv/private/skill-executor.ts:21:5)" + ); }); const failed = await skillExecutor.execute( @@ -134,6 +261,16 @@ test("skillExecutor turns handler errors and timeouts into error executions", as assert.equal(failed.status, "error"); assert.equal(failed.output, null); assert.match(failed.errorMessage, /handler exploded/); + assert.doesNotMatch( + String(failed.errorMessage), + /skill-db-secret|srv\/private|skill-executor\.ts|\bat execute\b/i + ); + const storedFailure = skillExecutor.getExecution(failed.id); + assert.match(String(storedFailure?.errorMessage), /handler exploded/); + assert.doesNotMatch( + String(storedFailure?.errorMessage), + /skill-db-secret|srv\/private|skill-executor\.ts|\bat execute\b/i + ); skillExecutor.registerHandler( "echo-handler", diff --git a/tests/unit/skills-interception.test.ts b/tests/unit/skills-interception.test.ts index f6c6e600f2..2cc8c6acc8 100644 --- a/tests/unit/skills-interception.test.ts +++ b/tests/unit/skills-interception.test.ts @@ -4,12 +4,20 @@ import fs from "node:fs"; import os from "node:os"; import path from "node:path"; -const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-skills-interception-")); +const TEST_ROOT = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-skills-interception-")); +const TEST_DATA_DIR = path.join(TEST_ROOT, "data"); +const TEST_PLUGINS_DIR = path.join(TEST_ROOT, "plugins"); +const ORIGINAL_DATA_DIR = process.env.DATA_DIR; +const ORIGINAL_PLUGINS_DIR = process.env.OMNIROUTE_PLUGINS_DIR; +fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +fs.mkdirSync(TEST_PLUGINS_DIR, { recursive: true }); process.env.DATA_DIR = TEST_DATA_DIR; +process.env.OMNIROUTE_PLUGINS_DIR = TEST_PLUGINS_DIR; const coreDb = await import("../../src/lib/db/core.ts"); const { skillRegistry } = await import("../../src/lib/skills/registry.ts"); const { skillExecutor } = await import("../../src/lib/skills/executor.ts"); +const { builtinSkills } = await import("../../src/lib/skills/builtins.ts"); const { interceptToolCalls, extractToolCalls, handleToolCallExecution, buildWebSearchCallItem } = await import("../../src/lib/skills/interception.ts"); const { OMNIROUTE_WEB_SEARCH_FALLBACK_TOOL_NAME } = @@ -71,7 +79,11 @@ test.beforeEach(async () => { test.after(() => { resetRuntime(); coreDb.resetDbInstance(); - fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + if (ORIGINAL_DATA_DIR === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = ORIGINAL_DATA_DIR; + if (ORIGINAL_PLUGINS_DIR === undefined) delete process.env.OMNIROUTE_PLUGINS_DIR; + else process.env.OMNIROUTE_PLUGINS_DIR = ORIGINAL_PLUGINS_DIR; + fs.rmSync(TEST_ROOT, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); }); test("buildWebSearchCallItem emits a native web_search_call item only for successful web-search fallback results", () => { @@ -231,6 +243,91 @@ test("interceptToolCalls returns outputs, execution errors and missing-skill err ]); }); +test("skill errors are sanitized before OpenAI tool-result response shapes", async () => { + const hostileMessage = + "skill failure access_token=skill-public-secret at /srv/private/skill-handler.ts\n" + + " at execute (/srv/private/skill-handler.ts:17:4)"; + skillExecutor.registerHandler("broken-handler", async () => { + throw new Error(hostileMessage); + }); + + const chatResult = await handleToolCallExecution( + { + choices: [ + { + message: { + tool_calls: [{ id: "chat-error", function: { name: "broken@1.0.0", arguments: "{}" } }], + }, + }, + ], + }, + "gpt-4o-mini", + executionContext + ); + const responsesResult = await handleToolCallExecution( + { + object: "response", + output: [ + { + type: "function_call", + call_id: "responses-error", + name: "broken@1.0.0", + arguments: "{}", + }, + ], + }, + "openai", + executionContext + ); + const thrownResult = await interceptToolCalls( + [{ id: "thrown-error", name: "/srv/private/missing.ts", arguments: {} }], + executionContext + ); + const serialized = JSON.stringify({ chatResult, responsesResult, thrownResult }); + + assert.match(serialized, /skill failure|Skill not found/i); + assert.doesNotMatch( + serialized, + /skill-public-secret|srv\/private|skill-handler\.ts|\bat execute\b/i + ); +}); + +test("failed builtin outputs are sanitized before public tool results", async () => { + const hostile = + "builtin failed access_token=builtin-output-secret at /srv/private/builtin-output.ts\n" + + " at run (/srv/private/builtin-output.ts:9:4)"; + const mutableBuiltins = builtinSkills as unknown as Record< + string, + ( + input: Record, + context: Record + ) => Promise> + >; + const originalHttpRequest = mutableBuiltins.http_request; + + try { + mutableBuiltins.http_request = async () => ({ + success: false, + status: 502, + headers: { authorization: "Bearer builtin-output-secret" }, + body: hostile, + }); + const results = await interceptToolCalls( + [{ id: "builtin-failure", name: "http_request", arguments: { url: "https://example.com" } }], + { ...executionContext, builtinToolNames: ["http_request"] } + ); + const serialized = JSON.stringify(results); + + assert.equal((results[0]?.result as Record)?.status, 502); + assert.doesNotMatch( + serialized, + /builtin-output-secret|srv\/private|builtin-output\.ts|\bat run\b/i + ); + } finally { + mutableBuiltins.http_request = originalHttpRequest; + } +}); + test("handleToolCallExecution appends OpenAI tool results and leaves empty responses untouched", async () => { const openaiResponse = await handleToolCallExecution( { diff --git a/tests/unit/stream-failure-persistent-classification.test.ts b/tests/unit/stream-failure-persistent-classification.test.ts new file mode 100644 index 0000000000..481cf8cc6f --- /dev/null +++ b/tests/unit/stream-failure-persistent-classification.test.ts @@ -0,0 +1,14 @@ +import test from "node:test"; + +import { runIsolatedBoundaryFixture } from "./helpers/runIsolatedBoundaryFixture.ts"; + +test("stream failure persistence boundaries pass in an isolated child process", () => { + runIsolatedBoundaryFixture({ + fixtureUrl: new URL( + "./fixtures/stream-failure-persistent-classification.fixture.ts", + import.meta.url + ), + expectedTests: 2, + label: "stream failure persistence boundaries", + }); +}); diff --git a/tests/unit/stream-passthrough-error-redaction.test.ts b/tests/unit/stream-passthrough-error-redaction.test.ts new file mode 100644 index 0000000000..9da6457def --- /dev/null +++ b/tests/unit/stream-passthrough-error-redaction.test.ts @@ -0,0 +1,446 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { createSSEStream } from "../../open-sse/utils/stream.ts"; +import { FORMATS } from "../../open-sse/translator/formats.ts"; + +type Failure = { status: number; message: string; code?: string; type?: string }; + +async function collectUntilFailure( + chunks: string[], + sourceFormat: string, + convertedLog: string[], + mode: "passthrough" | "translate" = "passthrough", + targetFormat: string = FORMATS.OPENAI +): Promise<{ output: string; error: unknown; failure: Failure | null }> { + let failure: Failure | null = null; + const source = new ReadableStream({ + start(controller) { + for (const chunk of chunks) controller.enqueue(new TextEncoder().encode(chunk)); + controller.close(); + }, + }); + const reader = source + .pipeThrough( + createSSEStream({ + mode, + ...(mode === "translate" ? { targetFormat } : {}), + sourceFormat, + ...(mode === "passthrough" ? { clientResponseFormat: sourceFormat } : {}), + provider: "hostile-upstream", + model: "hostile-model", + body: { input: "hello" }, + reqLogger: { + appendConvertedChunk(value: string) { + convertedLog.push(value); + }, + }, + onFailure(payload) { + failure = payload; + return true; + }, + }) + ) + .getReader(); + + let output = ""; + let error: unknown = null; + try { + while (true) { + const result = await reader.read(); + if (result.done) break; + output += new TextDecoder().decode(result.value); + } + } catch (caught) { + error = caught; + } + return { output, error, failure }; +} + +function assertNoHostileDetail(value: string): void { + assert.doesNotMatch(value, /private-runtime\.ts/); + assert.doesNotMatch(value, /sk-stream-secret/); + assert.doesNotMatch(value, /api_key/); +} + +test("translated root error frames notify onFailure and terminate with a public-safe error", async () => { + const convertedLog: string[] = []; + const raw = { + error: { + type: "server_error", + code: "opaque-provider-code", + message: + "translated failure at /srv/omniroute/private-runtime.ts:47:6 token=sk-stream-secret-xlate", + api_key: "sk-stream-secret-abcdef", + }, + }; + const result = await collectUntilFailure( + [`data: ${JSON.stringify(raw)}\n\n`], + FORMATS.CLAUDE, + convertedLog, + "translate" + ); + + assert.ok(result.error, "a translated upstream error must terminate the stream"); + assert.match(result.output, /event: error/); + assertNoHostileDetail(result.output); + assertNoHostileDetail(convertedLog.join("\n")); + assert.ok(result.failure, "translated failures must reach the internal classifier"); + assert.match(result.failure.message, /private-runtime\.ts/); + assert.equal(result.failure.code, "opaque-provider-code"); + assertNoHostileDetail(String(result.error)); +}); + +test("translated failed response.completed events cannot become successful Chat completions", async () => { + const convertedLog: string[] = []; + const raw = { + type: "response.completed", + response: { + id: "resp_translate_failed", + status: "failed", + output: [], + error: { + type: "server_error", + code: "translated_completed_failure", + message: + "completed translate failure at /srv/omniroute/private-runtime.ts:58:4 token=sk-stream-secret-completed-translate", + }, + }, + }; + const result = await collectUntilFailure( + [`data: ${JSON.stringify(raw)}\n\n`], + FORMATS.OPENAI, + convertedLog, + "translate", + FORMATS.OPENAI_RESPONSES + ); + + assert.ok(result.error, "a failed Responses completion must terminate translated Chat output"); + assert.match(result.output, /"error"/); + assert.doesNotMatch(result.output, /"finish_reason":"stop"/); + assertNoHostileDetail(result.output); + assertNoHostileDetail(convertedLog.join("\n")); + assert.ok(result.failure, "the translated failure must reach fallback classification"); + assert.equal(result.failure.code, "translated_completed_failure"); + assert.match(result.failure.message, /private-runtime\.ts/); + assertNoHostileDetail(String(result.error)); +}); + +test("a translated failed response.completed tail without a newline still terminates", async () => { + const convertedLog: string[] = []; + const raw = { + type: "response.completed", + response: { + id: "resp_translate_failed_tail", + status: "failed", + output: [], + error: { + code: "translated_completed_tail_failure", + message: + "completed tail failure at /srv/omniroute/private-runtime.ts:59:4 token=sk-stream-secret-completed-tail", + }, + }, + }; + const result = await collectUntilFailure( + [`data: ${JSON.stringify(raw)}`], + FORMATS.OPENAI, + convertedLog, + "translate", + FORMATS.OPENAI_RESPONSES + ); + + assert.ok(result.error, "a buffered failed Responses completion must terminate in flush"); + assert.match(result.output, /"error"/); + assert.doesNotMatch(result.output, /"finish_reason":"stop"/); + assertNoHostileDetail(result.output); + assertNoHostileDetail(convertedLog.join("\n")); + assert.ok(result.failure); + assert.equal(result.failure.code, "translated_completed_tail_failure"); + assert.match(result.failure.message, /private-runtime\.ts/); + assertNoHostileDetail(String(result.error)); +}); + +test("Responses response.failed is projected before forwarding, logging, and onFailure", async () => { + const convertedLog: string[] = []; + const raw = { + type: "response.failed", + response: { + id: "resp_hostile-/srv/omniroute/private-runtime.ts-token=sk-stream-secret-id", + model: "provider-model token=sk-stream-secret-model", + status: "failed", + output: [ + { + id: "msg_partial", + type: "message", + role: "assistant", + status: "in_progress", + diagnostics: { + stack: "at /srv/omniroute/private-runtime.ts:47:2", + api_key: "sk-stream-secret-output-diagnostics", + }, + content: [ + { + type: "output_text", + text: "safe partial output", + annotations: [ + { + type: "url_citation", + url: "https://example.invalid/?token=sk-stream-secret-annotation", + title: "at /srv/omniroute/private-runtime.ts:48:2", + }, + ], + }, + { + type: "output_text", + phase: "commentary", + text: "hidden nested commentary must not be public", + }, + { type: "refusal", refusal: "safe refusal" }, + ], + }, + { + id: "msg_commentary", + type: "message", + role: "assistant", + phase: "commentary", + content: [ + { + type: "output_text", + text: "hidden commentary at /srv/omniroute/private-runtime.ts:49:2", + }, + ], + }, + { + id: "msg_roleless", + type: "message", + content: [ + { + type: "output_text", + text: "roleless output must not be public", + annotations: [], + }, + ], + }, + { + id: "reasoning_private", + type: "reasoning", + encrypted_content: "sk-stream-secret-encrypted-reasoning", + summary: [ + { + type: "summary_text", + text: "at /srv/omniroute/private-runtime.ts:50:2", + }, + ], + }, + { + id: "call_private", + type: "function_call", + call_id: "call_private", + name: "read_private_file", + arguments: + '{"path":"/srv/omniroute/private-runtime.ts","api_key":"sk-stream-secret-tool"}', + }, + { + id: "provider_private", + type: "provider_diagnostics", + diagnostics: { + stack: "at /srv/omniroute/private-runtime.ts:51:2", + api_key: "sk-stream-secret-unknown-item", + }, + }, + ], + error: { + type: "server_error", + code: "server_error", + message: "failed at /srv/omniroute/private-runtime.ts:44:2 token=sk-stream-secret-123456", + api_key: "sk-stream-secret-abcdef", + }, + last_error: { + code: "server_error", + message: + "last failure at /srv/omniroute/private-runtime.ts:45:2 token=sk-stream-secret-last", + }, + message: + "sibling failure at /srv/omniroute/private-runtime.ts:46:2 token=sk-stream-secret-sibling", + diagnosis: { stack: "at /srv/omniroute/private-runtime.ts:46:2" }, + settings: { api_key: "sk-stream-secret-response-setting" }, + usage: { + input_tokens: 4, + output_tokens: 2, + total_tokens: 6, + input_tokens_details: { + cached_tokens: 1, + "sk-stream-secret-detail-key": 99, + }, + }, + }, + }; + const result = await collectUntilFailure( + [`event: response.failed\ndata: ${JSON.stringify(raw)}\n\n`], + FORMATS.OPENAI_RESPONSES, + convertedLog + ); + + assert.ok(result.error, "a failed Responses event must terminate the stream"); + assert.match(result.output, /response\.failed/); + assert.match(result.output, /"last_error":\{/); + assert.match(result.output, /safe partial output/); + assert.match(result.output, /safe refusal/); + assert.match(result.output, /"annotations":\[\]/); + assert.doesNotMatch(result.output, /hidden nested commentary must not be public/); + assert.doesNotMatch(result.output, /roleless output must not be public/); + assert.match(result.output, /"cached_tokens":1/); + assert.doesNotMatch(result.output, /\[truncated\]/); + assertNoHostileDetail(result.output); + assertNoHostileDetail(convertedLog.join("\n")); + assert.doesNotMatch( + result.output, + /"diagnosis"|"diagnostics"|"settings"|"encrypted_content"|"function_call"|"provider_diagnostics"|"phase"|"url_citation"/ + ); + assert.doesNotMatch( + convertedLog.join("\n"), + /"diagnosis"|"diagnostics"|"settings"|"encrypted_content"|"function_call"|"provider_diagnostics"|"phase"|"url_citation"/ + ); + assert.ok(result.failure); + assert.match(result.failure.message, /private-runtime\.ts/); + assertNoHostileDetail(String(result.error)); +}); + +test("failed response.completed events omit provider-only diagnostic siblings", async () => { + const convertedLog: string[] = []; + const raw = { + type: "response.completed", + response: { + id: "resp_failed_completed", + object: "response", + created_at: 1_777_777_777, + completed_at: 1_777_777_778, + status: "failed", + output: [], + error: { + code: "server_error", + message: + "completed failure at /srv/omniroute/private-runtime.ts:55:2 token=sk-stream-secret-completed", + }, + diagnosis: { stack: "at /srv/omniroute/private-runtime.ts:55:2" }, + settings: { api_key: "sk-stream-secret-completed-setting" }, + }, + }; + const result = await collectUntilFailure( + [`event: response.completed\ndata: ${JSON.stringify(raw)}\n\n`], + FORMATS.OPENAI_RESPONSES, + convertedLog + ); + + assert.ok(result.error, "a failed response.completed event must terminate the stream"); + assert.match(result.output, /"type":"response\.completed"/); + assert.match(result.output, /"id":"resp_failed_completed"/); + assert.match(result.output, /"created_at":1777777777/); + assert.match(result.output, /"completed_at":1777777778/); + assertNoHostileDetail(result.output); + assertNoHostileDetail(convertedLog.join("\n")); + assert.doesNotMatch(result.output, /"diagnosis"|"settings"/); + assert.doesNotMatch(convertedLog.join("\n"), /"diagnosis"|"settings"/); + assert.ok(result.failure); + assert.match(result.failure.message, /private-runtime\.ts/); + assertNoHostileDetail(String(result.error)); +}); + +test("OpenAI root error frames without a top-level type remain failures after projection", async () => { + const convertedLog: string[] = []; + const raw = { + error: { + type: "server_error", + code: "server_error", + message: "root failed at /srv/omniroute/private-runtime.ts:48:7 token=sk-stream-secret-root", + api_key: "sk-stream-secret-abcdef", + }, + }; + const result = await collectUntilFailure( + [`data: ${JSON.stringify(raw)}\n\n`], + FORMATS.OPENAI, + convertedLog + ); + + assert.ok(result.error, "an OpenAI error envelope must terminate the stream"); + assert.match(result.output, /"error"/); + assertNoHostileDetail(result.output); + assertNoHostileDetail(convertedLog.join("\n")); + assert.ok(result.failure); + assert.match(result.failure.message, /private-runtime\.ts/); + assertNoHostileDetail(String(result.error)); +}); + +test("OpenAI string error frames preserve raw classification but publish only safe text", async () => { + const convertedLog: string[] = []; + const raw = { + error: "string failure at /srv/omniroute/private-runtime.ts:49:8 token=sk-stream-secret-string", + }; + const result = await collectUntilFailure( + [`data: ${JSON.stringify(raw)}\n\n`], + FORMATS.OPENAI, + convertedLog + ); + + assert.ok(result.error, "a string OpenAI error must terminate the stream"); + assertNoHostileDetail(result.output); + assertNoHostileDetail(convertedLog.join("\n")); + assert.ok(result.failure); + assert.match(result.failure.message, /private-runtime\.ts/); + assertNoHostileDetail(String(result.error)); +}); + +test("Claude type:error is projected before forwarding and terminates the stream", async () => { + const convertedLog: string[] = []; + const raw = { + type: "error", + error: { + type: "server_error", + code: "server_error", + message: + "claude failed at /srv/omniroute/private-runtime.ts:51:3 token=sk-stream-secret-123456", + api_key: "sk-stream-secret-abcdef", + }, + }; + const result = await collectUntilFailure( + [`event: error\ndata: ${JSON.stringify(raw)}\n\n`], + FORMATS.CLAUDE, + convertedLog + ); + + assert.ok(result.error, "a Claude error event must terminate the stream"); + assert.match(result.output, /event: error/); + assertNoHostileDetail(result.output); + assertNoHostileDetail(convertedLog.join("\n")); + assert.ok(result.failure); + assert.match(result.failure.message, /private-runtime\.ts/); + assertNoHostileDetail(String(result.error)); +}); + +test("a final response.failed frame without a trailing newline is projected before flush", async () => { + const convertedLog: string[] = []; + const raw = { + type: "response.failed", + response: { + status: "failed", + error: { + code: "server_error", + message: + "tail failed at /srv/omniroute/private-runtime.ts:61:8 token=sk-stream-secret-123456", + api_key: "sk-stream-secret-abcdef", + }, + }, + }; + const result = await collectUntilFailure( + [`event: response.failed\ndata: ${JSON.stringify(raw)}`], + FORMATS.OPENAI_RESPONSES, + convertedLog + ); + + assert.ok(result.error, "a buffered failed event must terminate during flush"); + assert.match(result.output, /response\.failed/); + assertNoHostileDetail(result.output); + assertNoHostileDetail(convertedLog.join("\n")); + assert.ok(result.failure); + assert.match(result.failure.message, /private-runtime\.ts/); + assertNoHostileDetail(String(result.error)); +}); diff --git a/tests/unit/upstream-error-passthrough.test.ts b/tests/unit/upstream-error-passthrough.test.ts index 84458df5f7..a89303c964 100644 --- a/tests/unit/upstream-error-passthrough.test.ts +++ b/tests/unit/upstream-error-passthrough.test.ts @@ -4,6 +4,7 @@ import { shouldPassthroughUpstreamError, buildPassthroughErrorResponse, } from "../../open-sse/utils/upstreamErrorPassthrough.ts"; +import { buildSanitizedUpstreamErrorResponse } from "../../open-sse/utils/upstreamErrorResponse.ts"; test("upstream error passthrough", async (t) => { await t.test("4xx com corpo JSON de erro do provider é elegível", () => { @@ -53,13 +54,24 @@ test("upstream error passthrough", async (t) => { }), false ); + for (const message of [ + String.raw`rejected api_key\t=opaque-tab-secret-9382746`, + String.raw`rejected api_key\u0009=opaque-unicode-tab-9382746`, + String.raw`rejected Bearer\\topaque-bearer-secret-9382746`, + "spawn failed: helper --api-key opaque-cli-key-9382746", + 'spawn failed: helper --token "opaque cli token 9382746"', + "spawn failed: helper --password 'opaque-cli-password-9382746'", + `upstream echoed hf_${"A".repeat(34)}`, + ]) { + assert.equal(shouldPassthroughUpstreamError(422, { error: { message } }), false, message); + } } ); await t.test( "corpo de capacidade/quota sem segredo continua elegível (contrato Claude Code preservado)", () => { - // The common case must still relay verbatim so Claude Code can match the - // wording to auto-disable capabilities. + // The common safe case must preserve wording so Claude Code can match it + // after recursive sanitization and auto-disable capabilities. assert.equal( shouldPassthroughUpstreamError(400, { error: { message: "thinking.type: adaptive is not supported" }, @@ -74,7 +86,7 @@ test("upstream error passthrough", async (t) => { ); } ); - await t.test("buildPassthroughErrorResponse preserva corpo byte-a-byte", async () => { + await t.test("buildPassthroughErrorResponse preserves an already-safe JSON body", async () => { const body = { type: "error", error: { type: "invalid_request_error", message: "thinking.type: nope" }, @@ -89,9 +101,85 @@ test("upstream error passthrough", async (t) => { }); }); +test("passthrough preserves multiline capability wording without stack frames", async () => { + const message = "validation failed\nthinking.type: adaptive is not supported"; + const res = buildPassthroughErrorResponse(400, { + type: "error", + error: { type: "invalid_request_error", message }, + }); + assert.ok(res); + const body = (await res.json()) as { error?: { message?: string } }; + assert.equal(body.error?.message, message); +}); + +test("passthrough removes basename and URL stack frames while preserving prose URLs", async () => { + const hostileMessages = [ + "boom\n at handler (server.js:12:3)", + String.raw`boom\n at handler (server.js:12:3)`, + "boom at handler (http://127.0.0.1:3000/_next/server.js:12:3)", + "boom at handler (webpack-internal:///app/server.js:12:3)", + "boom\n at handler (http://127.0.0.1:3000/_next/server.js?build=abc:12:3)", + String.raw`boom\n at handler (webpack-internal:///app/server.js#chunk:12:3)`, + String.raw`boom at handler (\Windows\Temp\server.js:12:3)`, + "boom\nhandler@file:///home/runner/private.js:12:3", + String.raw`boom\nhandler@/home/runner/private.cts:12:3`, + "boom\nhandler@https://127.0.0.1:3000/_next/server.mts?build=abc:12:3", + "boom at handler (http://127.0.0.1:3000/_next/chunks/route:12:3)", + ]; + + for (const message of hostileMessages) { + const response = buildPassthroughErrorResponse(400, { + type: "error", + error: { type: "invalid_request_error", message }, + }); + assert.ok(response); + const body = (await response.json()) as { error?: { message?: string } }; + assert.equal(body.error?.message, "boom"); + } + + const prose = "See https://example.com/docs/error for recovery guidance"; + const proseResponse = buildPassthroughErrorResponse(400, { + type: "error", + error: { type: "invalid_request_error", message: prose }, + }); + assert.ok(proseResponse); + const proseBody = (await proseResponse.json()) as { error?: { message?: string } }; + assert.equal(proseBody.error?.message, prose); + + const proseWithCoordinates = "See https://example.com/docs/error:12:3 for recovery guidance"; + const proseWithCoordinatesResponse = buildPassthroughErrorResponse(400, { + type: "error", + error: { type: "invalid_request_error", message: proseWithCoordinates }, + }); + assert.ok(proseWithCoordinatesResponse); + const proseWithCoordinatesBody = (await proseWithCoordinatesResponse.json()) as { + error?: { message?: string }; + }; + assert.equal(proseWithCoordinatesBody.error?.message, proseWithCoordinates); +}); + +test("canonical upstream JSON projection redacts URL credentials", async () => { + const response = buildSanitizedUpstreamErrorResponse({ + status: 422, + rawBody: JSON.stringify({ + error: { + message: + "proxy failed https://svc-user:p4ss-opaque-9382@internal.example/v1?" + + "X-Amz-Signature=amz-secret&sig=sas-secret", + }, + }), + fallbackMessage: "Upstream validation failed", + }); + const serialized = await response.text(); + + assert.equal(response.status, 422); + assert.doesNotMatch(serialized, /svc-user|p4ss-opaque|amz-secret|sas-secret/i); + assert.match(serialized, /\[REDACTED\]/); +}); + test("createErrorResult opt-in passthrough (opts.passthrough)", async (t) => { await t.test( - "com opts.passthrough e corpo elegível, result.response é o corpo upstream verbatim", + "com opts.passthrough e corpo elegível, result.response preserva o JSON upstream seguro", async () => { const { createErrorResult } = await import("../../open-sse/utils/error.ts"); const upstreamBody = { From 16b0d4e3ad5d08d95a59fcc5b954cc6a5d6d6b37 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 23:26:32 -0300 Subject: [PATCH 074/143] fix(catalog): re-audit free-tier quotas against official pages (#12649) * test(catalog): pin the 2026-09-02 free-tier re-audit facts for gemini, ollama-cloud, groq, nara and mistral * fix(catalog): re-audit gemini, ollama-cloud, groq, nara and mistral against official pages * fix(catalog): restore the console-verified Mistral 1B pool and harden its regression test * docs(free-tiers): move headline to the re-audited ~1.50B and refresh pool counts * chore(free-tiers): retire stale Groq free-tier text and preset model; fix catalog header * docs(free-tiers): state the evidence-comment rule honestly and retire the last "14.4K RPD" Groq texts * docs(free-tiers): retire the stale Gemini onboarding quota text * docs(free-tier): refresh catalog-entry counts to 442 after base sync * docs(free-tiers): restore README spacing lost in the merge and re-sync the guide counts * docs(free-tiers): re-sync numbers after merging release/v3.8.51 (Cerebras reclassified upstream) * fix(docs): keep the NaraRouter plans endpoint out of the API-path checker; rebaseline gateways.ts (+3) --------- Co-authored-by: diegosouzapw --- README.md | 14 +-- config/quality/file-size-baseline.json | 3 +- docs/diagrams/README.md | 20 +-- docs/diagrams/free-tier-budget.svg | 100 ++++++++------- docs/diagrams/readme-hero.svg | 4 +- docs/getting-started/FREE-TIERS-GUIDE.md | 10 +- docs/reference/FREE_TIERS.md | 42 ++++--- docs/reference/PROVIDER_REFERENCE.md | 6 +- open-sse/config/freeModelCatalog.data.ts | 92 +++++++++----- open-sse/config/freeTierCatalog.ts | 9 +- .../config/providers/registry/groq/index.ts | 1 + .../config/providers/registry/nara/index.ts | 54 +++++++- src/app/(dashboard)/dashboard/combos/page.tsx | 2 +- src/i18n/messages/ar.json | 4 +- src/i18n/messages/az.json | 4 +- src/i18n/messages/bg.json | 4 +- src/i18n/messages/bn.json | 4 +- src/i18n/messages/cs.json | 4 +- src/i18n/messages/da.json | 4 +- src/i18n/messages/de.json | 4 +- src/i18n/messages/en.json | 4 +- src/i18n/messages/es.json | 4 +- src/i18n/messages/fa.json | 4 +- src/i18n/messages/fi.json | 4 +- src/i18n/messages/fr.json | 4 +- src/i18n/messages/gu.json | 4 +- src/i18n/messages/he.json | 4 +- src/i18n/messages/hi.json | 4 +- src/i18n/messages/hu.json | 4 +- src/i18n/messages/id.json | 4 +- src/i18n/messages/it.json | 4 +- src/i18n/messages/ja.json | 4 +- src/i18n/messages/ko.json | 4 +- src/i18n/messages/mr.json | 4 +- src/i18n/messages/ms.json | 4 +- src/i18n/messages/nl.json | 4 +- src/i18n/messages/no.json | 4 +- src/i18n/messages/phi.json | 4 +- src/i18n/messages/pl.json | 4 +- src/i18n/messages/pt-BR.json | 4 +- src/i18n/messages/pt.json | 4 +- src/i18n/messages/ro.json | 4 +- src/i18n/messages/ru.json | 4 +- src/i18n/messages/sk.json | 4 +- src/i18n/messages/sv.json | 4 +- src/i18n/messages/sw.json | 4 +- src/i18n/messages/ta.json | 4 +- src/i18n/messages/te.json | 4 +- src/i18n/messages/th.json | 4 +- src/i18n/messages/tr.json | 4 +- src/i18n/messages/uk-UA.json | 4 +- src/i18n/messages/ur.json | 4 +- src/i18n/messages/vi.json | 4 +- src/i18n/messages/zh-CN.json | 4 +- src/i18n/messages/zh-TW.json | 4 +- .../providers/apikey/frontier-labs.ts | 3 +- .../constants/providers/apikey/gateways.ts | 11 +- .../autoCombo/strict-zero-cost-filter.test.ts | 15 +-- .../unit/free-providers-batch-2026-07.test.ts | 6 +- tests/unit/free-tier-catalog.test.ts | 18 ++- tests/unit/free-tier-reaudit-2026-09.test.ts | 117 ++++++++++++++++++ 61 files changed, 454 insertions(+), 241 deletions(-) create mode 100644 tests/unit/free-tier-reaudit-2026-09.test.ts diff --git a/README.md b/README.md index ca92188728..74d8b0e0ba 100644 --- a/README.md +++ b/README.md @@ -7,19 +7,19 @@ # 🚀 OmniRoute — The Free AI Gateway -OmniRoute — Never stop coding. Every AI tool → 356 providers — 150+ free — through one endpoint. Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini with auto-fallback. RTK + Caveman stacked compression saves 15–95% tokens (~89% avg) — never hit limits. 356 AI providers · 150+ free tiers · ~1.48B free tokens/mo · 19 routing strategies · $0 to start. +OmniRoute — Never stop coding. Every AI tool → 356 providers — 150+ free — through one endpoint. Claude Code, Codex, Cursor, Cline, Copilot & Antigravity into FREE Claude / GPT / Gemini with auto-fallback. RTK + Caveman stacked compression saves 15–95% tokens (~89% avg) — never hit limits. 356 AI providers · 150+ free tiers · ~1.47B free tokens/mo · 19 routing strategies · $0 to start.
-## 💰 ~1.48B Free Tokens / Month +## 💰 ~1.47B Free Tokens / Month
-> Stacking free tiers by hand is painful — dozens of SDKs, dozens of rate limits, and no idea how much you actually have. OmniRoute catalogs **437 free-tier entries across 37 recurring pool keys** and computes the token headline from the **20 pools with a published positive monthly budget**, deduplicated by shared pool. The result stays visible on the dashboard (`/dashboard/free-tiers`). +> Stacking free tiers by hand is painful — dozens of SDKs, dozens of rate limits, and no idea how much you actually have. OmniRoute catalogs **442 free-tier entries across 34 recurring pool keys** and computes the token headline from the **16 pools with a published positive monthly budget plus five per-model Groq caps**, deduplicated by shared pool. The result stays visible on the dashboard (`/dashboard/free-tiers`). -OmniRoute free-tier budget card: ~1.48B free tokens per month steady, up to ~2.10B in the first month with signup credits, from 37 documented recurring pool keys covering 437 cataloged free-tier entries behind one endpoint. Honest pool-deduped math — each shared pool counted once, including 20 recurring pools with a published positive monthly token budget; 13 providers are marked avoid in the terms-risk catalog so you decide. Budget bar includes Mistral 1B, LLM7 150M, Nara 150M, Gemini 60M and smaller pools, plus first-month signup credits and permanently-free no-token-cap providers surfaced separately so they never inflate the headline. Live used/remaining on /dashboard/free-tiers. +OmniRoute free-tier budget card: ~1.47B free tokens per month steady, up to ~2.10B in the first month with signup credits, from 34 documented recurring pool keys covering 442 cataloged free-tier entries behind one endpoint. Honest pool-deduped math — each shared pool counted once, including 16 recurring pools with a published positive monthly token budget plus five per-model Groq caps; 13 providers are marked avoid in the terms-risk catalog so you decide. Budget bar includes Mistral 1B, Nara 210M, LLM7 150M, Groq 30M (five per-model caps) and smaller pools, plus first-month signup credits and permanently-free no-token-cap providers surfaced separately so they never inflate the headline. Live used/remaining on /dashboard/free-tiers. > Animated summary of the live `/dashboard/free-tiers` page. Full methodology (pool dedupe, credit tiers, provider terms): **[docs/reference/FREE_TIERS.md](docs/reference/FREE_TIERS.md)**. > @@ -518,7 +518,7 @@ Pix copia-e-cola: ## 📡 OmniRoute Radar -The main free-tier headline remains **~1.48B tokens/month** from the documented, +The main free-tier headline remains **~1.47B tokens/month** from the documented, pool-deduplicated catalog above. Temporary provider signup credits can separately lift the first month to **~2.10B**. Radar is an optional, signed catalog overlay for people who want fresher free-model availability between OmniRoute releases; the community catalog and every existing free @@ -648,7 +648,7 @@ of your shell history. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md) -> **352 registered providers** across the canonical chat, media, search, local, cloud-agent and system collections, including **152 carrying `hasFree: true` discovery metadata**. The chat model registry covers **229 providers / 2,554 distinct provider-model pairs / 1,283 raw model IDs**; the separate free-budget catalog has **437 per-model rows**, **37 recurring pools** and **52 recurring/keyless free-forever providers**. These are different denominators by design; definitions and pool-deduped calculations live in the [Provider Reference](docs/reference/PROVIDER_REFERENCE.md) and [Free Tiers](docs/reference/FREE_TIERS.md). +> **352 registered providers** across the canonical chat, media, search, local, cloud-agent and system collections, including **152 carrying `hasFree: true` discovery metadata**. The chat model registry covers **229 providers / 2,554 distinct provider-model pairs / 1,283 raw model IDs**; the separate free-budget catalog has **442 per-model rows**, **34 recurring pools** and **52 recurring/keyless free-forever providers**. These are different denominators by design; definitions and pool-deduped calculations live in the [Provider Reference](docs/reference/PROVIDER_REFERENCE.md) and [Free Tiers](docs/reference/FREE_TIERS.md).
@@ -1307,7 +1307,7 @@ Métricas canônicas em 2026-08-24: **1.029 vídeos únicos** · **11.132.922 vi Resilience GuideCircuit breakers, cooldowns, queue, anti-thundering herd, TLS spoofing Auto-Combo Engine16-factor scoring, mode packs, self-healing Proxy Guide3-level proxy system, 1proxy marketplace, registry CRUD - Free TiersConsolidated directory: 37 documented recurring pools / 437 cataloged free-tier entries + Free TiersConsolidated directory: 34 documented recurring pools / 442 cataloged free-tier entries Features GalleryVisual dashboard tour with screenshots Codebase DocumentationBeginner-friendly codebase walkthrough diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index 97b881c510..5ed9877db2 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -1,4 +1,5 @@ { + "_rebaseline_2026_09_03_12649_free_tier_reaudit_gateways": "PR #12649 (fix/free-tier-quota-reaudit) own growth: src/shared/constants/providers/apikey/gateways.ts 1459->1462 (+3 = the nara authHint rewritten for the re-audited 7M/day plan now wraps to two lines, plus the Prettier reflow of two pre-existing >100-col authHint lines (oneminai, freebuff) that lint-staged enforces on any touch of the file; additive text at the existing registry chokepoint, same god-file no-split rationale as prior gateways.ts rebaselines: #11786 seekai, #10987 logfare, #10531 freebuff). Covered by tests/unit/free-tier-reaudit-2026-09.test.ts and tests/unit/free-providers-batch-2026-07.test.ts.", "_rebaseline_2026_09_03_moonshot_native_quota": "PR feat/moonshot-native-quota own growth on release/v3.8.51: src/lib/db/migrationRunner.ts 1201->1206 (+5, case 172 retroactive guard for daily_quota_reset_* columns); src/sse/handlers/chat.ts 2434->2450 (+16, registerMoonshotQuotaFetcher + startup node scan at the existing quota-fetcher registration chokepoint); src/sse/services/auth.ts 3427->3450 (+23, resolveDailyResetForProvider + dailyReset arg on checkFallbackError); open-sse/services/accountFallback.ts 2422->2461 (+39, compatible-node credits_exhausted carve-out + TPD node-clock lock); tests/unit/account-fallback-service.test.ts 2008->2056 (+48, TPD/empty-wallet cases). Wiring at existing chokepoints; Moonshot host predicates, daily reset clock, and the balance fetcher live in new leaves under cap. Covered by tests/unit/moonshot-*.test.ts + account-fallback-service.test.ts (135/135 focused).", "_rebaseline_2026_09_02_11786_seekai_provider": "PR #11786 (feat/11786-seekai-provider, closes #11786) own growth: src/shared/constants/providers/apikey/gateways.ts 1438->1458 (check-file-size split-newline=1459; the seekai APIKEY_PROVIDERS_GATEWAYS catalog entry plus authHint, additive data at the existing registry chokepoint, same god-file no-split rationale as prior gateways.ts rebaselines: #10987 logfare, #10531 freebuff). Covered by tests/unit/seekai-provider.test.ts.", "_rebaseline_2026_09_02_12325_generic_429_invalidate": "PR #12325 own growth: open-sse/handlers/chatCore.ts 5946->5955 (+9 = the non-Codex 429 else-if that drops the generic quota wrapper and stamps force-refresh, plus a source-regex breadcrumb). Irreducible call-site wiring next to the existing Codex 429 invalidateCodexQuotaCache branch; not extractable without splitting handleChatCore mid-response. Covered by tests/unit/generic-quota-fetcher.test.ts (31/31) and tests/unit/antigravity-429-quota-cooldown.test.ts.", @@ -452,7 +453,7 @@ "src/lib/tailscaleTunnel.ts": 1208, "src/lib/tokenHealthCheck.ts": 1218, "src/shared/components/RequestLoggerV2.tsx": 1718, - "src/shared/constants/providers/apikey/gateways.ts": 1459, + "src/shared/constants/providers/apikey/gateways.ts": 1462, "src/shared/services/cliRuntime.ts": 1296, "src/sse/handlers/chat.ts": 2450, "src/sse/services/auth.ts": 3450, diff --git a/docs/diagrams/README.md b/docs/diagrams/README.md index 1382f41263..610e2a4dba 100644 --- a/docs/diagrams/README.md +++ b/docs/diagrams/README.md @@ -10,16 +10,16 @@ Mermaid sources (`.mmd`) and exported SVGs for OmniRoute v3.8.0 architecture flo ## Canonical diagrams -| Source | Exported | Used in | -| ---------------------------------------------------- | ----------------------------------------- | ------------------------------------------------------------------------------ | -| [request-pipeline.mmd](./request-pipeline.mmd) | [SVG](./exported/request-pipeline.svg) | docs/architecture/ARCHITECTURE.md, docs/architecture/CODEBASE_DOCUMENTATION.md | +| Source | Exported | Used in | +| -------------------------------------------------- | ---------------------------------------- | ------------------------------------------------------------------------------ | +| [request-pipeline.mmd](./request-pipeline.mmd) | [SVG](./exported/request-pipeline.svg) | docs/architecture/ARCHITECTURE.md, docs/architecture/CODEBASE_DOCUMENTATION.md | | [auto-combo-scoring.mmd](./auto-combo-scoring.mmd) | [SVG](./exported/auto-combo-scoring.svg) | docs/routing/AUTO-COMBO.md | -| [resilience-3layers.mmd](./resilience-3layers.mmd) | [SVG](./exported/resilience-3layers.svg) | docs/architecture/RESILIENCE_GUIDE.md, CLAUDE.md | -| [i18n-flow.mmd](./i18n-flow.mmd) | [SVG](./exported/i18n-flow.svg) | docs/guides/I18N.md | -| [mcp-tools.mmd](./mcp-tools.mmd) | [SVG](./exported/mcp-tools.svg) | docs/frameworks/MCP-SERVER.md | -| [cloud-agent-flow.mmd](./cloud-agent-flow.mmd) | [SVG](./exported/cloud-agent-flow.svg) | docs/frameworks/CLOUD_AGENT.md | -| [authz-pipeline.mmd](./authz-pipeline.mmd) | [SVG](./exported/authz-pipeline.svg) | docs/architecture/AUTHZ_GUIDE.md | -| [db-schema-overview.mmd](./db-schema-overview.mmd) | [SVG](./exported/db-schema-overview.svg) | docs/architecture/CODEBASE_DOCUMENTATION.md | +| [resilience-3layers.mmd](./resilience-3layers.mmd) | [SVG](./exported/resilience-3layers.svg) | docs/architecture/RESILIENCE_GUIDE.md, CLAUDE.md | +| [i18n-flow.mmd](./i18n-flow.mmd) | [SVG](./exported/i18n-flow.svg) | docs/guides/I18N.md | +| [mcp-tools.mmd](./mcp-tools.mmd) | [SVG](./exported/mcp-tools.svg) | docs/frameworks/MCP-SERVER.md | +| [cloud-agent-flow.mmd](./cloud-agent-flow.mmd) | [SVG](./exported/cloud-agent-flow.svg) | docs/frameworks/CLOUD_AGENT.md | +| [authz-pipeline.mmd](./authz-pipeline.mmd) | [SVG](./exported/authz-pipeline.svg) | docs/architecture/AUTHZ_GUIDE.md | +| [db-schema-overview.mmd](./db-schema-overview.mmd) | [SVG](./exported/db-schema-overview.svg) | docs/architecture/CODEBASE_DOCUMENTATION.md | ## Hand-authored animated diagrams @@ -34,7 +34,7 @@ inside GitHub's `` sandbox: | [combo-always-on.svg](./combo-always-on.svg) | style reference | Animated priority-combo fallback (4 layers, 16s loop). Edit the SVG directly — there is no `.mmd` source. | | [cli-terminal.svg](./cli-terminal.svg) | README.md (root) | Compact half-height animated terminal (1200×350): 3 real CLI commands cycling with typewriter + scrolling subcommand ticker; first frame = completed providers screen. Edit the SVG directly — there is no `.mmd` source. | | [compression-pipeline.svg](./compression-pipeline.svg) | README.md (root) | Animated 12-engine compression funnel (8s loop). Edit the SVG directly — there is no `.mmd` source. | -| [free-tier-budget.svg](./free-tier-budget.svg) | README.md (root) | Animated free-tier budget card (~1.48B/mo quantified headline, 20-pool budget bar, per-pool grid, signup credits, 10s loop). Edit the SVG directly — there is no `.mmd` source. | +| [free-tier-budget.svg](./free-tier-budget.svg) | README.md (root) | Animated free-tier budget card (~1.47B/mo quantified headline, 16-pool + Groq-caps budget bar, per-pool grid, signup credits, 10s loop). Edit the SVG directly — there is no `.mmd` source. | | [readme-hero.svg](./readme-hero.svg) | README.md (root) | Animated hero card (tagline, live provider/free-access headline, full-width compression bar demo, 6 stat chips). Edit the SVG directly — there is no `.mmd` source. | | [promise-pillars.svg](./promise-pillars.svg) | README.md (root) | Animated "The Promise" 6-pillar card (12s border-highlight sweep). Edit the SVG directly — there is no `.mmd` source. | | [why-pain-fix.svg](./why-pain-fix.svg) | README.md (root) | Animated "Why OmniRoute" 10-row pain-vs-fix ledger (15s green row sweep). Edit the SVG directly — there is no `.mmd` source. | diff --git a/docs/diagrams/free-tier-budget.svg b/docs/diagrams/free-tier-budget.svg index cdee0ab39c..19e0ca5f16 100644 --- a/docs/diagrams/free-tier-budget.svg +++ b/docs/diagrams/free-tier-budget.svg @@ -1,5 +1,5 @@ - - Pool-deduplicated chart of the 20 recurring free-token pools with positive published budgets, plus signup credits and uncapped providers shown separately. + + Pool-deduplicated chart of the 16 recurring free-token pools with positive published budgets (plus Groq's five per-model caps as one segment), plus signup credits and uncapped providers shown separately. @@ -61,10 +61,10 @@ - ~1.48B + ~1.47B FREE TOKENS / MONTH · STEADY up to ~2.10B in your first month — signup credits - documented free tiers · 37 recurring pools · 437 catalog entries · one endpoint + documented free tiers · 34 recurring pools · 442 catalog entries · one endpoint @@ -75,66 +75,60 @@ every rate limit · 24/7 we don't publish that - ~1.48B + ~1.47B each shared free pool counted once ✓ 13 providers ToS-flagged — we flag it · you decide - - WHERE IT COMES FROM · 20 QUANTIFIED RECURRING POOLS + + WHERE IT COMES FROM · 16 QUANTIFIED POOLS + 5 GROQ PER-MODEL CAPS - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + - each segment = one recurring pool · widths floored so every pool shows · audited pool budgets below + each segment = one recurring pool (Groq = its five per-model caps) · widths floored so every pool shows · audited pool budgets below - + Mistral 1.00B - LLM7 150M - Nara 150M - Gemini 60M - Cerebras 30M - Cloudflare AI 30M - API Airforce 24M - Ollama Cloud 20M - Groq 15M - Bluesminds 7.2M - SambaNova 6M - Arcee 4.8M - Navy 4.5M - BazaarLink 3.6M - OpenRouter 1.2M - Cohere 800K - HuggingChat 500K - Morph 400K - Hugging Face 200K - Kiro 25K + Nara 210M + LLM7 150M + Groq 30M · 5 caps + Cloudflare AI 30M + API Airforce 24M + Bluesminds 7.2M + SambaNova 6M + Arcee 4.8M + Navy 4.5M + BazaarLink 3.6M + OpenRouter 1.2M + Cohere 800K + HuggingChat 500K + Morph 400K + Hugging Face 200K + Kiro 25K @@ -178,10 +172,14 @@ OpenCode Zen baidu - - + + Gemini + + Ollama Cloud + + - $10 OpenRouter top-up → +24M/mo + $10 OpenRouter top-up → +24M/mo surfaced separately — never inflates the headline diff --git a/docs/diagrams/readme-hero.svg b/docs/diagrams/readme-hero.svg index 7dc4d749a2..6f56f7edfc 100644 --- a/docs/diagrams/readme-hero.svg +++ b/docs/diagrams/readme-hero.svg @@ -1,4 +1,4 @@ - + Animated hero card: a pulse travels the divider line and a compression bar demo repeatedly shrinks a prompt by up to 95 percent; all headline content is static and readable on the first frame. @@ -72,7 +72,7 @@ 90+ FREE TIERS - ~1.48B + ~1.47B FREE TOKENS / MO 15–95% diff --git a/docs/getting-started/FREE-TIERS-GUIDE.md b/docs/getting-started/FREE-TIERS-GUIDE.md index 4108f0babe..804707209d 100644 --- a/docs/getting-started/FREE-TIERS-GUIDE.md +++ b/docs/getting-started/FREE-TIERS-GUIDE.md @@ -1,6 +1,6 @@ # Free Tiers Guide: Understand and Combine Free AI Access -> **TL;DR**: OmniRoute registers 351 provider IDs, with **152 provider-catalog entries marked `hasFree`**. The stricter audited free-model catalog covers **39 recurring pool keys / 445 entries** (438 active + 7 discontinued). Connect several suitable providers for broader fallback capacity; every quota, approval rule, privacy policy, and paid-overage condition still applies. +> **TL;DR**: OmniRoute registers 352 provider IDs, with **152 provider-catalog entries marked `hasFree`**. The stricter audited free-model catalog covers **34 recurring pool keys / 442 entries** (435 active + 7 discontinued). Connect several suitable providers for broader fallback capacity; every quota, approval rule, privacy policy, and paid-overage condition still applies. --- @@ -161,11 +161,11 @@ The live, pool-deduplicated catalog currently reports: | Metric | Current audited value | Interpretation | | ---------------------------------------------------- | -----------------------------------------------: | ----------------------------------------------------------------------------------------- | -| Recurring quantified grant | **~1.48B tokens/month** | Shared pools counted once; excludes uncapped providers from the sum | +| Recurring quantified grant | **~1.47B tokens/month** | Shared pools counted once; excludes uncapped providers from the sum | | First month with signup grants | **~2.10B tokens** | Recurring total plus one-time and recurring credits | -| Audited free-model inventory | **39 recurring pool keys / 445 catalog entries** | 438 active + 7 discontinued; distinct from the 351-provider catalog | -| Recurring/keyless free-forever providers represented | **55** | Unique providers across recurring daily/monthly/credit/uncapped and keyless catalog types | -| Provider catalog entries marked `hasFree` | **152 / 351** | Broader provider metadata; not all have a quantifiable recurring quota | +| Audited free-model inventory | **34 recurring pool keys / 442 catalog entries** | 435 active + 7 discontinued; distinct from the 352-provider catalog | +| Recurring/keyless free-forever providers represented | **52** | Unique providers across recurring daily/monthly/credit/uncapped and keyless catalog types | +| Provider catalog entries marked `hasFree` | **152 / 352** | Broader provider metadata; not all have a quantifiable recurring quota | These values are computed from `open-sse/config/freeModelCatalog.ts`; see the [Free Tiers Reference](../reference/FREE_TIERS.md) for pool deduplication, ToS flags, diff --git a/docs/reference/FREE_TIERS.md b/docs/reference/FREE_TIERS.md index 0decf6fb3f..394142ccd2 100644 --- a/docs/reference/FREE_TIERS.md +++ b/docs/reference/FREE_TIERS.md @@ -1,37 +1,39 @@ --- title: "Free Tiers & Free-Token Budget" version: 3.8.50 -lastUpdated: 2026-08-31 +lastUpdated: 2026-09-02 --- # Free Tiers & Free-Token Budget > **For Users**: Looking for a simple guide? See the [Free Tiers Guide](../getting-started/FREE-TIERS-GUIDE.md) for step-by-step instructions on getting free AI. -> **Last researched:** 2026-06-17 — per-provider web research (official docs + last-7-days news, 50-agent pass with adversarial verification) refreshing every free-tier quota + ToS. +> **Last researched:** 2026-06-17 — per-provider web research (official docs + last-7-days news, 50-agent pass with adversarial verification) refreshing every free-tier quota + ToS. **Partial re-audit 2026-09-02** (`gemini`, `ollama-cloud`, `groq`, `nara`, `mistral` — see the dated note below). > **Source of truth (catalog):** `open-sse/config/freeModelCatalog.ts` (per-MODEL budgets, pool-deduped). The token-budget numbers below come from live web research and are an **approximation** — see [Methodology & caveats](#methodology--caveats). ## TL;DR — how much free inference does OmniRoute actually aggregate? -| Metric | Tokens / month | Meaning | -| ------------------------------------------- | ----------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| **Documented recurring grant (steady)** | **~1.48B** | Free-tier **pools** (per-model catalog), each shared pool counted **once**. The live source behind `/api/free-tier/summary` and the dashboard's Free-Tier Budget page. **Use this number.** | -| **+ first month with signup credits** | **~2.10B** | Steady + one-time signup credits (Together $25, Z.AI 20M, DeepSeek 5M, …), deduped per account. **First month only** — does not recur. | -| **+ permanently free, no published cap** | _un-quantifiable_ | `siliconflow`, `glm-cn` (GLM-4-Flash), `tencent`, `baidu`, `kilo-gateway`, `opencode-zen` — real recurring access, rate/concurrency-limited, **no token cap to count**. Listed, never summed (counting them at `RPM×24/7` is the inflation we reject). | -| **+ deposit-unlock boost** | **+~24M** | A one-time **$10** OpenRouter top-up raises its free pool from 50 → 1000 req/day. Reported separately so it never inflates the steady number. | -| Theoretical ceiling (all rate limits, 24/7) | ~10B | Sum of every provider rate limit extrapolated to non-stop use. **Not a guarantee** — do not headline this. | +| Metric | Tokens / month | Meaning | +| ------------------------------------------- | ----------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| **Documented recurring grant (steady)** | **~1.47B** | Free-tier **pools** (per-model catalog), each shared pool counted **once**. The live source behind `/api/free-tier/summary` and the dashboard's Free-Tier Budget page. **Use this number.** | +| **+ first month with signup credits** | **~2.10B** | Steady + one-time signup credits (Together $25, Z.AI 20M, DeepSeek 5M, …), deduped per account. **First month only** — does not recur. | +| **+ permanently free, no published cap** | _un-quantifiable_ | `siliconflow`, `glm-cn` (GLM-4-Flash), `tencent`, `baidu`, `kilo-gateway`, `opencode-zen`, `gemini`, `ollama-cloud` — real recurring access, rate/concurrency-limited, **no token cap to count**. Listed, never summed (counting them at `RPM×24/7` is the inflation we reject). | +| **+ deposit-unlock boost** | **+~24M** | A one-time **$10** OpenRouter top-up raises its free pool from 50 → 1000 req/day. Reported separately so it never inflates the steady number. | +| Theoretical ceiling (all rate limits, 24/7) | ~10B | Sum of every provider rate limit extrapolated to non-stop use. **Not a guarantee** — do not headline this. | -**Honest headline:** _OmniRoute aggregates **~1.48B documented free tokens per month** (up to ~2.10B in your first month with signup credits) across 37 free-tier pools — plus a long tail of permanently-free, no-cap providers — and RTK + Caveman compression (15–95% token savings) stretches that further._ +**Honest headline:** _OmniRoute aggregates **~1.47B documented free tokens per month** (up to ~2.10B in your first month with signup credits) across 34 free-tier pools — plus a long tail of permanently-free, no-cap providers — and RTK + Caveman compression (15–95% token savings) stretches that further._ > **Why this dropped from the previous ~1.94B.** The 2026-06-17 refresh is an honesty correction, not a loss: `gemini` is now pool-deduped (was inflated by counting each Flash variant separately, 462M → 60M), `cloudflare-ai` corrected to its real 10k-Neurons/day (122M → 30M), `doubao` reclassified as a one-time signup credit (not recurring), and shut-down tiers removed (`chutes`/`phind`/`kluster` discontinued). Partly offset by `llm7` (correct 5M/day → 150M) and new free providers (Kilo, OpenCode Zen, Z.AI GLM-Flash). > > **Further corrected to ~1.37B in v3.8.42:** `longcat` was reclassified from a 150M/mo recurring grant to a one-time 10M signup credit after its free preview ended. Same honesty rule — no provider was dropped by mistake. > -> **Corrected to ~1.48B on 2026-09-03 (#11773):** `cerebras` was reclassified from a 30M/mo recurring grant (old no-card 1M tokens/day trial) to a one-time $5 signup credit that requires a payment method. Same honesty rule as LongCat. +> **Updated on 2026-08-26 after retiring Felo Web:** Felo Web is excluded while its GPL-derived provenance/licensing remains on HOLD; the source reported 38 pool keys at the time. The pool count is live and CI-gated (`check:docs-counts` fails the build if the numbers above drift from `computeFreeModelTotals()`). > -> **Updated on 2026-08-26 after retiring Felo Web:** the source now reports 37 recurring pool keys. Felo Web is excluded while its GPL-derived provenance/licensing remains on HOLD. This is the live, CI-gated number (`check:docs-counts` fails the build if this drifts from `computeFreeModelTotals()`). +> **Re-audited on 2026-09-02 against the providers' own pages** (sources: the `// evidence:` comments next to each re-audited entry in `open-sse/config/freeModelCatalog.data.ts`): `gemini` and `ollama-cloud` no longer publish a token figure (Google removed the per-model free table on 2025-12-23; Ollama's Free plan is "starter usage credits") and are now listed as **uncapped**, never summed (−80M); `groq` is five **per-model** 200K-TPD caps (6M each, +15M) with three retired IDs dropped; `nara` is one 7M/day bucket (+60M, 210M). `mistral`'s 1B is visible only in the account console — see _Evidence classes_ under Methodology. The source reported 35 such keys at that point (−3: `gemini` and `ollama-cloud` moved to the uncapped list, and Groq's per-model caps are not a shared pool). +> +> **Corrected to ~1.47B on 2026-09-03 (#11773):** `cerebras` was reclassified from a 30M/mo recurring grant (old no-card 1M tokens/day trial) to a one-time $5 signup credit that requires a payment method. Same honesty rule as LongCat. The source now reports 34 recurring pool keys and ~1.47B steady. -Biggest **documented** contributors: `mistral` 1.00B, `llm7` 150M, `nara` 150M, `gemini` 60M, `cloudflare-ai` 30M, `api-airforce` 24M. (`longcat` is excluded — its 10M LongCat-2.0 grant is a one-time, KYC-gated signup credit, not a recurring monthly budget.) +Biggest **documented** contributors: `mistral` 1.00B, `nara` 210M, `llm7` 150M, `groq` 30M (five per-model caps), `cloudflare-ai` 30M, `api-airforce` 24M. (`longcat` is excluded — its 10M LongCat-2.0 grant is a one-time, KYC-gated signup credit, not a recurring monthly budget.) > ⚠️ The theoretical ceiling (~10B) is inflated by rate-limit-only providers with **no published token cap** (`tencent`, `siliconflow`, `nvidia`, `baidu`, `glm-cn`, `sparkdesk`) whose figures would be `RPM/TPM × 24/7 × 30d` — a theoretical maximum no single account will sustain. They are **excluded** from the defensible number (shown in the "permanently free, no cap" row instead). This is the same inflation that makes competitors' multi-billion claims unreliable. @@ -74,8 +76,9 @@ purpose. - **What an entry actually vouches for.** No entry carries a per-row confidence rating, and the API serves none — treat every figure above as an estimate of the same, unstated quality. Two facts are different, because they are curated by hand rather than inferred: 5 entries carry an independently documented hard stop, and 13 entries carry a prompt-training disclosure. `hardStopGuaranteed` is set only when the provider's own terms say that exceeding the free allowance refuses the request rather than silently starting to bill you, with the source in a comment next to the entry; it is never defaulted to `true`, and an entry nobody has verified stays unset. So a missing hard-stop flag means "not established", not "known to bill you". - `estMonthlyFreeTokens` = recurring monthly tokens only. **One-time signup credits do not recur** and count as 0. Discontinued tiers are also 0. - Daily token cap → `monthly = daily × 30`. Only RPD documented → `RPD × ~800 output tokens × 30`. Only RPM/TPM (no daily cap) → **uncapped** (see below). -- **Permanently free, but no published token cap** (`siliconflow`, `glm-cn`, `tencent`, `baidu`, `kilo-gateway`, `opencode-zen`): these are real recurring free access, rate/concurrency-limited. We classify them `recurring-uncapped` and **never sum them** — multiplying `RPM × 24/7 × 30d` would produce a fantasy ceiling (the inflation we reject). They are listed so you know they exist. +- **Permanently free, but no published token cap** (`siliconflow`, `glm-cn`, `tencent`, `baidu`, `kilo-gateway`, `opencode-zen`, `gemini`, `ollama-cloud`): these are real recurring free access, rate/concurrency-limited. We classify them `recurring-uncapped` and **never sum them** — multiplying `RPM × 24/7 × 30d` would produce a fantasy ceiling (the inflation we reject). They are listed so you know they exist. - **Deposit-unlock boost:** a one-time small top-up that permanently raises a free quota (OpenRouter: $10 → 1000 req/day ≈ +24M/mo). Reported as a separate figure, kept out of the steady headline. +- **Evidence classes.** The rule: a number in the catalog cites its source in an `// evidence:` comment next to the entry — `public-page` (a provider page anyone can read), `api-public` (an unauthenticated endpoint of the provider, e.g. NaraRouter's public plans endpoint at router.bynara.id), or `console-verified por ` (the figure is only visible inside an account console; the comment records who saw it and when, and the public page that says the cap exists). The state today: the five blocks re-audited on 2026-09-02 carry it (`gemini`, `groq`, `mistral`, `ollama-cloud`, `nara`); entries that predate the 2026-09-02 re-audit inherit the earlier research until they are touched; any **new or changed** number without an evidence comment is a bug. Today only `mistral` is console-verified. --- @@ -185,21 +188,20 @@ purpose. --- -## Per-provider free-tier (refreshed 2026-06-17) +## Per-provider free-tier (refreshed 2026-09-02 for the re-audited rows; 2026-06-17 otherwise) > Regenerated from the per-model catalog (`open-sse/config/freeModelCatalog.ts`), pool-deduped. Sorted by recurring steady tokens/mo. `uncapped*` = permanently free but no published token cap (rate/concurrency-limited) — real access, **not** summed into the headline. `—` = credit-only / keyless / not token-quantifiable. | Provider | Free type | Steady tokens/mo | First-month credit | ToS | Models | | ---------------- | ------------- | ---------------- | ------------------ | --------- | ------ | | `mistral` | recurring | ~1.00B | — | caution | 5 | +| `nara` | recurring | ~210M | — | caution | 8 | | `llm7` | recurring | ~150M | — | caution | 4 | | `longcat` | one-time | — | 10M | caution | 1 | -| `gemini` | recurring | ~60M | — | caution | 4 | | `cerebras` | one-time | — | $5 credit | caution | 2 | | `cloudflare-ai` | recurring | ~30M | — | caution | 9 | +| `groq` | recurring | ~30M | — | caution | 5 | | `api-airforce` | recurring | ~24M | — | caution | 7 | -| `ollama-cloud` | recurring | ~20M | — | ambiguous | 8 | -| `groq` | recurring | ~15M | — | caution | 5 | | `bluesminds` | recurring | ~7M | — | ambiguous | 22 | | `sambanova` | recurring | ~6M | — | caution | 5 | | `arcee-ai` | recurring | ~5M | — | caution | 1 | @@ -212,7 +214,9 @@ purpose. | `kiro` | recurring | ~25K | — | avoid | 12 | | `glm-cn` | uncapped | uncapped\* | ~20M | ok | 4 | | `baidu` | uncapped | uncapped\* | — | caution | 1 | +| `gemini` | uncapped | uncapped\* | — | caution | 4 | | `kilo-gateway` | uncapped | uncapped\* | — | caution | 7 | +| `ollama-cloud` | uncapped | uncapped\* | — | ambiguous | 8 | | `opencode-zen` | uncapped | uncapped\* | — | caution | 6 | | `siliconflow` | uncapped | uncapped\* | — | caution | 10 | | `tencent` | uncapped | uncapped\* | — | caution | 1 | @@ -294,7 +298,7 @@ purpose. - **`gemini`** — The shipped freeNote says "1,500 req/day for Gemini 2.5 Flash" — this was accurate before December 2025. Google cut free-tier limits by 50-80% in December 2025, reducing Gemini 2.5 Flash from 1,500 R… - **`gitlawb`** — The shipped freeNote "Free tier available" is effectively stale. The original free MiMo access was removed in May 2026; the only remaining "free" option is a temporary promotional model (Nemotron 3 U… - **`gitlawb-gmi`** — Partially still accurate — free tier exists but is now narrowed to a single model (Nemotron 3 Ultra) after MiMo free access was revoked in late May 2026. The shipped note "Free tier available" unders… -- **`groq`** — The shipped freeNote "30 RPM / 14.4K RPD" is accurate only for llama-3.1-8b-instant. Most other models (including llama-3.3-70b-versatile) have a much lower 1K RPD cap. The note omits model-specific … +- **`groq`** — The shipped freeNote "30 RPM / 14.4K RPD" is accurate only for llama-3.1-8b-instant. Most other models (including llama-3.3-70b-versatile) have a much lower 1K RPD cap. The note omits model-specific … **Resolved 2026-09-02:** the `freeNote` now reads "Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file." and the catalog carries five per-model 6M caps (llama-3.3-70b-versatile retired from the free tier on 2026-08-16) — see the [2026-09-02 re-audit note](#tldr--how-much-free-inference-does-omniroute-actually-aggregate). - **`huggingchat`** — The shipped freeNote ("Free LLM chat — no subscription required. Rate limits apply.") is partially accurate but significantly understates the restrictions. The free tier now operates on a hard $0.10/… - **`huggingface`** — Significantly tightened. The shipped freeNote ("Free Inference API for thousands of models") implied unlimited/generous free access, but as of mid-2025 the free tier is capped at $0.10/month in recur… - **`hyperbolic`** — Our shipped freeNote says "$1-5 trial credits on signup" — the $1 trial credit portion is accurate, but the "$5" figure refers to the minimum deposit required to unlock GPU rental (not free credits g… diff --git a/docs/reference/PROVIDER_REFERENCE.md b/docs/reference/PROVIDER_REFERENCE.md index e588bb04a3..54bfc297cd 100644 --- a/docs/reference/PROVIDER_REFERENCE.md +++ b/docs/reference/PROVIDER_REFERENCE.md @@ -210,7 +210,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `glm` | `glm` | GLM Coding | API key | [link](https://z.ai/subscribe) | — | | `glm-cn` | `glmcn` | GLM Coding (China) | API key | [link](https://open.bigmodel.cn) | — | | `glmt` | `glmt` | GLM Thinking | API key | [link](https://open.bigmodel.cn) | — | -| `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free tier: 30 RPM / 14.4K RPD — no credit card | +| `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file. | | `haiper` | `hp` | Haiper | API key, video | [link](https://haiper.ai) | Get API key at haiper.ai/haiper-api | | `hcnsec` | `hcnsec` | Huancheng Public API | API key | [link](https://api.hcnsec.cn) | Get API key at api.hcnsec.cn | | `helixmind` | `helixmind` | HelixMind | API key, aggregator | [link](https://helixmind.online) | Previously circulated 3 RPM/50 RPD and no-card claims were not confirmed during the 2026-08-02 audit; current quota and billing require account verification. | @@ -260,7 +260,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each | `naga-ac` | `naga` | Naga.ac | API key, aggregator | [link](https://naga.ac) | Get API key at naga.ac — Google/GitHub/Discord signup available. | | `naga-ai` | `naga-ai` | Naga AI | API key, aggregator | [link](https://naga.ac) | Models marked :free are publicly listed, but no numeric quota is confirmed. Naga's policy warns that free-tier prompts and outputs may be collected or used for training. | | `nanogpt` | `nanogpt` | NanoGPT | API key | [link](https://nano-gpt.com) | — | -| `nara` | `nara` | NaraRouter | API key | [link](https://bynara.id) | Get a free API key via NaraRouter's Telegram channel, then paste it here as a Bearer token. | +| `nara` | `nara` | NaraRouter | API key | [link](https://bynara.id) | Create a free NaraRouter account, link your Telegram (required before /v1 answers), then paste the key here as a Bearer token. | | `navy` | `navy` | NavyAI | API key | [link](https://api.navy) | Create a free API key from the NavyAI dashboard, then paste it here as a Bearer token. | | `nebius` | `nebius` | Nebius AI | API key | [link](https://nebius.com) | ~$1 trial credits on signup for API testing | | `nlpcloud` | `nlpc` | NLP Cloud | API key | [link](https://docs.nlpcloud.com) | Use your NLP Cloud API key in Authorization: Token . OmniRoute targets the chatbot endpoint on https://api.nlpcloud.io/v1/gpu//chatbot by default. | @@ -444,7 +444,7 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each - Catalog: [`src/shared/constants/providers.ts`](../../src/shared/constants/providers.ts) - Registry (per-model details): [`open-sse/config/providerRegistry.ts`](../../open-sse/config/providerRegistry.ts) -- Executors: [`open-sse/executors/`](../../open-sse/executors/) (108 implementations) +- Executors: [`open-sse/executors/`](../../open-sse/executors/) (109 implementations) - Translators: [`open-sse/translator/`](../../open-sse/translator/) ## See Also diff --git a/open-sse/config/freeModelCatalog.data.ts b/open-sse/config/freeModelCatalog.data.ts index a8ba20ae9c..08b54e455d 100644 --- a/open-sse/config/freeModelCatalog.data.ts +++ b/open-sse/config/freeModelCatalog.data.ts @@ -1,12 +1,18 @@ -// AUTO-GENERATED — refreshed by the 2026-06-17 per-provider free-tier research pass. -// 2026-07-20: added the free tiers of providers we could already route but had -// never mapped (requesty, ovhcloud, agnes, glm), plus two new providers (navy, -// aihorde), and reconciled kilo-gateway against its live /models list. -// Source: _tasks/features-v3.8.28/free-tier-research-2026-06-17.raw.json (50-agent web research + adversarial verification). +// HAND-CURATED free-tier catalog — there is no generator; edit the entries below directly. +// Provenance: seeded by the 2026-06-17 per-provider free-tier research pass (50-agent web research + +// adversarial verification); 2026-07-20 added the free tiers of providers we could already route but had +// never mapped (requesty, ovhcloud, agnes, glm), plus two new providers (navy, aihorde), and reconciled +// kilo-gateway against its live /models list; 2026-09-02 re-audited gemini, ollama-cloud, groq, nara and +// mistral against the providers' own pages. +// Evidence: every numeric block MUST carry an `// evidence:` comment naming its source class — +// public-page (a provider page anyone can read), api-public (an unauthenticated provider endpoint), or +// console-verified por (visible only inside an account console). No evidence ⇒ no number +// (the entry stays recurring-uncapped, monthlyTokens 0). Blocks that predate 2026-09-02 and still lack +// the comment inherit the 2026-06-17 research pass; add the comment whenever such a block is touched. // Methodology: honest pool-deduped recurring tokens. "recurring-uncapped" = permanently free but no // published token cap (rate/concurrency-limited) — NOT summed into the steady headline (see freeModelCatalog.ts). // Deposit-unlock boosts (e.g. OpenRouter $10 -> 1000 RPD) live in FREE_TIER_BOOSTS, not per-record. -// Do not edit by hand — re-run the patch generator to refresh. +// Bump FREE_CATALOG_CURATED_AT on every change to the entries below. import type { FreeModelBudget } from "./freeModelCatalog.ts"; /** @@ -16,7 +22,7 @@ import type { FreeModelBudget } from "./freeModelCatalog.ts"; * rewrites file timestamps on every deploy, which would report a months-old * catalog as "updated today". Bump this whenever the entries below change. */ -export const FREE_CATALOG_CURATED_AT = "2026-08-30"; +export const FREE_CATALOG_CURATED_AT = "2026-09-03"; export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "agentrouter", modelId: "claude-opus-4-8", displayName: "Claude Opus 4.8", monthlyTokens: 0, creditTokens: 200000000, freeType: "one-time-initial", poolKey: "agentrouter", tos: "caution" }, @@ -176,20 +182,31 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "freemodel-dev", modelId: "gpt-5.3-codex", displayName: "GPT-5.3 Codex", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "freemodel-dev", tos: "unknown" }, { provider: "friendliai", modelId: "meta-llama-3.1-70b-instruct", displayName: "meta-llama-3.1-70b-instruct", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "friendliai", tos: "avoid" }, { provider: "friendliai", modelId: "meta-llama-3.1-8b-instruct", displayName: "meta-llama-3.1-8b-instruct", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "friendliai", tos: "avoid" }, - { provider: "gemini", modelId: "gemini-2.5-flash", displayName: "Gemini 2.5 Flash", monthlyTokens: 60000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "gemini-free", tos: "caution" }, - { provider: "gemini", modelId: "gemini-2.5-flash-lite", displayName: "Gemini 2.5 Flash-Lite", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-daily", poolKey: "gemini-free", tos: "caution" }, - { provider: "gemini", modelId: "gemini-3-flash-preview", displayName: "Gemini 3 Flash Preview", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-daily", poolKey: "gemini-free", tos: "caution" }, - { provider: "gemini", modelId: "gemini-3.1-flash-lite", displayName: "Gemini 3.1 Flash-Lite", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-daily", poolKey: "gemini-free", tos: "caution" }, + // evidence: public-page https://ai.google.dev/gemini-api/docs/rate-limits (2026-08-18) — the per-model + // free-tier table was removed on 2025-12-23; the page now only says limits "can be viewed in Google AI + // Studio" and are "applied per project". No published token/RPD figure ⇒ recurring-uncapped (listed, + // never summed). Re-verify if Google republishes a table. + { provider: "gemini", modelId: "gemini-2.5-flash", displayName: "Gemini 2.5 Flash", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "gemini-free", tos: "caution" }, + { provider: "gemini", modelId: "gemini-2.5-flash-lite", displayName: "Gemini 2.5 Flash-Lite", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "gemini-free", tos: "caution" }, + { provider: "gemini", modelId: "gemini-3-flash-preview", displayName: "Gemini 3 Flash Preview", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "gemini-free", tos: "caution" }, + { provider: "gemini", modelId: "gemini-3.1-flash-lite", displayName: "Gemini 3.1 Flash-Lite", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "gemini-free", tos: "caution" }, { provider: "glm-cn", modelId: "glm-4-flash", displayName: "GLM-4-Flash", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "zhipu-flash-free", tos: "ok" }, { provider: "glm-cn", modelId: "glm-4.5-flash", displayName: "GLM-4.5-Flash", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "zhipu-flash-free", tos: "ok" }, { provider: "glm-cn", modelId: "glm-4.7-flash", displayName: "GLM-4.7-Flash", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "zhipu-flash-free", tos: "ok" }, { provider: "glm-cn", modelId: "glm-signup-bonus", displayName: "Z.AI — 20M signup bonus", monthlyTokens: 0, creditTokens: 20000000, freeType: "one-time-initial", poolKey: "zhipu-signup", tos: "ok" }, - // hardStopGuaranteed: Groq pricing page states "Free tier: 30 RPM / 14.4K RPD — no credit card" (open-sse/services/../providers/apikey/frontier-labs.ts:71-81). - { provider: "groq", modelId: "meta-llama/llama-4-scout-17b-16e-instruct", displayName: "Llama 4 Scout", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution", hardStopGuaranteed: true }, - { provider: "groq", modelId: "llama-3.3-70b-versatile", displayName: "Llama 3.3 70B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution", hardStopGuaranteed: true }, - { provider: "groq", modelId: "openai/gpt-oss-120b", displayName: "GPT-OSS 120B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution", hardStopGuaranteed: true }, - { provider: "groq", modelId: "openai/gpt-oss-20b", displayName: "GPT-OSS 20B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution", hardStopGuaranteed: true }, - { provider: "groq", modelId: "qwen/qwen3-32b", displayName: "Qwen3 32B", monthlyTokens: 15000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "groq", tos: "caution", hardStopGuaranteed: true }, + // evidence: public-page https://console.groq.com/docs/rate-limits (2026-09-02) — "Free Plan Limits": + // 200K TPD per model for the five chat models below; "Rate limits apply at the organization level". + // 200K × 30 = 6M per model; the cap is per model, so each row counts on its own (poolKey null). + // hardStopGuaranteed: same page — "When you exceed rate limits, our API returns a 429 Too Many Requests"; + // https://console.groq.com/docs/billing-faqs — the Free tier has no payment method on file ("To upgrade + // from the Free tier to the Developer tier, you'll need to provide a valid payment method"). + // Retired from the free tier (https://console.groq.com/docs/deprecations): llama-4-scout and qwen3-32b + // (2026-07-17), llama-3.3-70b-versatile (2026-08-16) — deliberately absent below. + { provider: "groq", modelId: "openai/gpt-oss-120b", displayName: "GPT-OSS 120B", monthlyTokens: 6000000, creditTokens: 0, freeType: "recurring-daily", poolKey: null, tos: "caution", hardStopGuaranteed: true }, + { provider: "groq", modelId: "openai/gpt-oss-20b", displayName: "GPT-OSS 20B", monthlyTokens: 6000000, creditTokens: 0, freeType: "recurring-daily", poolKey: null, tos: "caution", hardStopGuaranteed: true }, + { provider: "groq", modelId: "openai/gpt-oss-safeguard-20b", displayName: "GPT-OSS Safeguard 20B", monthlyTokens: 6000000, creditTokens: 0, freeType: "recurring-daily", poolKey: null, tos: "caution", hardStopGuaranteed: true }, + { provider: "groq", modelId: "qwen/qwen3.6-27b", displayName: "Qwen3.6 27B", monthlyTokens: 6000000, creditTokens: 0, freeType: "recurring-daily", poolKey: null, tos: "caution", hardStopGuaranteed: true }, + { provider: "groq", modelId: "qwen/qwen3.8-27b", displayName: "Qwen3.8 27B", monthlyTokens: 6000000, creditTokens: 0, freeType: "recurring-daily", poolKey: null, tos: "caution", hardStopGuaranteed: true }, { provider: "huggingchat", modelId: "baidu/ERNIE-4.5-VL-424B-A47B-Base-PT", displayName: "ERNIE 4.5 VL 424B A47B Base PT", monthlyTokens: 500000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "huggingchat", tos: "caution" }, { provider: "huggingchat", modelId: "CohereLabs/c4ai-command-r7b-12-2024", displayName: "Command R7B 12-2024", monthlyTokens: 500000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "huggingchat", tos: "caution" }, { provider: "huggingchat", modelId: "CohereLabs/command-a-reasoning-08-2025", displayName: "Command A Reasoning 08-2025", monthlyTokens: 500000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "huggingchat", tos: "caution" }, @@ -258,6 +275,13 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "llm7", modelId: "deepseek-r1-0528", displayName: "DeepSeek R1 (LLM7)", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-daily", poolKey: "llm7-free", tos: "caution" }, { provider: "llm7", modelId: "qwen2.5-coder-32b-instruct", displayName: "Qwen2.5 Coder 32B (LLM7)", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-daily", poolKey: "llm7-free", tos: "caution" }, { provider: "longcat", modelId: "LongCat-2.0", displayName: "LongCat-2.0", monthlyTokens: 0, creditTokens: 10000000, freeType: "one-time-initial", poolKey: "longcat-free", tos: "caution" }, + // evidence: console-verified 2026-09-02 por diegosouzapw (https://console.mistral.ai → Limits, Free mode, + // "Tokens per month" = 1,000,000,000). Public pages only confirm that the cap exists: + // https://docs.mistral.ai/admin/billing-usage/usage-limits — "Free mode lets you create API keys and use + // included monthly usage within the limits shown on the Limits page"; + // https://help.mistral.ai/en/articles/698531 — "Tokens per month: overall consumption cap", "set at the + // organization level". Re-verify in the console whenever this block is touched; without a dated + // console-verified line above, this pool MUST become recurring-uncapped (0). { provider: "mistral", modelId: "mistral-large-latest", displayName: "Mistral Large 3", monthlyTokens: 1000000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "mistral", tos: "caution" }, { provider: "mistral", modelId: "mistral-medium-3-5", displayName: "Mistral Medium 3.5", monthlyTokens: 1000000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "mistral", tos: "caution" }, { provider: "mistral", modelId: "mistral-small-latest", displayName: "Mistral Small 4", monthlyTokens: 1000000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "mistral", tos: "caution" }, @@ -283,14 +307,17 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "nvidia", modelId: "google/gemma-4-31b-it", displayName: "Gemma 4 31B", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, { provider: "nvidia", modelId: "nvidia/nemotron-3-super-120b-a12b", displayName: "Nemotron 3 Super 120B A12B", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, { provider: "nvidia", modelId: "openai/gpt-oss-120b", displayName: "GPT OSS 120B", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "nvidia", tos: "caution" }, - { provider: "ollama-cloud", modelId: "deepseek-v4-pro", displayName: "DeepSeek V4 Pro", monthlyTokens: 20000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "ollama-cloud", tos: "ambiguous" }, - { provider: "ollama-cloud", modelId: "deepseek-v4-flash", displayName: "DeepSeek V4 Flash", monthlyTokens: 20000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "ollama-cloud", tos: "ambiguous" }, - { provider: "ollama-cloud", modelId: "kimi-k2.6", displayName: "Kimi K2.6", monthlyTokens: 20000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "ollama-cloud", tos: "ambiguous" }, - { provider: "ollama-cloud", modelId: "glm-5.1", displayName: "GLM 5.1", monthlyTokens: 20000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "ollama-cloud", tos: "ambiguous" }, - { provider: "ollama-cloud", modelId: "minimax-m2.7", displayName: "MiniMax M2.7", monthlyTokens: 20000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "ollama-cloud", tos: "ambiguous" }, - { provider: "ollama-cloud", modelId: "gemma4:31b", displayName: "Gemma 4 31B", monthlyTokens: 20000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "ollama-cloud", tos: "ambiguous" }, - { provider: "ollama-cloud", modelId: "nemotron-3-super", displayName: "NVIDIA Nemotron 3 Super", monthlyTokens: 20000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "ollama-cloud", tos: "ambiguous" }, - { provider: "ollama-cloud", modelId: "qwen3.5:397b", displayName: "Qwen 3.5 397B", monthlyTokens: 20000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "ollama-cloud", tos: "ambiguous" }, + // evidence: public-page https://ollama.com/pricing (2026-09-02) — Free plan: "Starter usage credits + // included · Includes access to starter models · Add credits to unlock all models"; docs.ollama.com/cloud: + // "usage resets monthly". No token figure and no named starter-model list ⇒ recurring-uncapped. + { provider: "ollama-cloud", modelId: "deepseek-v4-pro", displayName: "DeepSeek V4 Pro", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "ollama-cloud", tos: "ambiguous" }, + { provider: "ollama-cloud", modelId: "deepseek-v4-flash", displayName: "DeepSeek V4 Flash", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "ollama-cloud", tos: "ambiguous" }, + { provider: "ollama-cloud", modelId: "kimi-k2.6", displayName: "Kimi K2.6", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "ollama-cloud", tos: "ambiguous" }, + { provider: "ollama-cloud", modelId: "glm-5.1", displayName: "GLM 5.1", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "ollama-cloud", tos: "ambiguous" }, + { provider: "ollama-cloud", modelId: "minimax-m2.7", displayName: "MiniMax M2.7", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "ollama-cloud", tos: "ambiguous" }, + { provider: "ollama-cloud", modelId: "gemma4:31b", displayName: "Gemma 4 31B", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "ollama-cloud", tos: "ambiguous" }, + { provider: "ollama-cloud", modelId: "nemotron-3-super", displayName: "NVIDIA Nemotron 3 Super", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "ollama-cloud", tos: "ambiguous" }, + { provider: "ollama-cloud", modelId: "qwen3.5:397b", displayName: "Qwen 3.5 397B", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "ollama-cloud", tos: "ambiguous" }, { provider: "opencode", modelId: "big-pickle", displayName: "Big Pickle", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "opencode", tos: "avoid" }, { provider: "opencode", modelId: "deepseek-v4-flash-free", displayName: "DeepSeek V4 Flash Free", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "opencode", tos: "avoid" }, { provider: "opencode", modelId: "minimax-m2.5-free", displayName: "MiniMax M2.5 Free", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "opencode", tos: "avoid" }, @@ -459,7 +486,16 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "routeway", modelId: "laguna-m.1:free", displayName: "Laguna M.1 (free)", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "routeway-free", tos: "caution" }, { provider: "routeway", modelId: "laguna-xs.2:free", displayName: "Laguna XS.2 (free)", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "routeway-free", tos: "caution" }, { provider: "routeway", modelId: "llama-3.2-3b-instruct:free", displayName: "Llama 3.2 3B Instruct (free)", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "routeway-free", tos: "caution" }, - { provider: "nara", modelId: "tencent-hy3", displayName: "Tencent Hy3", monthlyTokens: 150000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "nara-free", tos: "caution" }, - { provider: "nara", modelId: "mistral-large", displayName: "Mistral Large", monthlyTokens: 150000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "nara-free", tos: "caution" }, - { provider: "nara", modelId: "mistral-medium-3-5", displayName: "Mistral Medium 3.5", monthlyTokens: 150000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "nara-free", tos: "caution" }, + // evidence: api-public https://router.bynara.id/api/plans (2026-09-02) — plan "free": token_cap_daily=7000000, + // rpm_limit=15, models=[agnes-2.0-flash, agnes-2.5-flash, laguna-s-2.1, minimax-m3-free, mistral-large, + // mistral-medium-3-5, qwen3.8-27b, stepfun-3.7-flash]; home: "Token Cap 7M / day · Free tokens reset daily + // at 07:00 WIB". One daily bucket per account ⇒ single pool: 7M × 30 = 210M. Key requires linking Telegram. + { provider: "nara", modelId: "agnes-2.0-flash", displayName: "Agnes 2.0 Flash", monthlyTokens: 210000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "nara-free", tos: "caution" }, + { provider: "nara", modelId: "agnes-2.5-flash", displayName: "Agnes 2.5 Flash", monthlyTokens: 210000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "nara-free", tos: "caution" }, + { provider: "nara", modelId: "laguna-s-2.1", displayName: "Laguna S 2.1", monthlyTokens: 210000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "nara-free", tos: "caution" }, + { provider: "nara", modelId: "minimax-m3-free", displayName: "MiniMax M3 Free", monthlyTokens: 210000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "nara-free", tos: "caution" }, + { provider: "nara", modelId: "mistral-large", displayName: "Mistral Large", monthlyTokens: 210000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "nara-free", tos: "caution" }, + { provider: "nara", modelId: "mistral-medium-3-5", displayName: "Mistral Medium 3.5", monthlyTokens: 210000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "nara-free", tos: "caution" }, + { provider: "nara", modelId: "qwen3.8-27b", displayName: "Qwen3.8 27B", monthlyTokens: 210000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "nara-free", tos: "caution" }, + { provider: "nara", modelId: "stepfun-3.7-flash", displayName: "StepFun 3.7 Flash", monthlyTokens: 210000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "nara-free", tos: "caution" }, ]; diff --git a/open-sse/config/freeTierCatalog.ts b/open-sse/config/freeTierCatalog.ts index 68cc240caa..320ad62cb5 100644 --- a/open-sse/config/freeTierCatalog.ts +++ b/open-sse/config/freeTierCatalog.ts @@ -6,19 +6,20 @@ * (explicit daily/monthly token cap, or documented RPD × ~800 tokens × 30). * * Deliberately EXCLUDED (rate-limit-only, no published token cap — theoretical, - * not granted): tencent, siliconflow, nvidia, baidu, publicai, sparkdesk. + * not granted): tencent, siliconflow, nvidia, baidu, publicai, sparkdesk, + * gemini (no per-model limits published since 2025-12), ollama-cloud (starter + * credits, no figure). * One-time signup credits and discontinued tiers are excluded (do not recur). */ export type TosVerdict = "ok" | "caution" | "ambiguous" | "avoid" | "unknown"; export const FREE_TIER_BUDGETS: Record = { mistral: 1_000_000_000, + nara: 210_000_000, "cloudflare-ai": 122_000_000, - gemini: 60_000_000, doubao: 60_000_000, + groq: 30_000_000, "api-airforce": 24_000_000, - "ollama-cloud": 20_000_000, - groq: 15_000_000, bluesminds: 7_200_000, sambanova: 6_000_000, "arcee-ai": 4_800_000, diff --git a/open-sse/config/providers/registry/groq/index.ts b/open-sse/config/providers/registry/groq/index.ts index 974e24e710..154ca7a574 100644 --- a/open-sse/config/providers/registry/groq/index.ts +++ b/open-sse/config/providers/registry/groq/index.ts @@ -24,6 +24,7 @@ export const groqProvider: RegistryEntry = { { id: "openai/gpt-oss-20b", name: "GPT-OSS 20B" }, { id: "qwen/qwen3-32b", name: "Qwen3 32B" }, { id: "qwen/qwen3.6-27b", name: "Qwen3.6 27B" }, + { id: "qwen/qwen3.8-27b", name: "Qwen3.8 27B" }, { id: "openai/gpt-oss-safeguard-20b", name: "GPT-OSS Safeguard 20B" }, ], }; diff --git a/open-sse/config/providers/registry/nara/index.ts b/open-sse/config/providers/registry/nara/index.ts index e2f840d4e7..177e8f39ba 100644 --- a/open-sse/config/providers/registry/nara/index.ts +++ b/open-sse/config/providers/registry/nara/index.ts @@ -4,16 +4,60 @@ import { buildOpenAiCompatibleRegistryEntry } from "../../shared.ts"; /** * NaraRouter — OpenAI-compatible aggregator (router.bynara.id). * - * Free key issued via their Telegram channel. The free tier is a shared - * 5M-tokens/day pool; many models are gated behind - * credit/plan, so only the free-tier models are pinned. + * Free key issued after linking a Telegram account. The free plan is one + * 7M-tokens/day bucket per account (GET /api/plans, 2026-09-02); only the + * plan's own models are pinned. Context lengths mirror the same models in + * our own registry (agnes, poolside, novita, stepfun); qwen3.8-27b has no + * published context yet, so it carries none. */ export const naraProvider: RegistryEntry = buildOpenAiCompatibleRegistryEntry({ id: "nara", baseUrl: "https://router.bynara.id/v1/chat/completions", models: [ - { id: "tencent-hy3", name: "Tencent Hy3", contextLength: 1000000 }, + { + id: "agnes-2.0-flash", + name: "Agnes 2.0 Flash", + contextLength: 262144, + toolCalling: true, + supportsVision: true, + supportsReasoning: true, + }, + { + id: "agnes-2.5-flash", + name: "Agnes 2.5 Flash", + contextLength: 524288, + toolCalling: true, + supportsVision: true, + supportsReasoning: true, + }, + { + id: "laguna-s-2.1", + name: "Laguna S 2.1", + contextLength: 262144, + toolCalling: true, + supportsReasoning: true, + }, + { + id: "minimax-m3-free", + name: "MiniMax M3 (free)", + contextLength: 1000000, + supportsVision: true, + supportsReasoning: true, + }, { id: "mistral-large", name: "Mistral Large", contextLength: 252000, toolCalling: true }, - { id: "mistral-medium-3-5", name: "Mistral Medium 3.5", contextLength: 256000, toolCalling: true, supportsVision: true }, + { + id: "mistral-medium-3-5", + name: "Mistral Medium 3.5", + contextLength: 256000, + toolCalling: true, + supportsVision: true, + }, + { id: "qwen3.8-27b", name: "Qwen3.8 27B", toolCalling: true }, + { + id: "stepfun-3.7-flash", + name: "StepFun 3.7 Flash", + contextLength: 262144, + toolCalling: true, + }, ], }); diff --git a/src/app/(dashboard)/dashboard/combos/page.tsx b/src/app/(dashboard)/dashboard/combos/page.tsx index 0105dd8723..1a666b3634 100644 --- a/src/app/(dashboard)/dashboard/combos/page.tsx +++ b/src/app/(dashboard)/dashboard/combos/page.tsx @@ -2868,7 +2868,7 @@ function ComboFormModal({ isOpen, combo, onClose, onSave, activeProviders, combo { model: "if/qwen3-coder-plus", weight: 0 }, { model: "if/deepseek-v3.2", weight: 0 }, { model: "nvidia/llama-3.3-70b-instruct", weight: 0 }, - { model: "groq/llama-3.3-70b-versatile", weight: 0 }, + { model: "groq/openai/gpt-oss-120b", weight: 0 }, ]; const PAID_PREMIUM_PRESET_MODELS = [ diff --git a/src/i18n/messages/ar.json b/src/i18n/messages/ar.json index bdf4b7a4c7..f470c997b5 100644 --- a/src/i18n/messages/ar.json +++ b/src/i18n/messages/ar.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "بروكسي API مخفض لأكثر من 40 نموذجًا بما في ذلك GPT-5 و Claude Opus 4.6 و Claude Sonnet 4.6 و Qwen 3.5. احصل على مفتاح API الخاص بك من https://freeaiapikey.com/dashboard. عنوان URL الأساسي: https://freeaiapikey.com/v1.", "freemodel-dev": "احصل على رصيد API مجاني بقيمة 300 دولار على https://freemodel.dev — لا يلزم إدخال معلومات الدفع. نقطة نهاية متوافقة مع OpenAI. تتوفر نماذج GPT-5.4 و GPT-5.5.", "friendliai": "فئة مجانية للاستدلال بدون خادم — لا يلزم وجود بطاقة ائتمان", - "gemini": "مجاني للأبد: 1,500 طلب/يوم لـ Gemini 2.5 Flash — بدون بطاقة ائتمان، احصل على المفتاح من aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "ربط GigaChat (Sber) بمفتاح API.", "gitlab": "رمز وصول شخصي لـ GitLab لواجهة برمجة تطبيقات مقترحات الأكواد العامة. قم بتكوين عنوان URL أساسي مستضاف ذاتيًا عند عدم استخدام gitlab.com.", "gitlawb-gmi": "احصل على مفتاح API الخاص بك من لوحة تحكم Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "ربط GLM Coding (الصين) بمفتاح API.", "glmt": "ملف تعريف GLM مسبق الضبط بميزانية رموز أعلى، وتمكين التفكير، ومهلة أطول.", "getgoapi": "ربط GoAPI بمفتاح API.", - "groq": "الفئة المجانية: 30 طلبًا في الدقيقة / 14.4 ألف طلب في اليوم — بدون بطاقة ائتمان", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "احصل على مفتاح API من haiper.ai/haiper-api", "heroku": "ربط Heroku AI بمفتاح API.", "hcnsec": "احصل على مفتاح API من api.hcnsec.cn", diff --git a/src/i18n/messages/az.json b/src/i18n/messages/az.json index 543b6b60b0..10cd89ab9d 100644 --- a/src/i18n/messages/az.json +++ b/src/i18n/messages/az.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5 daxil olmaqla 40-dan çox model üçün endirimli API proksisi. API açarınızı https://freeaiapikey.com/dashboard ünvanından əldə edin. Baza URL: https://freeaiapikey.com/v1.", "freemodel-dev": "https://freemodel.dev ünvanında $300 pulsuz API krediti əldə edin — ödəniş məlumatı tələb olunmur. OpenAI ilə uyğun son nöqtə. GPT-5.4 və GPT-5.5 modelləri mövcuddur.", "friendliai": "Serverless çıxarış üçün pulsuz tarif — kredit kartı tələb olunmur", - "gemini": "Həmişə pulsuz: Gemini 2.5 Flash üçün gündə 1,500 sorğu — kredit kartı yoxdur, açarı aistudio.google.com ünvanından əldə edin", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "GigaChat (Sber)-ı API açarı ilə qoşun.", "gitlab": "İctimai Code Suggestions API üçün GitLab şəxsi giriş tokeni. gitlab.com istifadə etmədikdə, self-hosted baza URL-i konfiqurasiya edin.", "gitlawb-gmi": "API açarınızı Gitlawb Opengateway idarəetmə panelindən əldə edin.", @@ -6175,7 +6175,7 @@ "glm-cn": "GLM Coding (China)-i API açarı ilə qoşun.", "glmt": "Daha yüksək token büdcəsi, düşünmə aktivləşdirilmiş və daha uzun vaxt aşımı olan hazır GLM profili.", "getgoapi": "GoAPI-ni API açarı ilə qoşun.", - "groq": "Pulsuz tarif: 30 RPM / 14.4K RPD — kredit kartı tələb olunmur", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "API açarını haiper.ai/haiper-api ünvanından əldə edin", "heroku": "Heroku AI-ı API açarı ilə qoşun.", "hcnsec": "API açarını api.hcnsec.cn ünvanından əldə edin", diff --git a/src/i18n/messages/bg.json b/src/i18n/messages/bg.json index 3944642aff..a46d545e1c 100644 --- a/src/i18n/messages/bg.json +++ b/src/i18n/messages/bg.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "API прокси с отстъпка за над 40 модела, включително GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Вземете своя API ключ на https://freeaiapikey.com/dashboard. Базов URL адрес: https://freeaiapikey.com/v1.", "freemodel-dev": "Вземете $300 безплатни API кредити на https://freemodel.dev — не се изисква информация за плащане. Съвместима с OpenAI крайна точка. Налични са модели GPT-5.4 и GPT-5.5.", "friendliai": "Безплатен план за serverless inference — не се изисква кредитна карта", - "gemini": "Безплатно завинаги: 1,500 заявки/ден за Gemini 2.5 Flash — без кредитна карта, вземете ключ на aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Свържете GigaChat (Sber) с API ключ.", "gitlab": "Личен токен за достъп на GitLab за публичния Code Suggestions API. Конфигурирайте self-hosted базов URL адрес, когато не използвате gitlab.com.", "gitlawb-gmi": "Вземете своя API ключ от таблото за управление на Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "Свържете GLM Coding (China) с API ключ.", "glmt": "Предварително зададен GLM профил с по-висок бюджет за токени, активирано мислене и по-дълъг таймаут.", "getgoapi": "Свържете GoAPI с API ключ.", - "groq": "Безплатен план: 30 RPM / 14.4K RPD — без кредитна карта", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Вземете API ключ на haiper.ai/haiper-api", "heroku": "Свържете Heroku AI с API ключ.", "hcnsec": "Вземете API ключ на api.hcnsec.cn", diff --git a/src/i18n/messages/bn.json b/src/i18n/messages/bn.json index 8f6afadff3..20f15cafa2 100644 --- a/src/i18n/messages/bn.json +++ b/src/i18n/messages/bn.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5 সহ 40+ মডেলের জন্য ডিসকাউন্টেড API প্রক্সি। https://freeaiapikey.com/dashboard থেকে আপনার API কী পান। বেস URL: https://freeaiapikey.com/v1।", "freemodel-dev": "https://freemodel.dev থেকে $300 ফ্রি API ক্রেডিট পান — কোনো পেমেন্ট তথ্যের প্রয়োজন নেই। OpenAI-সামঞ্জস্যপূর্ণ এন্ডপয়েন্ট। GPT-5.4 and GPT-5.5 মডেলগুলো উপলব্ধ।", "friendliai": "সার্ভারলেস ইনফারেন্সের জন্য ফ্রি টিয়ার — কোনো ক্রেডিট কার্ডের প্রয়োজন নেই", - "gemini": "চিরকালের জন্য ফ্রি: Gemini 2.5 Flash-এর জন্য প্রতিদিন 1,500টি রিকোয়েস্ট — কোনো ক্রেডিট কার্ড লাগবে না, aistudio.google.com থেকে কী পান", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "একটি API কী দিয়ে GigaChat (Sber) কানেক্ট করুন।", "gitlab": "পাবলিক Code Suggestions API-এর জন্য GitLab পার্সোনাল অ্যাক্সেস টোকেন। gitlab.com ব্যবহার না করার সময় একটি সেলফ-হোস্টেড বেস URL কনফিগার করুন।", "gitlawb-gmi": "Gitlawb Opengateway ড্যাশবোর্ড থেকে আপনার API কী পান।", @@ -6175,7 +6175,7 @@ "glm-cn": "একটি API কী দিয়ে GLM Coding (China) কানেক্ট করুন।", "glmt": "উচ্চতর টোকেন বাজেট, থিংকিং সক্রিয় এবং দীর্ঘতর টাইমআউট সহ প্রিসেট GLM প্রোফাইল।", "getgoapi": "একটি API কী দিয়ে GoAPI কানেক্ট করুন।", - "groq": "ফ্রি টিয়ার: 30 RPM / 14.4K RPD — কোনো ক্রেডিট কার্ড লাগবে না", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "haiper.ai/haiper-api থেকে API কী পান", "heroku": "একটি API কী দিয়ে Heroku AI কানেক্ট করুন।", "hcnsec": "api.hcnsec.cn-এ API কী পান", diff --git a/src/i18n/messages/cs.json b/src/i18n/messages/cs.json index 29af05425a..427e6d2ef6 100644 --- a/src/i18n/messages/cs.json +++ b/src/i18n/messages/cs.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Zlevněná API proxy pro více než 40 modelů včetně GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Získejte svůj API klíč na https://freeaiapikey.com/dashboard. Základní URL: https://freeaiapikey.com/v1.", "freemodel-dev": "Získejte bezplatný API kredit 300 $ na https://freemodel.dev – nejsou vyžadovány žádné platební údaje. Koncový bod kompatibilní s OpenAI. K dispozici jsou modely GPT-5.4 a GPT-5.5.", "friendliai": "Bezplatná úroveň pro serverless inferenci – není vyžadována platební karta", - "gemini": "Navždy zdarma: 1 500 požadavků/den pro Gemini 2.5 Flash – bez platební karty, klíč získáte na aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Připojte GigaChat (Sber) pomocí API klíče.", "gitlab": "Osobní přístupový token (PAT) GitLab pro veřejné rozhraní API Code Suggestions. Pokud nepoužíváte gitlab.com, nakonfigurujte vlastní základní URL.", "gitlawb-gmi": "Získejte svůj API klíč z nástěnky Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "Připojte GLM Coding (Čína) pomocí API klíče.", "glmt": "Přednastavený profil GLM s vyšším rozpočtem tokenů, povoleným přemýšlením a delším časovým limitem.", "getgoapi": "Připojte GoAPI pomocí API klíče.", - "groq": "Bezplatná úroveň: 30 RPM / 14,4K RPD – bez platební karty", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Získejte API klíč na haiper.ai/haiper-api", "heroku": "Připojte Heroku AI pomocí API klíče.", "hcnsec": "Získejte API klíč na api.hcnsec.cn", diff --git a/src/i18n/messages/da.json b/src/i18n/messages/da.json index bea97fe340..fa3c227a46 100644 --- a/src/i18n/messages/da.json +++ b/src/i18n/messages/da.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Rabatbelagt API-proxy til 40+ modeller inklusive GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Hent din API-nøgle på https://freeaiapikey.com/dashboard. Base-URL: https://freeaiapikey.com/v1.", "freemodel-dev": "Få $300 gratis API-kredit på https://freemodel.dev — ingen betalingsoplysninger påkrævet. OpenAI-kompatibelt slutpunkt. GPT-5.4- og GPT-5.5-modeller tilgængelige.", "friendliai": "Gratis niveau til serverløs inferens — intet kreditkort påkrævet", - "gemini": "Gratis altid: 1.500 anmodninger/dag til Gemini 2.5 Flash — intet kreditkort, hent nøgle på aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Forbind GigaChat (Sber) med en API-nøgle.", "gitlab": "Personligt adgangstoken til GitLab til den offentlige Code Suggestions API. Konfigurer en selvhostet base-URL, når du ikke bruger gitlab.com.", "gitlawb-gmi": "Hent din API-nøgle fra Gitlawb Opengateway-dashboardet.", @@ -6175,7 +6175,7 @@ "glm-cn": "Forbind GLM Coding (Kina) med en API-nøgle.", "glmt": "Forudindstillet GLM-profil med højere token-budget, tænkning aktiveret og længere timeout.", "getgoapi": "Forbind GoAPI med en API-nøgle.", - "groq": "Gratis niveau: 30 RPM / 14,4K RPD — intet kreditkort", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Hent API-nøgle på haiper.ai/haiper-api", "heroku": "Forbind Heroku AI med en API-nøgle.", "hcnsec": "Få API-nøgle på api.hcnsec.cn", diff --git a/src/i18n/messages/de.json b/src/i18n/messages/de.json index d0c0322ac1..16e1370d91 100644 --- a/src/i18n/messages/de.json +++ b/src/i18n/messages/de.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Vergünstigter API-Proxy für über 40 Modelle, darunter GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Holen Sie sich Ihren API-Schlüssel unter https://freeaiapikey.com/dashboard. Basis-URL: https://freeaiapikey.com/v1.", "freemodel-dev": "Holen Sie sich 300 $ kostenloses API-Guthaben unter https://freemodel.dev – keine Zahlungsinformationen erforderlich. OpenAI-kompatibler Endpunkt. GPT-5.4- und GPT-5.5-Modelle verfügbar.", "friendliai": "Kostenlose Stufe für serverlose Inferenz – keine Kreditkarte erforderlich", - "gemini": "Für immer kostenlos: 1.500 Anfragen/Tag für Gemini 2.5 Flash — keine Kreditkarte, Schlüssel unter aistudio.google.com anfordern", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "GigaChat (Sber) mit einem API-Schlüssel verbinden.", "gitlab": "Persönliches GitLab-Zugriffstoken für die öffentliche Code Suggestions API. Konfigurieren Sie eine selbstgehostete Basis-URL, wenn Sie nicht gitlab.com verwenden.", "gitlawb-gmi": "Holen Sie sich Ihren API-Schlüssel aus dem Gitlawb Opengateway-Dashboard.", @@ -6175,7 +6175,7 @@ "glm-cn": "GLM Coding (China) mit einem API-Schlüssel verbinden.", "glmt": "Voreingestelltes GLM-Profil mit höherem Token-Budget, aktiviertem Denken und längerem Timeout.", "getgoapi": "GoAPI mit einem API-Schlüssel verbinden.", - "groq": "Kostenlose Stufe: 30 RPM / 14,4K RPD — keine Kreditkarte", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "API-Schlüssel unter haiper.ai/haiper-api anfordern", "heroku": "Heroku AI mit einem API-Schlüssel verbinden.", "hcnsec": "API-Schlüssel unter api.hcnsec.cn anfordern", diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index 89659dc23c..163a1fe581 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -6169,7 +6169,7 @@ "freeaiapikey": "Discounted API proxy for 40+ models including GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Get your API key at https://freeaiapikey.com/dashboard. Base URL: https://freeaiapikey.com/v1.", "freemodel-dev": "Get $300 free API credits at https://freemodel.dev — no payment info required. OpenAI-compatible endpoint. GPT-5.4 and GPT-5.5 models available.", "friendliai": "Free tier for serverless inference — no credit card required", - "gemini": "Free forever: 1,500 req/day for Gemini 2.5 Flash — no credit card, get key at aistudio.google.com", + "gemini": "Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Connect GigaChat (Sber) with an API key.", "gitlab": "GitLab personal access token for the public Code Suggestions API. Configure a self-hosted base URL when not using gitlab.com.", "gitlawb-gmi": "Get your API key from Gitlawb Opengateway dashboard.", @@ -6178,7 +6178,7 @@ "glm-cn": "Connect GLM Coding (China) with an API key.", "glmt": "Preset GLM profile with higher token budget, thinking enabled, and longer timeout.", "getgoapi": "Connect GoAPI with an API key.", - "groq": "Free tier: 30 RPM / 14.4K RPD — no credit card", + "groq": "Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Get API key at haiper.ai/haiper-api", "heroku": "Connect Heroku AI with an API key.", "hcnsec": "Get API key at api.hcnsec.cn", diff --git a/src/i18n/messages/es.json b/src/i18n/messages/es.json index 09b8296a82..73dd31e3b8 100644 --- a/src/i18n/messages/es.json +++ b/src/i18n/messages/es.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Discounted API proxy for 40+ models including GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Get your API key at https://freeaiapikey.com/dashboard. Base URL: https://freeaiapikey.com/v1.", "freemodel-dev": "Get $300 free API credits at https://freemodel.dev — no payment info required. OpenAI-compatible endpoint. GPT-5.4 and GPT-5.5 models available.", "friendliai": "Free tier for serverless inference — no credit card required", - "gemini": "Free forever: 1,500 req/day for Gemini 2.5 Flash — no credit card, get key at aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Connect GigaChat (Sber) with an API key.", "gitlab": "GitLab personal access token for the public Code Suggestions API. Configure a self-hosted base URL when not using gitlab.com.", "gitlawb-gmi": "Get your API key from Gitlawb Opengateway dashboard.", @@ -6175,7 +6175,7 @@ "glm-cn": "Connect GLM Coding (China) with an API key.", "glmt": "Preset GLM profile with higher token budget, thinking enabled, and longer timeout.", "getgoapi": "Connect GoAPI with an API key.", - "groq": "Free tier: 30 RPM / 14.4K RPD — no credit card", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Get API key at haiper.ai/haiper-api", "heroku": "Connect Heroku AI with an API key.", "hcnsec": "Get API key at api.hcnsec.cn", diff --git a/src/i18n/messages/fa.json b/src/i18n/messages/fa.json index 79c0ec653f..f6507ed49e 100644 --- a/src/i18n/messages/fa.json +++ b/src/i18n/messages/fa.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "پروکسی API تخفیف‌خورده برای بیش از ۴۰ مدل از جمله GPT-5، Claude Opus 4.6، Claude Sonnet 4.6، Qwen 3.5. کلید API خود را در https://freeaiapikey.com/dashboard دریافت کنید. URL پایه: https://freeaiapikey.com/v1.", "freemodel-dev": "۳۰۰ دلار اعتبار رایگان API در https://freemodel.dev دریافت کنید — بدون نیاز به اطلاعات پرداخت. نقطه پایانی سازگار با OpenAI. مدل‌های GPT-5.4 و GPT-5.5 در دسترس هستند.", "friendliai": "سطح رایگان برای استنتاج بدون سرور (serverless inference) — بدون نیاز به کارت اعتباری", - "gemini": "رایگان برای همیشه: ۱,۵۰۰ درخواست در روز برای Gemini 2.5 Flash — بدون نیاز به کارت اعتباری، کلید را در aistudio.google.com دریافت کنید", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "اتصال GigaChat (Sber) با یک کلید API.", "gitlab": "توکن دسترسی شخصی GitLab برای API عمومی Code Suggestions. در صورت عدم استفاده از gitlab.com، یک URL پایه خودمیزبانی‌شده (self-hosted) پیکربندی کنید.", "gitlawb-gmi": "کلید API خود را از داشبورد Gitlawb Opengateway دریافت کنید.", @@ -6175,7 +6175,7 @@ "glm-cn": "اتصال GLM Coding (China) با یک کلید API.", "glmt": "پروفایل پیش‌فرض GLM با بودجه توکن بالاتر، فعال بودن تفکر (thinking) و زمان انتظار (timeout) طولانی‌تر.", "getgoapi": "اتصال GoAPI با یک کلید API.", - "groq": "سطح رایگان: ۳۰ RPM / ۱۴.۴K RPD — بدون نیاز به کارت اعتباری", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "کلید API را در haiper.ai/haiper-api دریافت کنید", "heroku": "اتصال Heroku AI با یک کلید API.", "hcnsec": "دریافت کلید API در api.hcnsec.cn", diff --git a/src/i18n/messages/fi.json b/src/i18n/messages/fi.json index c83a56cb2c..253e58687f 100644 --- a/src/i18n/messages/fi.json +++ b/src/i18n/messages/fi.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Alennettu API-välityspalvelin yli 40 mallille, mukaan lukien GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Hanki API-avaimesi osoitteesta https://freeaiapikey.com/dashboard. Perus-URL: https://freeaiapikey.com/v1.", "freemodel-dev": "Hanki 300 $ ilmaista API-saldoa osoitteesta https://freemodel.dev — maksutietoja ei vaadita. OpenAI-yhteensopiva päätepiste. GPT-5.4- ja GPT-5.5-mallit saatavilla.", "friendliai": "Ilmainen taso palvelimettomaan päättelyyn — luottokorttia ei vaadita", - "gemini": "Aina ilmainen: 1 500 pyyntöä/päivä Gemini 2.5 Flashille — ei luottokorttia, hanki avain osoitteesta aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Yhdistä GigaChat (Sber) API-avaimella.", "gitlab": "GitLabin henkilökohtainen käyttöoikeustunniste julkiselle Code Suggestions API:lle. Määritä itse isännöity perus-URL, kun et käytä gitlab.com-palvelua.", "gitlawb-gmi": "Hanki API-avaimesi Gitlawb Opengateway -hallintapaneelista.", @@ -6175,7 +6175,7 @@ "glm-cn": "Yhdistä GLM Coding (China) API-avaimella.", "glmt": "Esiasetettu GLM-profiili suuremmalla token-budjetilla, ajattelu käytössä ja pidemmällä aikakatkaisulla.", "getgoapi": "Yhdistä GoAPI API-avaimella.", - "groq": "Ilmainen taso: 30 RPM / 14,4K RPD — ei luottokorttia", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Hanki API-avain osoitteesta haiper.ai/haiper-api", "heroku": "Yhdistä Heroku AI API-avaimella.", "hcnsec": "Hanki API-avain osoitteesta api.hcnsec.cn", diff --git a/src/i18n/messages/fr.json b/src/i18n/messages/fr.json index 1bb74a6c87..9870e03616 100644 --- a/src/i18n/messages/fr.json +++ b/src/i18n/messages/fr.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Proxy API à tarif réduit pour plus de 40 modèles, dont GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Obtenez votre clé API sur https://freeaiapikey.com/dashboard. URL de base : https://freeaiapikey.com/v1.", "freemodel-dev": "Obtenez 300 $ de crédits API gratuits sur https://freemodel.dev — aucune information de paiement requise. Point de terminaison compatible avec OpenAI. Modèles GPT-5.4 et GPT-5.5 disponibles.", "friendliai": "Offre gratuite pour l'inférence serverless — aucune carte de crédit requise", - "gemini": "Gratuit à vie : 1 500 req/jour pour Gemini 2.5 Flash — sans carte de crédit, obtenez la clé sur aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Connectez GigaChat (Sber) avec une clé API.", "gitlab": "Jeton d'accès personnel GitLab pour l'API publique Code Suggestions. Configurez une URL de base auto-hébergée si vous n'utilisez pas gitlab.com.", "gitlawb-gmi": "Obtenez votre clé API depuis le tableau de bord Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "Connectez GLM Coding (Chine) avec une clé API.", "glmt": "Profil GLM prédéfini avec un budget de tokens plus élevé, mode pensée activé et délai d'attente plus long.", "getgoapi": "Connectez GoAPI avec une clé API.", - "groq": "Offre gratuite : 30 RPM / 14,4K RPD — sans carte de crédit", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Obtenez une clé API sur haiper.ai/haiper-api", "heroku": "Connectez Heroku AI avec une clé API.", "hcnsec": "Obtenir une clé API sur api.hcnsec.cn", diff --git a/src/i18n/messages/gu.json b/src/i18n/messages/gu.json index 7fbc8ac29b..e6e20f58c2 100644 --- a/src/i18n/messages/gu.json +++ b/src/i18n/messages/gu.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5 સહિત 40+ મોડલ્સ માટે ડિસ્કાઉન્ટેડ API પ્રોક્સી. https://freeaiapikey.com/dashboard પર તમારી API કી મેળવો. બેઝ URL: https://freeaiapikey.com/v1.", "freemodel-dev": "https://freemodel.dev પર $300 મફત API ક્રેડિટ્સ મેળવો — કોઈ ચુકવણી માહિતીની જરૂર નથી. OpenAI-સુસંગત એન્ડપોઇન્ટ. GPT-5.4 અને GPT-5.5 મોડલ્સ ઉપલબ્ધ છે.", "friendliai": "સર્વરલેસ ઇન્ફરન્સ માટે મફત સ્તર — કોઈ ક્રેડિટ કાર્ડની જરૂર નથી", - "gemini": "હંમેશા માટે મફત: Gemini 2.5 Flash માટે દરરોજ 1,500 વિનંતીઓ — કોઈ ક્રેડિટ કાર્ડ નહીં, aistudio.google.com પર કી મેળવો", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "API કી વડે GigaChat (Sber) ને કનેક્ટ કરો.", "gitlab": "સાર્વજનિક Code Suggestions API માટે GitLab વ્યક્તિગત ઍક્સેસ ટોકન. જ્યારે gitlab.com નો ઉપયોગ ન કરી રહ્યા હોવ ત્યારે સેલ્ફ-હોસ્ટેડ બેઝ URL કન્ફિગર કરો.", "gitlawb-gmi": "Gitlawb Opengateway ડેશબોર્ડ પરથી તમારી API કી મેળવો.", @@ -6175,7 +6175,7 @@ "glm-cn": "API કી વડે GLM Coding (China) ને કનેક્ટ કરો.", "glmt": "ઉચ્ચ ટોકન બજેટ, વિચારવાની ક્ષમતા સક્ષમ અને લાંબા સમયસમાપ્તિ સાથે પ્રીસેટ GLM પ્રોફાઇલ.", "getgoapi": "API કી વડે GoAPI ને કનેક્ટ કરો.", - "groq": "મફત સ્તર: 30 RPM / 14.4K RPD — કોઈ ક્રેડિટ કાર્ડ નહીં", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "haiper.ai/haiper-api પર API કી મેળવો", "heroku": "API કી વડે Heroku AI ને કનેક્ટ કરો.", "hcnsec": "api.hcnsec.cn પર API કી મેળવો", diff --git a/src/i18n/messages/he.json b/src/i18n/messages/he.json index 66bea20bb8..654be87cc9 100644 --- a/src/i18n/messages/he.json +++ b/src/i18n/messages/he.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "פרוקסי API מוזל עבור יותר מ-40 מודלים כולל GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. קבל את מפתח ה-API שלך ב-https://freeaiapikey.com/dashboard. כתובת URL בסיסית: https://freeaiapikey.com/v1.", "freemodel-dev": "קבל קרדיט API חינם בסך $300 ב-https://freemodel.dev — ללא צורך בפרטי תשלום. נקודת קצה תואמת OpenAI. מודלים של GPT-5.4 ו-GPT-5.5 זמינים.", "friendliai": "מסלול חינמי להסקה ללא שרת — ללא צורך בכרטיס אשראי", - "gemini": "חינם לתמיד: 1,500 בקשות/יום עבור Gemini 2.5 Flash — ללא כרטיס אשראי, קבל מפתח ב-aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "חבר את GigaChat (Sber) באמצעות מפתח API.", "gitlab": "טוקן גישה אישי של GitLab עבור ה-API הציבורי של Code Suggestions. הגדר כתובת URL בסיסית באירוח עצמי כאשר אינך משתמש ב-gitlab.com.", "gitlawb-gmi": "קבל את מפתח ה-API שלך מלוח הבקרה של Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "חבר את GLM Coding (China) באמצעות מפתח API.", "glmt": "פרופיל GLM מוגדר מראש עם תקציב טוקנים גבוה יותר, חשיבה מופעלת ופסק זמן ארוך יותר.", "getgoapi": "חבר את GoAPI באמצעות מפתח API.", - "groq": "מסלול חינמי: 30 RPM / 14.4K RPD — ללא כרטיס אשראי", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "קבל מפתח API ב-haiper.ai/haiper-api", "heroku": "חבר את Heroku AI באמצעות מפתח API.", "hcnsec": "קבל מפתח API ב-api.hcnsec.cn", diff --git a/src/i18n/messages/hi.json b/src/i18n/messages/hi.json index f45c0fad90..f02e4c2e5c 100644 --- a/src/i18n/messages/hi.json +++ b/src/i18n/messages/hi.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5 सहित 40+ मॉडलों के लिए रियायती API प्रॉक्सी। https://freeaiapikey.com/dashboard पर अपनी API कुंजी प्राप्त करें। बेस URL: https://freeaiapikey.com/v1।", "freemodel-dev": "https://freemodel.dev पर $300 का निःशुल्क API क्रेडिट प्राप्त करें — किसी भुगतान जानकारी की आवश्यकता नहीं है। OpenAI-संगत एंडपॉइंट। GPT-5.4 और GPT-5.5 मॉडल उपलब्ध हैं।", "friendliai": "सर्वरलेस इनफेरेंस के लिए निःशुल्क टियर — किसी क्रेडिट कार्ड की आवश्यकता नहीं है", - "gemini": "हमेशा के लिए निःशुल्क: Gemini 2.5 Flash के लिए 1,500 req/day — कोई क्रेडिट कार्ड नहीं, aistudio.google.com पर कुंजी प्राप्त करें", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "GigaChat (Sber) को एक API कुंजी से कनेक्ट करें।", "gitlab": "सार्वजनिक Code Suggestions API के लिए GitLab व्यक्तिगत एक्सेस टोकन। gitlab.com का उपयोग न करते समय एक स्व-होस्टेड बेस URL कॉन्फ़िगर करें।", "gitlawb-gmi": "Gitlawb Opengateway डैशबोर्ड से अपनी API कुंजी प्राप्त करें।", @@ -6175,7 +6175,7 @@ "glm-cn": "GLM Coding (China) को एक API कुंजी से कनेक्ट करें।", "glmt": "उच्च टोकन बजट, थिंकिंग (thinking) सक्षम और लंबे टाइमआउट के साथ प्रीसेट GLM प्रोफ़ाइल।", "getgoapi": "GoAPI को एक API कुंजी से कनेक्ट करें।", - "groq": "निःशुल्क टियर: 30 RPM / 14.4K RPD — कोई क्रेडिट कार्ड नहीं", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "haiper.ai/haiper-api पर API कुंजी प्राप्त करें", "heroku": "Heroku AI को एक API कुंजी से कनेक्ट करें।", "hcnsec": "api.hcnsec.cn पर API कुंजी प्राप्त करें", diff --git a/src/i18n/messages/hu.json b/src/i18n/messages/hu.json index b1275355b9..7f30948405 100644 --- a/src/i18n/messages/hu.json +++ b/src/i18n/messages/hu.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Kedvezményes API-proxy több mint 40 modellhez, beleértve a GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5 modelleket. Szerezze be API-kulcsát a https://freeaiapikey.com/dashboard oldalon. Alap URL: https://freeaiapikey.com/v1.", "freemodel-dev": "Szerezzen $300 ingyenes API-kreditet a https://freemodel.dev oldalon — fizetési adat nem szükséges. OpenAI-kompatibilis végpont. GPT-5.4 és GPT-5.5 modellek érhetők el.", "friendliai": "Ingyenes csomag szerver nélküli következtetéshez — bankkártya nem szükséges", - "gemini": "Örökké ingyenes: 1500 kérés/nap a Gemini 2.5 Flash-hez — bankkártya nem szükséges, kulcs beszerzése: aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Csatlakoztassa a GigaChatet (Sber) egy API-kulccsal.", "gitlab": "GitLab személyes hozzáférési token a nyilvános Code Suggestions API-hoz. Állítson be saját üzemeltetésű alap URL-t, ha nem a gitlab.com-ot használja.", "gitlawb-gmi": "Szerezze be API-kulcsát a Gitlawb Opengateway irányítópultjáról.", @@ -6175,7 +6175,7 @@ "glm-cn": "Csatlakoztassa a GLM Coding (Kína) szolgáltatást egy API-kulccsal.", "glmt": "Előre beállított GLM-profil magasabb tokenkerettel, engedélyezett gondolkodással és hosszabb időtúllépéssel.", "getgoapi": "Csatlakoztassa a GoAPI-t egy API-kulccsal.", - "groq": "Ingyenes csomag: 30 RPM / 14,4K RPD — bankkártya nem szükséges", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Szerezzen API-kulcsot a haiper.ai/haiper-api oldalon", "heroku": "Csatlakoztassa a Heroku AI-t egy API-kulccsal.", "hcnsec": "Szerezzen API-kulcsot itt: api.hcnsec.cn", diff --git a/src/i18n/messages/id.json b/src/i18n/messages/id.json index 474a26973d..d92469fca9 100644 --- a/src/i18n/messages/id.json +++ b/src/i18n/messages/id.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Proksi API berdiskon untuk 40+ model termasuk GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Dapatkan kunci API Anda di https://freeaiapikey.com/dashboard. URL Dasar: https://freeaiapikey.com/v1.", "freemodel-dev": "Dapatkan kredit API gratis senilai $300 di https://freemodel.dev — tidak memerlukan informasi pembayaran. Endpoint yang kompatibel dengan OpenAI. Model GPT-5.4 dan GPT-5.5 tersedia.", "friendliai": "Tingkat gratis untuk inferensi serverless — tidak memerlukan kartu kredit", - "gemini": "Gratis selamanya: 1.500 req/hari untuk Gemini 2.5 Flash — tanpa kartu kredit, dapatkan kunci di aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Hubungkan GigaChat (Sber) dengan kunci API.", "gitlab": "Token akses pribadi GitLab untuk API Code Suggestions publik. Konfigurasikan URL dasar yang di-host sendiri saat tidak menggunakan gitlab.com.", "gitlawb-gmi": "Dapatkan kunci API Anda dari dasbor Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "Hubungkan GLM Coding (China) dengan kunci API.", "glmt": "Profil GLM prasetel dengan anggaran token yang lebih tinggi, proses berpikir diaktifkan, dan batas waktu yang lebih lama.", "getgoapi": "Hubungkan GoAPI dengan kunci API.", - "groq": "Tingkat gratis: 30 RPM / 14,4K RPD — tanpa kartu kredit", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Dapatkan kunci API di haiper.ai/haiper-api", "heroku": "Hubungkan Heroku AI dengan kunci API.", "hcnsec": "Dapatkan kunci API di api.hcnsec.cn", diff --git a/src/i18n/messages/it.json b/src/i18n/messages/it.json index 86e1226241..0b1d244f02 100644 --- a/src/i18n/messages/it.json +++ b/src/i18n/messages/it.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Proxy API scontato per oltre 40 modelli tra cui GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Ottieni la tua chiave API su https://freeaiapikey.com/dashboard. URL di base: https://freeaiapikey.com/v1.", "freemodel-dev": "Ottieni 300 $ di crediti API gratuiti su https://freemodel.dev — nessuna informazione di pagamento richiesta. Endpoint compatibile con OpenAI. Modelli GPT-5.4 e GPT-5.5 disponibili.", "friendliai": "Piano gratuito per inferenza serverless — nessuna carta di credito richiesta", - "gemini": "Gratis per sempre: 1.500 richieste/giorno per Gemini 2.5 Flash — nessuna carta di credito, ottieni la chiave su aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Connetti GigaChat (Sber) con una chiave API.", "gitlab": "Token di accesso personale GitLab per l'API pubblica Code Suggestions. Configura un URL di base self-hosted quando non utilizzi gitlab.com.", "gitlawb-gmi": "Ottieni la tua chiave API dalla dashboard di Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "Connetti GLM Coding (Cina) con una chiave API.", "glmt": "Profilo GLM preimpostato con budget di token più elevato, pensiero abilitato e timeout più lungo.", "getgoapi": "Connetti GoAPI con una chiave API.", - "groq": "Piano gratuito: 30 RPM / 14,4K RPD — nessuna carta di credito", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Ottieni la chiave API su haiper.ai/haiper-api", "heroku": "Connetti Heroku AI con una chiave API.", "hcnsec": "Ottieni la chiave API su api.hcnsec.cn", diff --git a/src/i18n/messages/ja.json b/src/i18n/messages/ja.json index ab2f52c289..61f4d19d6e 100644 --- a/src/i18n/messages/ja.json +++ b/src/i18n/messages/ja.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "GPT-5、Claude Opus 4.6、Claude Sonnet 4.6、Qwen 3.5を含む40以上のモデルに対応した割引APIプロキシ。https://freeaiapikey.com/dashboard でAPIキーを取得してください。ベースURL: https://freeaiapikey.com/v1。", "freemodel-dev": "https://freemodel.dev で$300の無料APIクレジットを取得 — 支払い情報は不要です。OpenAI互換エンドポイント。GPT-5.4およびGPT-5.5モデルが利用可能です。", "friendliai": "サーバーレス推論の無料枠 — クレジットカード不要", - "gemini": "永久無料: Gemini 2.5 Flashが1,500 req/日 — クレジットカード不要、aistudio.google.com でキーを取得", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "APIキーでGigaChat (Sber)に接続します。", "gitlab": "パブリックCode Suggestions API用のGitLabパーソナルアクセストークン。gitlab.comを使用しない場合は、セルフホストのベースURLを設定してください。", "gitlawb-gmi": "Gitlawb OpengatewayダッシュボードからAPIキーを取得します。", @@ -6175,7 +6175,7 @@ "glm-cn": "APIキーでGLM Coding (China)に接続します。", "glmt": "より大きなトークンバジェット、思考の有効化、およびより長いタイムアウトを備えたプリセットGLMプロファイル。", "getgoapi": "APIキーでGoAPIに接続します。", - "groq": "無料枠: 30 RPM / 14.4K RPD — クレジットカード不要", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "haiper.ai/haiper-api でAPIキーを取得", "heroku": "APIキーでHeroku AIに接続します。", "hcnsec": "api.hcnsec.cn でAPIキーを取得", diff --git a/src/i18n/messages/ko.json b/src/i18n/messages/ko.json index ad26677619..c4b71001df 100644 --- a/src/i18n/messages/ko.json +++ b/src/i18n/messages/ko.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5를 포함한 40개 이상의 모델을 위한 할인된 API 프록시. https://freeaiapikey.com/dashboard 에서 API 키를 가져오세요. 베이스 URL: https://freeaiapikey.com/v1.", "freemodel-dev": "https://freemodel.dev 에서 $300 무료 API 크레딧을 받으세요 — 결제 정보 불필요. OpenAI 호환 엔드포인트. GPT-5.4 및 GPT-5.5 모델 사용 가능.", "friendliai": "서버리스 추론을 위한 무료 티어 — 신용카드 불필요", - "gemini": "평생 무료: Gemini 2.5 Flash 기준 1,500회 요청/일 — 신용카드 불필요, aistudio.google.com 에서 키 발급", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "API 키로 GigaChat (Sber) 연결.", "gitlab": "공개 Code Suggestions API용 GitLab 개인 액세스 토큰. gitlab.com을 사용하지 않는 경우 자체 호스팅 베이스 URL을 구성하세요.", "gitlawb-gmi": "Gitlawb Opengateway 대시보드에서 API 키를 가져오세요.", @@ -6175,7 +6175,7 @@ "glm-cn": "API 키로 GLM Coding (China) 연결.", "glmt": "더 높은 토큰 예산, 생각하기(thinking) 활성화 및 더 긴 타임아웃이 설정된 프리셋 GLM 프로필.", "getgoapi": "API 키로 GoAPI 연결.", - "groq": "무료 티어: 30 RPM / 14.4K RPD — 신용카드 불필요", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "haiper.ai/haiper-api 에서 API 키를 가져오세요.", "heroku": "API 키로 Heroku AI 연결.", "hcnsec": "api.hcnsec.cn 에서 API 키를 가져오세요.", diff --git a/src/i18n/messages/mr.json b/src/i18n/messages/mr.json index 7f5ee425c9..eabefd5464 100644 --- a/src/i18n/messages/mr.json +++ b/src/i18n/messages/mr.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5 सह 40+ मॉडेल्ससाठी सवलतीचा API प्रॉक्सी. https://freeaiapikey.com/dashboard वर तुमची API की मिळवा. बेस URL: https://freeaiapikey.com/v1.", "freemodel-dev": "https://freemodel.dev वर $300 विनामूल्य API क्रेडिट्स मिळवा — पेमेंट माहितीची आवश्यकता नाही. OpenAI-सुसंगत एंडपॉइंट. GPT-5.4 आणि GPT-5.5 मॉडेल्स उपलब्ध आहेत.", "friendliai": "सर्व्हरलेस इन्फरन्ससाठी विनामूल्य टियर — क्रेडिट कार्डची आवश्यकता नाही", - "gemini": "कायमचे विनामूल्य: Gemini 2.5 Flash साठी 1,500 विनंत्या/दिवस — क्रेडिट कार्ड नाही, aistudio.google.com वर की मिळवा", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "API की सह GigaChat (Sber) कनेक्ट करा.", "gitlab": "सार्वजनिक Code Suggestions API साठी GitLab वैयक्तिक ॲक्सेस टोकन. gitlab.com न वापरताना सेल्फ-होस्टेड बेस URL कॉन्फिगर करा.", "gitlawb-gmi": "Gitlawb Opengateway डॅशबोर्डवरून तुमची API की मिळवा.", @@ -6175,7 +6175,7 @@ "glm-cn": "API की सह GLM Coding (China) कनेक्ट करा.", "glmt": "उच्च टोकन बजेट, थिंकिंग सक्षम आणि दीर्घ टाइमआउटसह प्रीसेट GLM प्रोफाइल.", "getgoapi": "API की सह GoAPI कनेक्ट करा.", - "groq": "विनामूल्य टियर: 30 RPM / 14.4K RPD — क्रेडिट कार्ड नाही", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "haiper.ai/haiper-api वर API की मिळवा", "heroku": "API की सह Heroku AI कनेक्ट करा.", "hcnsec": "api.hcnsec.cn वर API की मिळवा", diff --git a/src/i18n/messages/ms.json b/src/i18n/messages/ms.json index a780d09bcb..98a8e04778 100644 --- a/src/i18n/messages/ms.json +++ b/src/i18n/messages/ms.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Proksi API berdiskaun untuk 40+ model termasuk GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Dapatkan kunci API anda di https://freeaiapikey.com/dashboard. URL asas: https://freeaiapikey.com/v1.", "freemodel-dev": "Dapatkan kredit API percuma $300 di https://freemodel.dev — tiada maklumat pembayaran diperlukan. Titik akhir serasi OpenAI. Model GPT-5.4 dan GPT-5.5 tersedia.", "friendliai": "Peringkat percuma untuk inferens tanpa pelayan — tiada kad kredit diperlukan", - "gemini": "Percuma selamanya: 1,500 req/hari untuk Gemini 2.5 Flash — tiada kad kredit, dapatkan kunci di aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Sambungkan GigaChat (Sber) dengan kunci API.", "gitlab": "Token akses peribadi GitLab untuk API Code Suggestions awam. Konfigurasikan URL asas dihoskan sendiri apabila tidak menggunakan gitlab.com.", "gitlawb-gmi": "Dapatkan kunci API anda daripada papan pemuka Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "Sambungkan GLM Coding (China) dengan kunci API.", "glmt": "Profil GLM pratetap dengan belanjawan token yang lebih tinggi, pemikiran didayakan dan tamat masa yang lebih lama.", "getgoapi": "Sambungkan GoAPI dengan kunci API.", - "groq": "Peringkat percuma: 30 RPM / 14.4K RPD — tiada kad kredit", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Dapatkan kunci API di haiper.ai/haiper-api", "heroku": "Sambungkan Heroku AI dengan kunci API.", "hcnsec": "Dapatkan kunci API di api.hcnsec.cn", diff --git a/src/i18n/messages/nl.json b/src/i18n/messages/nl.json index 706a5ccd35..dc251e05db 100644 --- a/src/i18n/messages/nl.json +++ b/src/i18n/messages/nl.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "API-proxy met korting voor meer dan 40 modellen, waaronder GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Haal je API-sleutel op via https://freeaiapikey.com/dashboard. Basis-URL: https://freeaiapikey.com/v1.", "freemodel-dev": "Ontvang $300 gratis API-tegoed op https://freemodel.dev — geen betalingsgegevens vereist. OpenAI-compatibel eindpunt. GPT-5.4- en GPT-5.5-modellen beschikbaar.", "friendliai": "Gratis abonnement voor serverloze inferentie — geen creditcard vereist", - "gemini": "Altijd gratis: 1.500 verzoeken/dag voor Gemini 2.5 Flash — geen creditcard, haal de sleutel op via aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Verbind GigaChat (Sber) met een API-sleutel.", "gitlab": "Persoonlijk toegangstoken van GitLab voor de openbare Code Suggestions-API. Configureer een zelf-gehoste basis-URL als je gitlab.com niet gebruikt.", "gitlawb-gmi": "Haal je API-sleutel op uit het Gitlawb Opengateway-dashboard.", @@ -6175,7 +6175,7 @@ "glm-cn": "Verbind GLM Coding (China) met een API-sleutel.", "glmt": "Vooraf ingesteld GLM-profiel met een hoger tokenbudget, denken ingeschakeld en een langere time-out.", "getgoapi": "Verbind GoAPI met een API-sleutel.", - "groq": "Gratis abonnement: 30 RPM / 14,4K RPD — geen creditcard", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Haal de API-sleutel op via haiper.ai/haiper-api", "heroku": "Verbind Heroku AI met een API-sleutel.", "hcnsec": "Haal de API-sleutel op via api.hcnsec.cn", diff --git a/src/i18n/messages/no.json b/src/i18n/messages/no.json index 9576e4b6a6..0051f48f05 100644 --- a/src/i18n/messages/no.json +++ b/src/i18n/messages/no.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Rabattert API-proxy for 40+ modeller inkludert GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Hent API-nøkkelen din på https://freeaiapikey.com/dashboard. Base-URL: https://freeaiapikey.com/v1.", "freemodel-dev": "Få $300 i gratis API-kreditter på https://freemodel.dev — ingen betalingsinformasjon påkrevd. OpenAI-kompatibelt endepunkt. GPT-5.4- og GPT-5.5-modeller tilgjengelig.", "friendliai": "Gratisnivå for serverløs inferens — ingen kredittkort påkrevd", - "gemini": "Gratis for alltid: 1,500 req/dag for Gemini 2.5 Flash — ingen kredittkort, hent nøkkel på aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Koble til GigaChat (Sber) med en API-nøkkel.", "gitlab": "GitLab personlig adgangstoken for det offentlige Code Suggestions-API-et. Konfigurer en selvvertet base-URL når du ikke bruker gitlab.com.", "gitlawb-gmi": "Hent API-nøkkelen din fra Gitlawb Opengateway-dashbordet.", @@ -6175,7 +6175,7 @@ "glm-cn": "Koble til GLM Coding (Kina) med en API-nøkkel.", "glmt": "Forhåndsinnstilt GLM-profil med høyere token-budsjett, tenkning aktivert og lengre tidsavbrudd.", "getgoapi": "Koble til GoAPI med en API-nøkkel.", - "groq": "Gratisnivå: 30 RPM / 14,4K RPD — uten kredittkort", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Hent API-nøkkel på haiper.ai/haiper-api", "heroku": "Koble til Heroku AI med en API-nøkkel.", "hcnsec": "Hent API-nøkkel på api.hcnsec.cn", diff --git a/src/i18n/messages/phi.json b/src/i18n/messages/phi.json index 856b906cfb..6263e50e01 100644 --- a/src/i18n/messages/phi.json +++ b/src/i18n/messages/phi.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "May diskwentong API proxy para sa 40+ na modelo kabilang ang GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Kumuha ng iyong API key sa https://freeaiapikey.com/dashboard. Base URL: https://freeaiapikey.com/v1.", "freemodel-dev": "Kumuha ng $300 libreng API credits sa https://freemodel.dev — walang kinakailangang impormasyon sa pagbabayad. OpenAI-compatible na endpoint. Available ang mga modelong GPT-5.4 at GPT-5.5.", "friendliai": "Libreng tier para sa serverless inference — walang kinakailangang credit card", - "gemini": "Libre magpakailanman: 1,500 req/araw para sa Gemini 2.5 Flash — walang credit card, kumuha ng key sa aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Ikonekta ang GigaChat (Sber) gamit ang isang API key.", "gitlab": "GitLab personal access token para sa pampublikong Code Suggestions API. Mag-configure ng self-hosted na base URL kapag hindi gumagamit ng gitlab.com.", "gitlawb-gmi": "Kumuha ng iyong API key mula sa Gitlawb Opengateway dashboard.", @@ -6175,7 +6175,7 @@ "glm-cn": "Ikonekta ang GLM Coding (China) gamit ang isang API key.", "glmt": "Preset na GLM profile na may mas mataas na token budget, naka-enable ang thinking, at mas mahabang timeout.", "getgoapi": "Ikonekta ang GoAPI gamit ang isang API key.", - "groq": "Libreng tier: 30 RPM / 14.4K RPD — walang credit card", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Kumuha ng API key sa haiper.ai/haiper-api", "heroku": "Ikonekta ang Heroku AI gamit ang isang API key.", "hcnsec": "Kumuha ng API key sa api.hcnsec.cn", diff --git a/src/i18n/messages/pl.json b/src/i18n/messages/pl.json index b9ecc290db..29fc91d100 100644 --- a/src/i18n/messages/pl.json +++ b/src/i18n/messages/pl.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Zrabatowane proxy API dla ponad 40 modeli, w tym GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Pobierz swój klucz API na stronie https://freeaiapikey.com/dashboard. Bazowy adres URL: https://freeaiapikey.com/v1.", "freemodel-dev": "Odbierz $300 darmowych środków API na stronie https://freemodel.dev — dane płatnicze nie są wymagane. Punkt końcowy zgodny z OpenAI. Dostępne modele GPT-5.4 i GPT-5.5.", "friendliai": "Bezpłatny pakiet do wnioskowania bezserwerowego — karta kredytowa nie jest wymagana", - "gemini": "Darmowe na zawsze: 1500 zapytań/dzień dla Gemini 2.5 Flash — bez karty kredytowej, pobierz klucz na aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Połącz z GigaChat (Sber) za pomocą klucza API.", "gitlab": "Osobisty token dostępu GitLab dla publicznego API Code Suggestions. Skonfiguruj bazowy adres URL dla instancji self-hosted, jeśli nie korzystasz z gitlab.com.", "gitlawb-gmi": "Pobierz swój klucz API z panelu Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "Połącz z GLM Coding (China) za pomocą klucza API.", "glmt": "Wstępnie zdefiniowany profil GLM z większym budżetem tokenów, włączonym myśleniem i dłuższym limitem czasu.", "getgoapi": "Połącz z GoAPI za pomocą klucza API.", - "groq": "Darmowy plan: 30 RPM / 14.4K RPD — bez karty kredytowej", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Pobierz klucz API na haiper.ai/haiper-api", "heroku": "Połącz z Heroku AI za pomocą klucza API.", "hcnsec": "Pobierz klucz API na api.hcnsec.cn", diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index e65916a54d..15b6f6df86 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -6170,7 +6170,7 @@ "freeaiapikey": "Proxy de API com desconto para mais de 40 modelos, incluindo GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Obtenha sua chave de API em https://freeaiapikey.com/dashboard. URL base: https://freeaiapikey.com/v1.", "freemodel-dev": "Obtenha $300 em créditos de API gratuitos em https://freemodel.dev — sem necessidade de dados de pagamento. Endpoint compatível com OpenAI. Modelos GPT-5.4 e GPT-5.5 disponíveis.", "friendliai": "Nível gratuito para inferência serverless — sem necessidade de cartão de crédito", - "gemini": "Gratuito para sempre: 1.500 solicitações/dia para o Gemini 2.5 Flash — sem cartão de crédito, obtenha a chave em aistudio.google.com", + "gemini": "Plano gratuito pelo Google AI Studio; as quotas por modelo não são mais publicadas (veja a página de quota no AI Studio) — sem cartão de crédito.", "gigachat": "Conecte o GigaChat (Sber) com uma chave de API.", "gitlab": "Token de acesso pessoal do GitLab para a API pública de Code Suggestions. Configure uma URL base self-hosted quando não estiver usando gitlab.com.", "gitlawb-gmi": "Obtenha sua chave de API no dashboard do Gitlawb Opengateway.", @@ -6179,7 +6179,7 @@ "glm-cn": "Conecte o GLM Coding (China) com uma chave de API.", "glmt": "Perfil GLM pré-configurado com orçamento de tokens maior, thinking ativado e timeout mais longo.", "getgoapi": "Conecte o GoAPI com uma chave de API.", - "groq": "Nível gratuito: 30 RPM / 14,4K RPD — sem cartão de crédito", + "groq": "Plano gratuito: limites por modelo (200K tokens/dia por modelo de chat; veja console.groq.com/docs/rate-limits para RPM/RPD) — sem meio de pagamento cadastrado.", "haiper": "Obtenha a chave de API em haiper.ai/haiper-api", "heroku": "Conecte o Heroku AI com uma chave de API.", "hcnsec": "Obtenha a chave de API em api.hcnsec.cn", diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index 8e518e2e0e..3ecb98b353 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Proxy de API com desconto para mais de 40 modelos, incluindo GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Obtenha a sua chave de API em https://freeaiapikey.com/dashboard. URL base: https://freeaiapikey.com/v1.", "freemodel-dev": "Obtenha $300 em créditos de API gratuitos em https://freemodel.dev — sem necessidade de dados de pagamento. Endpoint compatível com OpenAI. Modelos GPT-5.4 e GPT-5.5 disponíveis.", "friendliai": "Nível gratuito para inferência serverless — sem necessidade de cartão de crédito", - "gemini": "Gratuito para sempre: 1.500 req/dia para o Gemini 2.5 Flash — sem cartão de crédito, obtenha a chave em aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Ligue o GigaChat (Sber) com uma chave de API.", "gitlab": "Token de acesso pessoal do GitLab para a API pública Code Suggestions. Configure um URL base autoalojado quando não estiver a utilizar o gitlab.com.", "gitlawb-gmi": "Obtenha a sua chave de API no painel do Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "Ligue o GLM Coding (China) com uma chave de API.", "glmt": "Perfil predefinido do GLM com maior orçamento de tokens, raciocínio ativado e tempo limite mais longo.", "getgoapi": "Ligue a GoAPI com uma chave de API.", - "groq": "Nível gratuito: 30 RPM / 14,4K RPD — sem cartão de crédito", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Obtenha a chave de API em haiper.ai/haiper-api", "heroku": "Ligue o Heroku AI com uma chave de API.", "hcnsec": "Obtenha a chave de API em api.hcnsec.cn", diff --git a/src/i18n/messages/ro.json b/src/i18n/messages/ro.json index 0bfdea697e..124b1a610d 100644 --- a/src/i18n/messages/ro.json +++ b/src/i18n/messages/ro.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Proxy API cu reducere pentru peste 40 de modele, inclusiv GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Obțineți cheia API la https://freeaiapikey.com/dashboard. URL de bază: https://freeaiapikey.com/v1.", "freemodel-dev": "Obțineți $300 credite API gratuite la https://freemodel.dev — nu sunt necesare informații de plată. Endpoint compatibil cu OpenAI. Modele GPT-5.4 și GPT-5.5 disponibile.", "friendliai": "Nivel gratuit pentru inferență serverless — nu este necesar card de credit", - "gemini": "Gratuit pentru totdeauna: 1.500 req/zi pentru Gemini 2.5 Flash — fără card de credit, obțineți cheia la aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Conectați GigaChat (Sber) cu o cheie API.", "gitlab": "Token de acces personal GitLab pentru API-ul public Code Suggestions. Configurați un URL de bază self-hosted când nu utilizați gitlab.com.", "gitlawb-gmi": "Obțineți cheia API din tabloul de bord Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "Conectați GLM Coding (China) cu o cheie API.", "glmt": "Profil GLM prestabilit cu un buget de tokenuri mai mare, gândire activată și timeout mai lung.", "getgoapi": "Conectați GoAPI cu o cheie API.", - "groq": "Nivel gratuit: 30 RPM / 14.4K RPD — fără card de credit", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Obțineți cheia API la haiper.ai/haiper-api", "heroku": "Conectați Heroku AI cu o cheie API.", "hcnsec": "Obțineți cheia API la api.hcnsec.cn", diff --git a/src/i18n/messages/ru.json b/src/i18n/messages/ru.json index c63a9fab99..7b65d8eafa 100644 --- a/src/i18n/messages/ru.json +++ b/src/i18n/messages/ru.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "API-прокси со скидкой для более чем 40 моделей, включая GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Получите API-ключ на https://freeaiapikey.com/dashboard. Базовый URL: https://freeaiapikey.com/v1.", "freemodel-dev": "Получите бесплатный баланс API $300 на https://freemodel.dev — платежная информация не требуется. Совместимая с OpenAI конечная точка. Доступны модели GPT-5.4 и GPT-5.5.", "friendliai": "Бесплатный тариф для бессерверного инференса — кредитная карта не требуется", - "gemini": "Бесплатно навсегда: 1 500 запросов в день для Gemini 2.5 Flash — без кредитной карты, получите ключ на aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Подключите GigaChat (Сбер) с помощью API-ключа.", "gitlab": "Персональный токен доступа GitLab для публичного Code Suggestions API. Настройте собственный базовый URL-адрес, если не используете gitlab.com.", "gitlawb-gmi": "Получите API-ключ в панели управления Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "Подключите GLM Coding (Китай) с помощью API-ключа.", "glmt": "Предустановленный профиль GLM с увеличенным лимитом токенов, включенным режимом рассуждения и более длительным таймаутом.", "getgoapi": "Подключите GoAPI с помощью API-ключа.", - "groq": "Бесплатный тариф: 30 RPM / 14.4K RPD — без кредитной карты", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Получите API-ключ на haiper.ai/haiper-api", "heroku": "Подключите Heroku AI с помощью API-ключа.", "hcnsec": "Получите API-ключ на api.hcnsec.cn", diff --git a/src/i18n/messages/sk.json b/src/i18n/messages/sk.json index de5778a2cd..2f8c814357 100644 --- a/src/i18n/messages/sk.json +++ b/src/i18n/messages/sk.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Zľavnená API proxy pre viac ako 40 modelov vrátane GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Získajte svoj API kľúč na adrese https://freeaiapikey.com/dashboard. Základná URL: https://freeaiapikey.com/v1.", "freemodel-dev": "Získajte bezplatný API kredit 300 $ na adrese https://freemodel.dev — nevyžadujú sa žiadne platobné údaje. Koncový bod kompatibilný s OpenAI. K dispozícii sú modely GPT-5.4 a GPT-5.5.", "friendliai": "Bezplatná úroveň pre serverless inferenciu — nevyžaduje sa kreditná karta", - "gemini": "Navždy zadarmo: 1 500 požiadaviek/deň pre Gemini 2.5 Flash — bez kreditnej karty, kľúč získate na aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Pripojte GigaChat (Sber) pomocou API kľúča.", "gitlab": "Osobný prístupový token GitLab pre verejné API Code Suggestions. Ak nepoužívate gitlab.com, nakonfigurujte vlastnú základnú URL.", "gitlawb-gmi": "Získajte svoj API kľúč z nástenky Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "Pripojte GLM Coding (Čína) pomocou API kľúča.", "glmt": "Prednastavený profil GLM s vyšším rozpočtom tokenov, povoleným premýšľaním a dlhším časovým limitom.", "getgoapi": "Pripojte GoAPI pomocou API kľúča.", - "groq": "Bezplatná úroveň: 30 RPM / 14,4K RPD — bez kreditnej karty", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Získajte API kľúč na adrese haiper.ai/haiper-api", "heroku": "Pripojte Heroku AI pomocou API kľúča.", "hcnsec": "Získajte API kľúč na adrese api.hcnsec.cn", diff --git a/src/i18n/messages/sv.json b/src/i18n/messages/sv.json index 0408b0806b..73a2cea1dd 100644 --- a/src/i18n/messages/sv.json +++ b/src/i18n/messages/sv.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Rabatterad API-proxy för 40+ modeller inklusive GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Hämta din API-nyckel på https://freeaiapikey.com/dashboard. Bas-URL: https://freeaiapikey.com/v1.", "freemodel-dev": "Få $300 i gratis API-krediter på https://freemodel.dev — ingen betalningsinformation krävs. OpenAI-kompatibel slutpunkt. GPT-5.4- och GPT-5.5-modeller tillgängliga.", "friendliai": "Gratisnivå för serverlös inferens — inget kreditkort krävs", - "gemini": "Gratis för alltid: 1 500 anrop/dag för Gemini 2.5 Flash — inget kreditkort, hämta nyckel på aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Anslut GigaChat (Sber) med en API-nyckel.", "gitlab": "Personlig åtkomsttoken för GitLab för det offentliga Code Suggestions-API:et. Konfigurera en egenvärd bas-URL när du inte använder gitlab.com.", "gitlawb-gmi": "Hämta din API-nyckel från instrumentpanelen för Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "Anslut GLM Coding (Kina) med en API-nyckel.", "glmt": "Förinställd GLM-profil med högre tokenbudget, tänkande aktiverat och längre tidsgräns.", "getgoapi": "Anslut GoAPI med en API-nyckel.", - "groq": "Gratisnivå: 30 RPM / 14,4K RPD — inget kreditkort", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Hämta API-nyckel på haiper.ai/haiper-api", "heroku": "Anslut Heroku AI med en API-nyckel.", "hcnsec": "Hämta API-nyckel på api.hcnsec.cn", diff --git a/src/i18n/messages/sw.json b/src/i18n/messages/sw.json index 88b9200348..d1974ce945 100644 --- a/src/i18n/messages/sw.json +++ b/src/i18n/messages/sw.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "Proksi ya API yenye punguzo kwa miundo 40+ ikijumuisha GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Pata ufunguo wako wa API kwenye https://freeaiapikey.com/dashboard. URL ya Msingi: https://freeaiapikey.com/v1.", "freemodel-dev": "Pata salio la bure la API la $300 kwenye https://freemodel.dev — hakuna maelezo ya malipo yanayohitajika. Endpoint inayoendana na OpenAI. Miundo ya GPT-5.4 na GPT-5.5 inapatikana.", "friendliai": "Kiwango cha bure cha makisio yasiyo na seva — hakuna kadi ya mkopo inayohitajika", - "gemini": "Bure milele: maombi 1,500/siku kwa Gemini 2.5 Flash — hakuna kadi ya mkopo, pata ufunguo kwenye aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Unganisha GigaChat (Sber) kwa kutumia ufunguo wa API.", "gitlab": "Tokeni ya ufikiaji wa kibinafsi ya GitLab kwa ajili ya API ya umma ya Code Suggestions. Sanidi URL ya msingi ya self-hosted wakati hutumii gitlab.com.", "gitlawb-gmi": "Pata ufunguo wako wa API kutoka kwenye dashibodi ya Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "Unganisha GLM Coding (China) kwa kutumia ufunguo wa API.", "glmt": "Wasifu uliowekwa awali wa GLM wenye bajeti ya juu ya tokeni, kufikiri kumewashwa, na muda mrefu zaidi wa kuisha.", "getgoapi": "Unganisha GoAPI kwa kutumia ufunguo wa API.", - "groq": "Kiwango cha bure: 30 RPM / 14.4K RPD — hakuna kadi ya mkopo", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Pata ufunguo wa API kwenye haiper.ai/haiper-api", "heroku": "Unganisha Heroku AI kwa kutumia ufunguo wa API.", "hcnsec": "Pata ufunguo wa API kwenye api.hcnsec.cn", diff --git a/src/i18n/messages/ta.json b/src/i18n/messages/ta.json index 95da2289ce..1d12b14de6 100644 --- a/src/i18n/messages/ta.json +++ b/src/i18n/messages/ta.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5 உட்பட 40+ மாதிரிகளுக்கான தள்ளுபடி செய்யப்பட்ட API ப்ராக்ஸி. https://freeaiapikey.com/dashboard இல் உங்கள் API விசையைப் பெறவும். அடிப்படை URL: https://freeaiapikey.com/v1.", "freemodel-dev": "https://freemodel.dev இல் $300 இலவச API கிரெடிட்களைப் பெறவும் — கட்டணத் தகவல் தேவையில்லை. OpenAI-இணக்கமான எண்ட்பாயிண்ட். GPT-5.4 மற்றும் GPT-5.5 மாதிரிகள் கிடைக்கின்றன.", "friendliai": "சர்வர்லெஸ் இன்ஃபெரன்ஸிற்கான இலவச அடுக்கு — கிரெடிட் கார்டு தேவையில்லை", - "gemini": "எப்போதும் இலவசம்: Gemini 2.5 Flash-க்கு ஒரு நாளைக்கு 1,500 கோரிக்கைகள் — கிரெடிட் கார்டு தேவையில்லை, aistudio.google.com இல் விசையைப் பெறவும்", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "GigaChat (Sber) ஐ ஒரு API விசையுடன் இணைக்கவும்.", "gitlab": "பொது குறியீடு பரிந்துரைகள் API க்கான GitLab தனிப்பட்ட அணுகல் டோக்கன். gitlab.com ஐப் பயன்படுத்தாதபோது சுய-ஹோஸ்ட் செய்யப்பட்ட அடிப்படை URL ஐ உள்ளமைக்கவும்.", "gitlawb-gmi": "Gitlawb Opengateway டாஷ்போர்டிலிருந்து உங்கள் API விசையைப் பெறவும்.", @@ -6175,7 +6175,7 @@ "glm-cn": "GLM Coding (China) ஐ ஒரு API விசையுடன் இணைக்கவும்.", "glmt": "அதிக டோக்கன் பட்ஜெட், சிந்தனை இயக்கப்பட்டது மற்றும் நீண்ட காலாவதி நேரத்துடன் கூடிய முன்னமைக்கப்பட்ட GLM சுயவிவரம்.", "getgoapi": "GoAPI ஐ ஒரு API விசையுடன் இணைக்கவும்.", - "groq": "இலவச அடுக்கு: 30 RPM / 14.4K RPD — கிரெடிட் கார்டு தேவையில்லை", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "haiper.ai/haiper-api இல் API விசையைப் பெறவும்", "heroku": "Heroku AI ஐ ஒரு API விசையுடன் இணைக்கவும்.", "hcnsec": "api.hcnsec.cn இல் API விசையைப் பெறுக", diff --git a/src/i18n/messages/te.json b/src/i18n/messages/te.json index 3526a50b41..dc5998cee4 100644 --- a/src/i18n/messages/te.json +++ b/src/i18n/messages/te.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5 తో సహా 40+ మోడల్‌ల కోసం డిస్కౌంట్ పొందిన API ప్రాక్సీ. https://freeaiapikey.com/dashboard వద్ద మీ API కీని పొందండి. బేస్ URL: https://freeaiapikey.com/v1.", "freemodel-dev": "https://freemodel.dev వద్ద $300 ఉచిత API క్రెడిట్‌లను పొందండి — చెల్లింపు సమాచారం అవసరం లేదు. OpenAI-అనుకూల ఎండ్‌పాయింట్. GPT-5.4 మరియు GPT-5.5 మోడల్‌లు అందుబాటులో ఉన్నాయి.", "friendliai": "సర్వర్‌లెస్ ఇన్ఫరెన్స్ కోసం ఉచిత టైర్ — క్రెడిట్ కార్డ్ అవసరం లేదు", - "gemini": "ఎప్పటికీ ఉచితం: Gemini 2.5 Flash కోసం రోజుకు 1,500 అభ్యర్థనలు — క్రెడిట్ కార్డ్ అవసరం లేదు, aistudio.google.com వద్ద కీని పొందండి", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "API కీతో GigaChat (Sber) ని కనెక్ట్ చేయండి.", "gitlab": "పబ్లిక్ Code Suggestions API కోసం GitLab వ్యక్తిగత యాక్సెస్ టోకెన్. gitlab.com ని ఉపయోగించనప్పుడు సెల్ఫ్-హోస్టెడ్ బేస్ URLని కాన్ఫిగర్ చేయండి.", "gitlawb-gmi": "Gitlawb Opengateway డ్యాష్‌బోర్డ్ నుండి మీ API కీని పొందండి.", @@ -6175,7 +6175,7 @@ "glm-cn": "API కీతో GLM Coding (China) ని కనెక్ట్ చేయండి.", "glmt": "ఎక్కువ టోకెన్ బడ్జెట్, థింకింగ్ ఎనేబుల్ చేయబడిన మరియు ఎక్కువ టైమ్‌అవుట్‌తో కూడిన ప్రీసెట్ GLM ప్రొఫైల్.", "getgoapi": "API కీతో GoAPI ని కనెక్ట్ చేయండి.", - "groq": "ఉచిత టైర్: 30 RPM / 14.4K RPD — క్రెడిట్ కార్డ్ అవసరం లేదు", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "haiper.ai/haiper-api వద్ద API కీని పొందండి", "heroku": "API కీతో Heroku AI ని కనెక్ట్ చేయండి.", "hcnsec": "api.hcnsec.cn వద్ద API కీని పొందండి", diff --git a/src/i18n/messages/th.json b/src/i18n/messages/th.json index e19eb88c20..13b002b810 100644 --- a/src/i18n/messages/th.json +++ b/src/i18n/messages/th.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "พร็อกซี API ราคาพิเศษสำหรับโมเดลมากกว่า 40 โมเดล รวมถึง GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5 รับคีย์ API ของคุณได้ที่ https://freeaiapikey.com/dashboard URL ฐาน: https://freeaiapikey.com/v1", "freemodel-dev": "รับเครดิต API ฟรี $300 ที่ https://freemodel.dev — ไม่ต้องใช้ข้อมูลการชำระเงิน ปลายทางที่เข้ากันได้กับ OpenAI มีโมเดล GPT-5.4 และ GPT-5.5 ให้บริการ", "friendliai": "ระดับการใช้งานฟรีสำหรับการอนุมานแบบ Serverless — ไม่ต้องใช้บัตรเครดิต", - "gemini": "ฟรีตลอดชีพ: 1,500 คำขอ/วันสำหรับ Gemini 2.5 Flash — ไม่ต้องใช้บัตรเครดิต รับคีย์ได้ที่ aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "เชื่อมต่อ GigaChat (Sber) ด้วยคีย์ API", "gitlab": "โทเค็นการเข้าถึงส่วนตัวของ GitLab สำหรับ Code Suggestions API สาธารณะ กำหนดค่า URL ฐานแบบโฮสต์เองเมื่อไม่ได้ใช้งาน gitlab.com", "gitlawb-gmi": "รับคีย์ API ของคุณจากแดชบอร์ด Gitlawb Opengateway", @@ -6175,7 +6175,7 @@ "glm-cn": "เชื่อมต่อ GLM Coding (China) ด้วยคีย์ API", "glmt": "โปรไฟล์ GLM ที่ตั้งค่าไว้ล่วงหน้าพร้อมงบประมาณโทเค็นที่สูงขึ้น เปิดใช้งานการคิด และหมดเวลาการทำงานที่นานขึ้น", "getgoapi": "เชื่อมต่อ GoAPI ด้วยคีย์ API", - "groq": "ระดับการใช้งานฟรี: 30 RPM / 14.4K RPD — ไม่ต้องใช้บัตรเครดิต", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "รับคีย์ API ได้ที่ haiper.ai/haiper-api", "heroku": "เชื่อมต่อ Heroku AI ด้วยคีย์ API", "hcnsec": "รับ API key ได้ที่ api.hcnsec.cn", diff --git a/src/i18n/messages/tr.json b/src/i18n/messages/tr.json index 12b3973933..3b90a96aac 100644 --- a/src/i18n/messages/tr.json +++ b/src/i18n/messages/tr.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5 dahil 40'tan fazla model için indirimli API proxy'si. API anahtarınızı https://freeaiapikey.com/dashboard adresinden alın. Temel URL: https://freeaiapikey.com/v1.", "freemodel-dev": "https://freemodel.dev adresinden 300$ ücretsiz API kredisi alın — ödeme bilgisi gerekmez. OpenAI uyumlu uç nokta. GPT-5.4 ve GPT-5.5 modelleri mevcuttur.", "friendliai": "Sunucusuz çıkarım için ücretsiz katman — kredi kartı gerekmez", - "gemini": "Sonsuza kadar ücretsiz: Gemini 2.5 Flash için günlük 1.500 istek — kredi kartı gerekmez, anahtarı aistudio.google.com adresinden alın", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "GigaChat'i (Sber) bir API anahtarı ile bağlayın.", "gitlab": "Genel Code Suggestions API'si için GitLab kişisel erişim belirteci. gitlab.com kullanmadığınızda barındırılan bir temel URL yapılandırın.", "gitlawb-gmi": "API anahtarınızı Gitlawb Opengateway panelinden alın.", @@ -6175,7 +6175,7 @@ "glm-cn": "GLM Coding'i (Çin) bir API anahtarı ile bağlayın.", "glmt": "Daha yüksek token bütçesi, düşünme etkinleştirilmiş ve daha uzun zaman aşımına sahip önceden ayarlanmış GLM profili.", "getgoapi": "GoAPI'yi bir API anahtarı ile bağlayın.", - "groq": "Ücretsiz katman: 30 RPM / 14.4K RPD — kredi kartı gerekmez", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "API anahtarını haiper.ai/haiper-api adresinden alın", "heroku": "Heroku AI'ı bir API anahtarı ile bağlayın.", "hcnsec": "API anahtarını api.hcnsec.cn adresinden alın", diff --git a/src/i18n/messages/uk-UA.json b/src/i18n/messages/uk-UA.json index fd9a0732d3..a84e6b9b77 100644 --- a/src/i18n/messages/uk-UA.json +++ b/src/i18n/messages/uk-UA.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "API-проксі зі знижкою для понад 40 моделей, включаючи GPT-5, Claude Opus 4.6, Claude Sonnet 4.6, Qwen 3.5. Отримайте свій API-ключ на https://freeaiapikey.com/dashboard. Базова URL-адреса: https://freeaiapikey.com/v1.", "freemodel-dev": "Отримайте $300 безкоштовних API-кредитів на https://freemodel.dev — платіжна інформація не потрібна. Сумісна з OpenAI кінцева точка. Доступні моделі GPT-5.4 та GPT-5.5.", "friendliai": "Безкоштовний тариф для безсерверного виведення — кредитна картка не потрібна", - "gemini": "Безкоштовно назавжди: 1500 зап/день для Gemini 2.5 Flash — без кредитної картки, отримайте ключ на aistudio.google.com", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "Підключіть GigaChat (Сбер) за допомогою API-ключа.", "gitlab": "Особистий токен доступу GitLab для публічного API Code Suggestions. Налаштуйте власну базову URL-адресу, якщо не використовуєте gitlab.com.", "gitlawb-gmi": "Отримайте свій API-ключ на панелі керування Gitlawb Opengateway.", @@ -6175,7 +6175,7 @@ "glm-cn": "Підключіть GLM Coding (Китай) за допомогою API-ключа.", "glmt": "Попередньо встановлений профіль GLM із більшим бюджетом токенів, увімкненим мисленням та довшим таймаутом.", "getgoapi": "Підключіть GoAPI за допомогою API-ключа.", - "groq": "Безкоштовний тариф: 30 RPM / 14.4K RPD — без кредитної картки", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "Отримайте API-ключ на haiper.ai/haiper-api", "heroku": "Підключіть Heroku AI за допомогою API-ключа.", "hcnsec": "Отримайте API-ключ на api.hcnsec.cn", diff --git a/src/i18n/messages/ur.json b/src/i18n/messages/ur.json index 02dba16c45..fa38b24d3e 100644 --- a/src/i18n/messages/ur.json +++ b/src/i18n/messages/ur.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "GPT-5، Claude Opus 4.6، Claude Sonnet 4.6، Qwen 3.5 سمیت 40+ ماڈلز کے لیے رعایتی API پراکسی۔ اپنی API کی https://freeaiapikey.com/dashboard پر حاصل کریں۔ بیس URL: https://freeaiapikey.com/v1۔", "freemodel-dev": "https://freemodel.dev پر $300 کے مفت API کریڈٹس حاصل کریں — ادائیگی کی معلومات درکار نہیں۔ OpenAI-compatible اینڈ پوائنٹ۔ GPT-5.4 اور GPT-5.5 ماڈلز دستیاب ہیں۔", "friendliai": "سرور لیس انفیرنس کے لیے مفت ٹیر — کسی کریڈٹ کارڈ کی ضرورت نہیں", - "gemini": "ہمیشہ کے لیے مفت: Gemini 2.5 Flash کے لیے 1,500 درخواستیں/دن — کوئی کریڈٹ کارڈ نہیں، aistudio.google.com پر کی حاصل کریں", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "GigaChat (Sber) کو ایک API کی کے ساتھ منسلک کریں۔", "gitlab": "عوامی Code Suggestions API کے لیے GitLab پرسنل ایکسیس ٹوکن۔ جب gitlab.com استعمال نہ کر رہے ہوں تو ایک سیلف ہوسٹڈ بیس URL کنفیگر کریں۔", "gitlawb-gmi": "Gitlawb Opengateway ڈیش بورڈ سے اپنی API کی حاصل کریں۔", @@ -6175,7 +6175,7 @@ "glm-cn": "GLM Coding (China) کو ایک API کی کے ساتھ منسلک کریں۔", "glmt": "زیادہ ٹوکن بجٹ، تھنکنگ فعال، اور طویل ٹائم آؤٹ کے ساتھ پہلے سے سیٹ کردہ GLM پروفائل۔", "getgoapi": "GoAPI کو ایک API کی کے ساتھ منسلک کریں۔", - "groq": "مفت ٹیر: 30 RPM / 14.4K RPD — کوئی کریڈٹ کارڈ نہیں", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "haiper.ai/haiper-api پر API کی حاصل کریں", "heroku": "Heroku AI کو ایک API کی کے ساتھ منسلک کریں۔", "hcnsec": "api.hcnsec.cn پر API کی حاصل کریں", diff --git a/src/i18n/messages/vi.json b/src/i18n/messages/vi.json index cf7235f61b..cb5dc787d4 100644 --- a/src/i18n/messages/vi.json +++ b/src/i18n/messages/vi.json @@ -6170,7 +6170,7 @@ "freeaiapikey": "Proxy API giảm giá cho hơn 40 mô hình, gồm GPT-5, Claude Opus 4.6, Claude Sonnet 4.6 và Qwen 3.5. Lấy khóa API tại https://freeaiapikey.com/dashboard. URL cơ sở: https://freeaiapikey.com/v1.", "freemodel-dev": "Nhận 300 USD tín dụng API miễn phí tại https://freemodel.dev — không cần thông tin thanh toán. Endpoint tương thích OpenAI. Có GPT-5.4 và GPT-5.5.", "friendliai": "Gói miễn phí cho suy luận serverless — không cần thẻ tín dụng", - "gemini": "Miễn phí vĩnh viễn: 1.500 yêu cầu/ngày cho Gemini 2.5 Flash — không cần thẻ tín dụng, lấy khóa tại aistudio.google.com", + "gemini": "Gói miễn phí qua Google AI Studio; hạn mức theo từng mô hình không còn được công bố (xem trang hạn mức trong AI Studio) — không cần thẻ tín dụng.", "gigachat": "Kết nối GigaChat (Sber) bằng khóa API.", "gitlab": "Personal access token GitLab cho API Code Suggestions công khai. Cấu hình URL cơ sở tự lưu trữ khi không dùng gitlab.com.", "gitlawb-gmi": "Lấy khóa API của bạn từ Gitlawb Opengateway dashboard.", @@ -6179,7 +6179,7 @@ "glm-cn": "Kết nối GLM Coding (China) bằng khóa API.", "glmt": "Hồ sơ GLM đặt sẵn với ngân sách token cao hơn, bật thinking và thời gian chờ dài hơn.", "getgoapi": "Kết nối GoAPI bằng khóa API.", - "groq": "Gói miễn phí: 30 RPM / 14,4 nghìn RPD — không cần thẻ tín dụng", + "groq": "Gói miễn phí: giới hạn theo từng mô hình (200K token/ngày cho mỗi mô hình chat; xem console.groq.com/docs/rate-limits để biết RPM/RPD) — không cần đăng ký phương thức thanh toán.", "haiper": "Lấy khóa API tại haiper.ai/haiper-api", "heroku": "Kết nối Heroku AI bằng khóa API.", "hcnsec": "Lấy khóa API tại api.hcnsec.cn", diff --git a/src/i18n/messages/zh-CN.json b/src/i18n/messages/zh-CN.json index e53f191cf9..14413d9130 100644 --- a/src/i18n/messages/zh-CN.json +++ b/src/i18n/messages/zh-CN.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "适用于 40+ 种模型的折扣 API 代理,包括 GPT-5、Claude Opus 4.6、Claude Sonnet 4.6、Qwen 3.5。在 https://freeaiapikey.com/dashboard 获取您的 API 密钥。Base URL: https://freeaiapikey.com/v1.", "freemodel-dev": "在 https://freemodel.dev 获取 $300 免费 API 额度 — 无需支付信息。兼容 OpenAI 的端点。提供 GPT-5.4 和 GPT-5.5 模型。", "friendliai": "无服务器推理免费层 — 无需信用卡", - "gemini": "永久免费:Gemini 2.5 Flash 每天 1,500 次请求 — 无需信用卡,在 aistudio.google.com 获取密钥", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "使用 API 密钥连接 GigaChat (Sber)。", "gitlab": "用于公共 Code Suggestions API 的 GitLab 个人访问令牌。不使用 gitlab.com 时请配置自托管 Base URL。", "gitlawb-gmi": "从 Gitlawb Opengateway 控制面板获取您的 API 密钥。", @@ -6175,7 +6175,7 @@ "glm-cn": "使用 API 密钥连接 GLM Coding (China)。", "glmt": "预设 GLM 配置文件,具有更高的 Token 预算、启用思考功能以及更长的超时时间。", "getgoapi": "使用 API 密钥连接 GoAPI。", - "groq": "免费层:30 RPM / 14.4K RPD — 无需信用卡", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "在 haiper.ai/haiper-api 获取 API 密钥", "heroku": "使用 API 密钥连接 Heroku AI。", "hcnsec": "在 api.hcnsec.cn 获取 API 密钥", diff --git a/src/i18n/messages/zh-TW.json b/src/i18n/messages/zh-TW.json index 2cddce94b2..77932434c9 100644 --- a/src/i18n/messages/zh-TW.json +++ b/src/i18n/messages/zh-TW.json @@ -6166,7 +6166,7 @@ "freeaiapikey": "40+ 種模型的折扣 API 代理,包括 GPT-5、Claude Opus 4.6、Claude Sonnet 4.6、Qwen 3.5。在 https://freeaiapikey.com/dashboard 取得 API 金鑰。基本 URL:https://freeaiapikey.com/v1。", "freemodel-dev": "在 https://freemodel.dev 取得 $300 美元免費 API 額度 — 無需付款資訊。OpenAI 相容端點。提供 GPT-5.4 和 GPT-5.5 模型。", "friendliai": "無伺服器推論的免費方案 — 無需信用卡", - "gemini": "永久免費:Gemini 2.5 Flash 每天 1,500 次請求 — 無需信用卡,在 aistudio.google.com 取得金鑰", + "gemini": "__MISSING__:Free tier through Google AI Studio; per-model quotas are no longer published (check the quota page in AI Studio) — no credit card.", "gigachat": "使用 API 金鑰連線 GigaChat(Sber)。", "gitlab": "用於公開 Code Suggestions API 的 GitLab 個人存取權杖。不使用 gitlab.com 時,請設定自託管的基本 URL。", "gitlawb-gmi": "從 Gitlawb Opengateway 儀表板取得 API 金鑰。", @@ -6175,7 +6175,7 @@ "glm-cn": "使用 API 金鑰連線 GLM Coding(中國)。", "glmt": "預設 GLM 設定檔,具有較高的 token 預算、啟用思考功能,以及更長的超時時間。", "getgoapi": "使用 API 金鑰連線 GoAPI。", - "groq": "免費方案:每分鐘 30 次 / 每天 14,400 次請求 — 無需信用卡", + "groq": "__MISSING__:Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", "haiper": "在 haiper.ai/haiper-api 取得 API 金鑰", "heroku": "使用 API 金鑰連線 Heroku AI。", "hcnsec": "在 api.hcnsec.cn 取得 API 金鑰", diff --git a/src/shared/constants/providers/apikey/frontier-labs.ts b/src/shared/constants/providers/apikey/frontier-labs.ts index 71609ef544..e22a0e4330 100644 --- a/src/shared/constants/providers/apikey/frontier-labs.ts +++ b/src/shared/constants/providers/apikey/frontier-labs.ts @@ -92,7 +92,8 @@ export const APIKEY_PROVIDERS_FRONTIER = { textIcon: "GQ", website: "https://groq.com", hasFree: true, - freeNote: "Free tier: 30 RPM / 14.4K RPD — no credit card", + freeNote: + "Free plan: per-model caps (200K tokens/day per chat model; see console.groq.com/docs/rate-limits for RPM/RPD) — no payment method on file.", serviceKinds: ["llm", "imageToText"], }, blackbox: { diff --git a/src/shared/constants/providers/apikey/gateways.ts b/src/shared/constants/providers/apikey/gateways.ts index 7f09204e3d..c1f87a6d75 100644 --- a/src/shared/constants/providers/apikey/gateways.ts +++ b/src/shared/constants/providers/apikey/gateways.ts @@ -15,7 +15,8 @@ export const APIKEY_PROVIDERS_GATEWAYS = { color: "#6366F1", textIcon: "1M", website: "https://1min.ai", - authHint: "Create an API key at https://docs.1min.ai/docs/api/create-api-key, then paste it here.", + authHint: + "Create an API key at https://docs.1min.ai/docs/api/create-api-key, then paste it here.", apiHint: "1min.ai uses a proprietary chat API (single prompt string + SSE) instead of OpenAI chat/completions. OmniRoute flattens OpenAI messages into a labeled prompt and translates the SSE stream.", passthroughModels: true, @@ -47,7 +48,8 @@ export const APIKEY_PROVIDERS_GATEWAYS = { website: "https://freebuff.com", hasFree: true, serviceKinds: ["llm"], - authHint: "Enter Freebuff / Codebuff Auth Token (obtained via CLI login or automated harvester).", + authHint: + "Enter Freebuff / Codebuff Auth Token (obtained via CLI login or automated harvester).", freeNote: "Free Codebuff / Freebuff AI models.", apiHint: "Token is authenticated against Codebuff upstream session pool.", passthroughModels: true, @@ -1327,9 +1329,10 @@ export const APIKEY_PROVIDERS_GATEWAYS = { passthroughModels: true, website: "https://bynara.id", hasFree: true, - freeNote: "Free tier is a shared 5M tokens/day pool; some models are gated behind credit/plan.", + freeNote: + "Free plan: one 7M tokens/day bucket per account (15 req/min) across the plan's 8 models; others need credit.", authHint: - "Get a free API key via NaraRouter's Telegram channel, then paste it here as a Bearer token.", + "Create a free NaraRouter account, link your Telegram (required before /v1 answers), then paste the key here as a Bearer token.", apiHint: "OpenAI-compatible endpoint at https://router.bynara.id/v1. Free-tier models are pinned; others need credit.", }, diff --git a/tests/unit/autoCombo/strict-zero-cost-filter.test.ts b/tests/unit/autoCombo/strict-zero-cost-filter.test.ts index 59ab8c226c..6966ec6f94 100644 --- a/tests/unit/autoCombo/strict-zero-cost-filter.test.ts +++ b/tests/unit/autoCombo/strict-zero-cost-filter.test.ts @@ -50,7 +50,8 @@ const KEYLESS = { }; // A real quota-based entry with hardStopGuaranteed: true (added by this feature), // as a concrete single-connection candidate. -const QUOTA_SAFE = { provider: "groq", model: "llama-3.3-70b-versatile", connectionId: REAL_CONN }; +// 2026-09-02: was groq/llama-3.3-70b-versatile, retired from the Groq free tier on 2026-08-16. +const QUOTA_SAFE = { provider: "groq", model: "openai/gpt-oss-120b", connectionId: REAL_CONN }; // A real quota-based entry WITHOUT hardStopGuaranteed (agentrouter: one-time-initial, // no usage adapter, no documented "no credit card" claim — must never pass). const QUOTA_UNGUARANTEED = { @@ -69,7 +70,7 @@ const PAID = { provider: "openai", model: "gpt-4o", connectionId: REAL_CONN }; test("sanity: fixtures exist in the real catalog with the metadata these tests assume", () => { const groqEntry = FREE_MODEL_BUDGETS.find( - (m) => m.provider === "groq" && m.modelId === "llama-3.3-70b-versatile" + (m) => m.provider === "groq" && m.modelId === "openai/gpt-oss-120b" ); assert.equal(groqEntry?.hardStopGuaranteed, true, "groq must carry hardStopGuaranteed: true"); const arEntry = FREE_MODEL_BUDGETS.find( @@ -121,7 +122,7 @@ test("model absent from the free catalog is excluded even under a known provider // 5. quota SAFE + fresh + hardStop → PASS test("quota-based candidate with hardStopGuaranteed, fresh SAFE state above threshold passes", () => { const entry = FREE_MODEL_BUDGETS.find( - (m) => m.provider === "groq" && m.modelId === "llama-3.3-70b-versatile" + (m) => m.provider === "groq" && m.modelId === "openai/gpt-oss-120b" ); assert.deepEqual( evaluateCandidateConnections( @@ -137,7 +138,7 @@ test("quota-based candidate with hardStopGuaranteed, fresh SAFE state above thre // 6. quota exhausted → EXCLUDE test("EXHAUSTED status excludes even with a fresh checkedAt", () => { const entry = FREE_MODEL_BUDGETS.find( - (m) => m.provider === "groq" && m.modelId === "llama-3.3-70b-versatile" + (m) => m.provider === "groq" && m.modelId === "openai/gpt-oss-120b" ); const state = freshState({ status: "EXHAUSTED", remainingFreeAllowance: 0 }); assert.deepEqual( @@ -149,7 +150,7 @@ test("EXHAUSTED status excludes even with a fresh checkedAt", () => { // 7. usage adapter absent (no state resolvable) → EXCLUDE test("quota-based candidate with no resolvable state is excluded, not assumed safe", () => { const entry = FREE_MODEL_BUDGETS.find( - (m) => m.provider === "groq" && m.modelId === "llama-3.3-70b-versatile" + (m) => m.provider === "groq" && m.modelId === "openai/gpt-oss-120b" ); assert.deepEqual( evaluateCandidateConnections(QUOTA_SAFE, entry, () => undefined, BASE_OPTIONS), @@ -162,7 +163,7 @@ test("quota-based candidate with no resolvable state is excluded, not assumed sa // here — same assertion as #7, the important contract is "never falls back to SAFE"). test("UNKNOWN status excludes", () => { const entry = FREE_MODEL_BUDGETS.find( - (m) => m.provider === "groq" && m.modelId === "llama-3.3-70b-versatile" + (m) => m.provider === "groq" && m.modelId === "openai/gpt-oss-120b" ); const state = freshState({ status: "UNKNOWN", remainingFreeAllowance: null }); assert.deepEqual( @@ -174,7 +175,7 @@ test("UNKNOWN status excludes", () => { // 9. usage state stale → EXCLUDE test("stale checkedAt excludes even when status is SAFE", () => { const entry = FREE_MODEL_BUDGETS.find( - (m) => m.provider === "groq" && m.modelId === "llama-3.3-70b-versatile" + (m) => m.provider === "groq" && m.modelId === "openai/gpt-oss-120b" ); const stale = freshState({ checkedAt: "2026-08-19T00:00:00.000Z" }); // >24h before NOW assert.deepEqual( diff --git a/tests/unit/free-providers-batch-2026-07.test.ts b/tests/unit/free-providers-batch-2026-07.test.ts index e58c6f5dd2..5c913d562e 100644 --- a/tests/unit/free-providers-batch-2026-07.test.ts +++ b/tests/unit/free-providers-batch-2026-07.test.ts @@ -47,12 +47,12 @@ test("providers with no published token quota never inflate the headline", () => } }); -test("nara is a single shared 5M/day pool, counted once", () => { +test("nara is a single shared 7M/day pool, counted once", () => { const rows = byProvider("nara"); assert.ok(rows.length >= 1); - // 5M tokens/day shared across all models => 150M/month, deduped by poolKey. + // 7M tokens/day shared across all plan models => 210M/month (re-audited 2026-09-02, GET /api/plans). assert.ok(rows.every((m) => m.poolKey === "nara-free")); - assert.ok(rows.every((m) => m.monthlyTokens === 150_000_000)); + assert.ok(rows.every((m) => m.monthlyTokens === 210_000_000)); assert.ok(rows.every((m) => m.freeType === "recurring-daily")); }); diff --git a/tests/unit/free-tier-catalog.test.ts b/tests/unit/free-tier-catalog.test.ts index 25282d5ef1..4a804261ca 100644 --- a/tests/unit/free-tier-catalog.test.ts +++ b/tests/unit/free-tier-catalog.test.ts @@ -7,7 +7,9 @@ import { } from "../../open-sse/config/freeTierCatalog.ts"; test("FREE_TIER_BUDGETS holds positive integer monthly-token budgets", () => { - assert.ok(Object.keys(FREE_TIER_BUDGETS).length >= 18); + // 2026-09-02 re-audit: gemini + ollama-cloud left (no published cap), nara joined; + // #12591 then dropped cerebras (one-time credit) → 17 legacy keys. + assert.ok(Object.keys(FREE_TIER_BUDGETS).length >= 17); for (const [id, tokens] of Object.entries(FREE_TIER_BUDGETS)) { assert.ok(Number.isInteger(tokens) && tokens > 0, `${id} must be a positive integer`); } @@ -28,16 +30,20 @@ test("FREE_TIER_TOS marks proxy-prohibited providers as avoid", () => { test("computeFreeTierTotals sums the documented budgets", () => { const t = computeFreeTierTotals(); - assert.equal(t.providerCount, 18); - assert.ok(t.documentedMonthlyTokens >= 1_320_000_000); - assert.ok(t.documentedMonthlyTokens <= 1_420_000_000); + // 2026-09-02 re-audit: gemini + ollama-cloud left the legacy map (no published cap), + // nara joined, groq → 30M; #12591 dropped cerebras → 17 providers, legacy sum 1,475,025,000. + assert.equal(t.providerCount, 17); + assert.ok(t.documentedMonthlyTokens >= 1_425_000_000); + assert.ok(t.documentedMonthlyTokens <= 1_525_000_000); assert.equal(typeof t.headline, "string"); - assert.match(t.headline, /1\.3/); + // 2026-09-02 re-audit + #12591 (cerebras dropped): headline reads "over 1.48B …". + assert.match(t.headline, /1\.4/); }); test("computeFreeTierTotals can exclude ToS-avoid providers", () => { const all = computeFreeTierTotals(); const clean = computeFreeTierTotals({ excludeTosAvoid: true }); assert.equal(all.documentedMonthlyTokens - clean.documentedMonthlyTokens, 25_000); - assert.equal(clean.providerCount, 17); + // 2026-09-02 re-audit + #12591: 17 legacy providers, minus kiro (ToS avoid) → 16. + assert.equal(clean.providerCount, 16); }); diff --git a/tests/unit/free-tier-reaudit-2026-09.test.ts b/tests/unit/free-tier-reaudit-2026-09.test.ts new file mode 100644 index 0000000000..f227616349 --- /dev/null +++ b/tests/unit/free-tier-reaudit-2026-09.test.ts @@ -0,0 +1,117 @@ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import test from "node:test"; + +import { + FREE_MODEL_BUDGETS, + computeFreeModelTotals, +} from "@omniroute/open-sse/config/freeModelCatalog.ts"; +import { FREE_TIER_BUDGETS } from "@omniroute/open-sse/config/freeTierCatalog.ts"; +import { REGISTRY } from "@omniroute/open-sse/config/providerRegistry.ts"; + +/** + * 2026-09-02 re-audit against the providers' own pages — the official pages + * cited in the `// evidence:` comments next to each entry in + * `open-sse/config/freeModelCatalog.data.ts`. Each test pins one verified + * fact so the catalog cannot drift back to an invented number. + */ +const rows = (id: string) => FREE_MODEL_BUDGETS.filter((m) => m.provider === id); +const ids = (id: string) => + rows(id) + .map((m) => m.modelId) + .sort(); + +test("gemini publishes no per-model free limits any more — uncapped, never summed", () => { + const g = rows("gemini"); + assert.ok(g.length >= 1); + assert.ok(g.every((m) => m.freeType === "recurring-uncapped" && m.monthlyTokens === 0)); + assert.ok(computeFreeModelTotals().uncappedProviders.includes("gemini")); +}); + +test("ollama-cloud free plan has no published token cap — uncapped, never summed", () => { + const o = rows("ollama-cloud"); + assert.ok(o.length >= 1); + assert.ok(o.every((m) => m.freeType === "recurring-uncapped" && m.monthlyTokens === 0)); + assert.ok(computeFreeModelTotals().uncappedProviders.includes("ollama-cloud")); +}); + +test("groq free plan: 200K TPD per model × 30 = 6M per model, each cap independent", () => { + assert.deepEqual(ids("groq"), [ + "openai/gpt-oss-120b", + "openai/gpt-oss-20b", + "openai/gpt-oss-safeguard-20b", + "qwen/qwen3.6-27b", + "qwen/qwen3.8-27b", + ]); + for (const m of rows("groq")) { + assert.equal(m.monthlyTokens, 6_000_000, m.modelId); + assert.equal(m.poolKey, null, `${m.modelId} is a per-model cap, not a shared pool`); + assert.equal(m.freeType, "recurring-daily"); + assert.equal(m.hardStopGuaranteed, true); + } + // retired from the free tier on 2026-07-17 / 2026-08-16 (console.groq.com/docs/deprecations) + for (const dead of [ + "llama-3.3-70b-versatile", + "meta-llama/llama-4-scout-17b-16e-instruct", + "qwen/qwen3-32b", + ]) { + assert.ok(!ids("groq").includes(dead), `${dead} must not be in the free catalog`); + } + const registryIds = new Set(REGISTRY.groq.models.map((m) => m.id)); + for (const id of ids("groq")) assert.ok(registryIds.has(id), `${id} must be routable`); +}); + +test("nara free plan: 7M tokens/day account-wide → 210M/month, one pool, the 8 plan models", () => { + const free = [ + "agnes-2.0-flash", + "agnes-2.5-flash", + "laguna-s-2.1", + "minimax-m3-free", + "mistral-large", + "mistral-medium-3-5", + "qwen3.8-27b", + "stepfun-3.7-flash", + ]; + assert.deepEqual(ids("nara"), free); + for (const m of rows("nara")) { + assert.equal(m.monthlyTokens, 210_000_000, m.modelId); + assert.equal(m.poolKey, "nara-free"); + assert.equal(m.freeType, "recurring-daily"); + } + assert.deepEqual(REGISTRY.nara.models.map((m) => m.id).sort(), free); +}); + +test("mistral keeps its 1B pool only with a dated console verification on record", () => { + const src = readFileSync( + new URL("../../open-sse/config/freeModelCatalog.data.ts", import.meta.url), + "utf8" + ); + const m = rows("mistral"); + assert.ok(m.length >= 1); + const pooled = m.filter((r) => r.monthlyTokens > 0); + if (pooled.length > 0) { + // All-or-nothing: a mixed 1B/0 state is neither console-verified nor honestly uncapped. + assert.equal(pooled.length, m.length, "every mistral row must carry the pooled 1B"); + assert.ok( + pooled.every( + (r) => + r.monthlyTokens === 1_000_000_000 && + r.poolKey === "mistral" && + r.freeType === "recurring-monthly" + ) + ); + assert.match( + src, + /evidence: console-verified 20\d\d-\d\d-\d\d por \S+ \(https:\/\/console\.mistral\.ai/ + ); + } else { + assert.ok(m.every((r) => r.monthlyTokens === 0 && r.freeType === "recurring-uncapped")); + } +}); + +test("legacy provider-level catalog agrees with the per-model catalog for the re-audited providers", () => { + assert.equal(FREE_TIER_BUDGETS.gemini, undefined); + assert.equal(FREE_TIER_BUDGETS["ollama-cloud"], undefined); + assert.equal(FREE_TIER_BUDGETS.groq, 30_000_000); + assert.equal(FREE_TIER_BUDGETS.nara, 210_000_000); +}); From d6771779f7ac08e9fb9f029d5822327958326731 Mon Sep 17 00:00:00 2001 From: Krzysztof Skomra <159253490+KrzysiekSko@users.noreply.github.com> Date: Fri, 4 Sep 2026 04:26:59 +0200 Subject: [PATCH 075/143] fix(cli): preserve Claude settings on config set (#12432) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 6 PRs destas duas levas sobre o tip de `release/v3.8.51`: os seis boardaram **sem um único conflito**, `typecheck:core` limpo e **54/54** nos 7 arquivos de teste que trazem. O drift de `i18n:check` (`docs/security/GUARDRAILS.md`, `STEALTH_GUIDE.md` — source-changed) foi medido também no tip puro e é idêntico: base-red pré-existente, não desta leva. --- bin/cli/commands/config.mjs | 30 +++++++- src/lib/cli-helper/config-generator/claude.ts | 10 ++- .../unit/cli/claude-config-set-12407.test.ts | 77 +++++++++++++++++++ 3 files changed, 113 insertions(+), 4 deletions(-) create mode 100644 tests/unit/cli/claude-config-set-12407.test.ts diff --git a/bin/cli/commands/config.mjs b/bin/cli/commands/config.mjs index 6376ba9217..e8a43e0a6c 100644 --- a/bin/cli/commands/config.mjs +++ b/bin/cli/commands/config.mjs @@ -16,6 +16,29 @@ function ensureBackup(configPath) { return backupPath; } +function mergeClaudeSettings(existingContent, generatedContent) { + const generated = JSON.parse(generatedContent); + let current = {}; + if (existingContent && existingContent.trim()) { + current = JSON.parse(existingContent); + if (!current || typeof current !== "object" || Array.isArray(current)) current = {}; + } + return JSON.stringify( + { + ...current, + ...generated, + env: { + ...(current.env && typeof current.env === "object" && !Array.isArray(current.env) + ? current.env + : {}), + ...(generated.env || {}), + }, + }, + null, + 2 + ); +} + async function runConfigListCommand(opts = {}) { const { detectAllTools } = await import("../../../src/lib/cli-helper/tool-detector.ts"); const tools = await detectAllTools(); @@ -120,7 +143,12 @@ async function runConfigSetCommand(toolId, opts = {}) { const backupPath = ensureBackup(result.configPath); if (backupPath) printInfo(`Backup saved to: ${backupPath}`); - fs.writeFileSync(result.configPath, result.content, "utf-8"); + let content = result.content; + if (toolId === "claude" && fs.existsSync(result.configPath)) { + content = mergeClaudeSettings(fs.readFileSync(result.configPath, "utf-8"), result.content); + } + + fs.writeFileSync(result.configPath, content, "utf-8"); printSuccess(`Config written to ${result.configPath}`); return 0; } diff --git a/src/lib/cli-helper/config-generator/claude.ts b/src/lib/cli-helper/config-generator/claude.ts index 2b2490690f..645da8fc04 100644 --- a/src/lib/cli-helper/config-generator/claude.ts +++ b/src/lib/cli-helper/config-generator/claude.ts @@ -16,9 +16,13 @@ export function generateClaudeConfig(options: { const model = options.model || "claude-3-5-sonnet-20241022"; const config = { - baseUrl: `${base}/v1`, - authToken: options.apiKey, - models: [{ id: model }], + model, + env: { + ANTHROPIC_BASE_URL: base, + ANTHROPIC_AUTH_TOKEN: options.apiKey, + ANTHROPIC_MODEL: model, + CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY: "1", + }, }; return JSON.stringify(config, null, 2); diff --git a/tests/unit/cli/claude-config-set-12407.test.ts b/tests/unit/cli/claude-config-set-12407.test.ts new file mode 100644 index 0000000000..12b0653fe4 --- /dev/null +++ b/tests/unit/cli/claude-config-set-12407.test.ts @@ -0,0 +1,77 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { execFile } from "node:child_process"; +import { promisify } from "node:util"; +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; + +const execFileAsync = promisify(execFile); +const repoRoot = path.resolve(import.meta.dirname, "../../.."); +const cliPath = path.join(repoRoot, "bin", "omniroute.mjs"); + +test("#12407: config set claude preserves existing settings and writes Claude Code env keys", async () => { + const home = await fs.mkdtemp(path.join(os.tmpdir(), "omniroute-claude-config-")); + try { + const settingsPath = path.join(home, ".claude", "settings.json"); + await fs.mkdir(path.dirname(settingsPath), { recursive: true }); + await fs.writeFile( + settingsPath, + JSON.stringify( + { + model: "existing-model", + effortLevel: "high", + hooks: { PreToolUse: [{ command: "echo keep" }] }, + statusLine: { type: "command", command: "omniroute status" }, + env: { KEEP_ME: "1", ANTHROPIC_BASE_URL: "http://old" }, + }, + null, + 2 + ) + ); + + const { stdout, stderr } = await execFileAsync( + process.execPath, + [ + cliPath, + "config", + "set", + "claude", + "--model", + "claude-fallback", + "--yes", + "--non-interactive", + "--allow-container-write", + ], + { + cwd: repoRoot, + env: { + ...process.env, + HOME: home, + USERPROFILE: home, + OMNIROUTE_API_KEY: "sk_test_12407", + OMNIROUTE_BASE_URL: "http://localhost:20128/v1", + }, + timeout: 30_000, + } + ); + + assert.match(stdout + stderr, /Config written/); + const written = JSON.parse(await fs.readFile(settingsPath, "utf8")); + + assert.equal(written.model, "claude-fallback"); + assert.equal(written.effortLevel, "high"); + assert.deepEqual(written.hooks, { PreToolUse: [{ command: "echo keep" }] }); + assert.deepEqual(written.statusLine, { type: "command", command: "omniroute status" }); + assert.equal(written.env.KEEP_ME, "1"); + assert.equal(written.env.ANTHROPIC_BASE_URL, "http://localhost:20128"); + assert.equal(written.env.ANTHROPIC_AUTH_TOKEN, "sk_test_12407"); + assert.equal(written.env.ANTHROPIC_MODEL, "claude-fallback"); + assert.equal(written.env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY, "1"); + assert.equal("baseUrl" in written, false); + assert.equal("authToken" in written, false); + assert.equal("models" in written, false); + } finally { + await fs.rm(home, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + } +}); From 11e1c79e6532245a19a64017305d469e5dccbf8f Mon Sep 17 00:00:00 2001 From: Krzysztof Skomra <159253490+KrzysiekSko@users.noreply.github.com> Date: Fri, 4 Sep 2026 04:27:15 +0200 Subject: [PATCH 076/143] fix(combos): clear LKGP pins on delete (#12425) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 6 PRs destas duas levas sobre o tip de `release/v3.8.51`: os seis boardaram **sem um único conflito**, `typecheck:core` limpo e **54/54** nos 7 arquivos de teste que trazem. O drift de `i18n:check` (`docs/security/GUARDRAILS.md`, `STEALTH_GUIDE.md` — source-changed) foi medido também no tip puro e é idêntico: base-red pré-existente, não desta leva. --- .../fixes/12326-combo-delete-lkgp-cleanup.md | 1 + .../db/repositories/sqliteComboRepository.ts | 24 ++- src/lib/db/settings.ts | 2 + src/lib/db/settings/lkgp.ts | 49 ++++++ .../combo-delete-lkgp-cleanup-12326.test.ts | 152 ++++++++++++++++++ 5 files changed, 226 insertions(+), 2 deletions(-) create mode 100644 changelog.d/fixes/12326-combo-delete-lkgp-cleanup.md create mode 100644 tests/unit/combo-delete-lkgp-cleanup-12326.test.ts diff --git a/changelog.d/fixes/12326-combo-delete-lkgp-cleanup.md b/changelog.d/fixes/12326-combo-delete-lkgp-cleanup.md new file mode 100644 index 0000000000..be5bc5678f --- /dev/null +++ b/changelog.d/fixes/12326-combo-delete-lkgp-cleanup.md @@ -0,0 +1 @@ +- **fix(combos):** deleting a combo now clears its persisted LKGP pins instead of leaving unreachable `key_value` rows behind ([#12326](https://github.com/diegosouzapw/OmniRoute/issues/12326)) diff --git a/src/lib/db/repositories/sqliteComboRepository.ts b/src/lib/db/repositories/sqliteComboRepository.ts index 4263f51034..1a623ea136 100644 --- a/src/lib/db/repositories/sqliteComboRepository.ts +++ b/src/lib/db/repositories/sqliteComboRepository.ts @@ -11,6 +11,7 @@ import type { import { normalizeComboRecord } from "@/lib/combos/steps"; import { validateComboInvariant } from "@/lib/combos/invariants"; import { getDbInstance } from "../core"; +import { deleteLKGPRowsByComboName } from "../settings/lkgp"; type JsonRecord = Record; @@ -349,8 +350,27 @@ export async function reorderCombos(comboIds: string[]): Promise { + const combo = db.prepare("SELECT name FROM combos WHERE id = ?").get(id) as + { name?: string } | undefined; + const result = db.prepare("DELETE FROM combos WHERE id = ?").run(id); + if (result.changes === 0) return { deleted: false, lkgpKeys: [] as string[] }; + return { + deleted: true, + lkgpKeys: combo?.name ? deleteLKGPRowsByComboName(combo.name) : ([] as string[]), + }; + }); + + const { deleted, lkgpKeys } = deleteTransaction(); + if (!deleted) return false; + + if (lkgpKeys.length > 0) { + const { invalidateCachedLKGP } = await import("../readCache"); + for (const key of lkgpKeys) { + invalidateCachedLKGP(key); + } + } + return true; } diff --git a/src/lib/db/settings.ts b/src/lib/db/settings.ts index 769c6d08ce..8b4aa803af 100644 --- a/src/lib/db/settings.ts +++ b/src/lib/db/settings.ts @@ -840,6 +840,8 @@ export { setLKGP, clearAllLKGP, clearLKGP, + deleteLKGPByComboName, + deleteLKGPRowsByComboName, deleteLKGPByConnectionIds, } from "./settings/lkgp"; diff --git a/src/lib/db/settings/lkgp.ts b/src/lib/db/settings/lkgp.ts index a57208bfeb..0100c5bd18 100644 --- a/src/lib/db/settings/lkgp.ts +++ b/src/lib/db/settings/lkgp.ts @@ -4,6 +4,34 @@ import { getDbInstance } from "../core"; +/** + * Escape SQLite `LIKE` wildcards so a combo name containing `%` or `_` cannot + * widen the prefix match into unrelated combos' pins. + */ +function escapeLikePattern(value: string): string { + return value.replace(/[\\%_]/g, (char) => `\\${char}`); +} + +export function deleteLKGPRowsByComboName(comboName: string): string[] { + if (!comboName) return []; + + const db = getDbInstance(); + const prefix = `${comboName}:`; + const rows = db + .prepare("SELECT key FROM key_value WHERE namespace = 'lkgp' AND key LIKE ? ESCAPE '\\'") + .all(`${escapeLikePattern(prefix)}%`) as Array<{ key?: string }>; + + const staleKeys = rows.map((row) => row?.key).filter((key): key is string => Boolean(key)); + if (staleKeys.length === 0) return []; + + const deleteStatement = db.prepare("DELETE FROM key_value WHERE namespace = 'lkgp' AND key = ?"); + for (const key of staleKeys) { + deleteStatement.run(key); + } + + return staleKeys; +} + export interface LKGPRecord { provider: string; connectionId?: string; @@ -67,6 +95,27 @@ export async function clearLKGP(comboName: string, modelId: string): Promise { + const staleKeys = deleteLKGPRowsByComboName(comboName); + + if (staleKeys.length === 0) return 0; + + const { invalidateCachedLKGP } = await import("../readCache"); + for (const key of staleKeys) { + invalidateCachedLKGP(key); + } + + return staleKeys.length; +} + /** * Delete persisted LKGP pins whose connectionId references a removed provider * connection (#8887). A pin persisted by `setLKGP()` carries the connection it diff --git a/tests/unit/combo-delete-lkgp-cleanup-12326.test.ts b/tests/unit/combo-delete-lkgp-cleanup-12326.test.ts new file mode 100644 index 0000000000..a35920535d --- /dev/null +++ b/tests/unit/combo-delete-lkgp-cleanup-12326.test.ts @@ -0,0 +1,152 @@ +/** + * Issue #12326 — deleting a combo must remove the LKGP pins keyed by its name + * without disturbing surviving combos' pins. + */ + +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-lkgp-12326-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const comboRepo = await import("../../src/lib/db/repositories/sqliteComboRepository.ts"); +const lkgpDb = await import("../../src/lib/db/settings/lkgp.ts"); +const readCache = await import("../../src/lib/db/readCache.ts"); + +async function resetStorage() { + core.resetDbInstance(); + + for (let attempt = 0; attempt < 10; attempt++) { + try { + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + break; + } catch (error: unknown) { + const code = + error && typeof error === "object" && "code" in error + ? String((error as { code?: unknown }).code) + : ""; + + if ((code === "EBUSY" || code === "EPERM") && attempt < 9) { + await new Promise((resolve) => setTimeout(resolve, 50 * (attempt + 1))); + continue; + } + + throw error; + } + } + + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +async function createCombo(name: string): Promise { + const combo = await comboRepo.createCombo({ + name, + models: [{ provider: "berry", model: "model-x" }], + } as Parameters[0]); + + assert.equal(typeof combo.id, "string", "combo fixture must return an id"); + return combo.id as string; +} + +test.beforeEach(async () => { + await resetStorage(); +}); + +test.after(() => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("#12326: deleting a combo removes its LKGP pins", async () => { + const doomedId = await createCombo("doomed-combo"); + await createCombo("survivor-combo"); + + await lkgpDb.setLKGP("doomed-combo", "model-x", "berry", "conn-1"); + await lkgpDb.setLKGP("doomed-combo", "model-y", "berry", "conn-2"); + await lkgpDb.setLKGP("survivor-combo", "model-x", "berry", "conn-3"); + + assert.equal(await comboRepo.deleteCombo(doomedId), true); + + assert.equal(await lkgpDb.getLKGP("doomed-combo", "model-x"), null); + assert.equal(await lkgpDb.getLKGP("doomed-combo", "model-y"), null); + assert.deepEqual(await lkgpDb.getLKGP("survivor-combo", "model-x"), { + provider: "berry", + connectionId: "conn-3", + }); +}); + +test("#12326: deleting a combo invalidates warmed LKGP read-cache entries", async () => { + const doomedId = await createCombo("cached-combo"); + + await lkgpDb.setLKGP("cached-combo", "model-x", "berry", "conn-1"); + + assert.deepEqual(await readCache.getCachedLKGP("cached-combo", "model-x"), { + provider: "berry", + connectionId: "conn-1", + }); + + assert.equal(await comboRepo.deleteCombo(doomedId), true); + + assert.equal( + await readCache.getCachedLKGP("cached-combo", "model-x"), + null, + "deleted combos' LKGP pins must not survive in the read cache" + ); +}); + +test("#12326: a combo whose name prefixes another keeps the sibling's pins", async () => { + const doomedId = await createCombo("prod"); + await createCombo("prod-canary"); + + await lkgpDb.setLKGP("prod", "model-x", "berry", "conn-1"); + await lkgpDb.setLKGP("prod-canary", "model-x", "berry", "conn-2"); + + assert.equal(await comboRepo.deleteCombo(doomedId), true); + + assert.equal(await lkgpDb.getLKGP("prod", "model-x"), null); + assert.deepEqual( + await lkgpDb.getLKGP("prod-canary", "model-x"), + { provider: "berry", connectionId: "conn-2" }, + "the ':' delimiter must keep a prefix-sharing sibling's pins intact" + ); +}); + +test("#12326: LIKE wildcards in a combo name do not widen the cleanup", async () => { + const doomedId = await createCombo("temp_a"); + await createCombo("tempXa"); + + await lkgpDb.setLKGP("temp_a", "model-x", "berry", "conn-1"); + await lkgpDb.setLKGP("tempXa", "model-x", "berry", "conn-2"); + + assert.equal(await comboRepo.deleteCombo(doomedId), true); + + assert.equal(await lkgpDb.getLKGP("temp_a", "model-x"), null); + assert.deepEqual( + await lkgpDb.getLKGP("tempXa", "model-x"), + { provider: "berry", connectionId: "conn-2" }, + "'_' must be escaped so it cannot match an arbitrary character" + ); +}); + +test("#12326: deleting an unknown combo id leaves LKGP state untouched", async () => { + await createCombo("untouched-combo"); + await lkgpDb.setLKGP("untouched-combo", "model-x", "berry", "conn-1"); + + assert.equal(await comboRepo.deleteCombo("00000000-0000-0000-0000-000000000000"), false); + + assert.deepEqual(await lkgpDb.getLKGP("untouched-combo", "model-x"), { + provider: "berry", + connectionId: "conn-1", + }); +}); + +test("#12326: deleting a combo without pins succeeds", async () => { + const doomedId = await createCombo("no-pins-combo"); + + assert.equal(await comboRepo.deleteCombo(doomedId), true); + assert.equal(await lkgpDb.getLKGP("no-pins-combo", "model-x"), null); +}); From 9271a34ec1f01a7c2a841d8276daf5e4f79671b0 Mon Sep 17 00:00:00 2001 From: Krzysztof Skomra <159253490+KrzysiekSko@users.noreply.github.com> Date: Fri, 4 Sep 2026 04:27:33 +0200 Subject: [PATCH 077/143] fix(api): preserve API key ACL on creation (#12352) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 6 PRs destas duas levas sobre o tip de `release/v3.8.51`: os seis boardaram **sem um único conflito**, `typecheck:core` limpo e **54/54** nos 7 arquivos de teste que trazem. O drift de `i18n:check` (`docs/security/GUARDRAILS.md`, `STEALTH_GUIDE.md` — source-changed) foi medido também no tip puro e é idêntico: base-red pré-existente, não desta leva. --- src/app/api/keys/route.ts | 13 ++++- src/lib/db/apiKeys.ts | 83 ++++++++++++++++----------- src/shared/validation/schemas/keys.ts | 24 +++++++- tests/integration/api-keys.test.ts | 70 ++++++++++++++++++++++ 4 files changed, 154 insertions(+), 36 deletions(-) diff --git a/src/app/api/keys/route.ts b/src/app/api/keys/route.ts index 11f4b94b97..51016b9f52 100644 --- a/src/app/api/keys/route.ts +++ b/src/app/api/keys/route.ts @@ -71,6 +71,9 @@ export async function POST(request) { } const { name, + modelAccessMode, + allowedModels, + allowedCombos, noLog, scopes, allowedConnections, @@ -84,7 +87,12 @@ export async function POST(request) { // Always get machineId from server const machineId = await getConsistentMachineId(); const normalizedScopes = normalizeSelfServiceScopesForCreate(scopes); - const apiKey = await createApiKey(name, machineId, normalizedScopes, { allowedConnections }); + const apiKey = await createApiKey(name, machineId, normalizedScopes, { + modelAccessMode, + allowedModels, + allowedCombos, + allowedConnections, + }); if ( noLog === true || allowUsageCommand === true || @@ -119,6 +127,9 @@ export async function POST(request) { name: apiKey.name, id: apiKey.id, machineId: apiKey.machineId, + modelAccessMode: apiKey.modelAccessMode, + allowedModels: apiKey.allowedModels, + allowedCombos: apiKey.allowedCombos, allowedConnections: apiKey.allowedConnections, noLog: noLog === true, allowUsageCommand: allowUsageCommand === true, diff --git a/src/lib/db/apiKeys.ts b/src/lib/db/apiKeys.ts index eeb1f70a0e..9ffa78159d 100644 --- a/src/lib/db/apiKeys.ts +++ b/src/lib/db/apiKeys.ts @@ -78,6 +78,13 @@ interface CacheEntry { value: TValue; } +interface CreateApiKeyOptions { + modelAccessMode?: ModelAccessMode; + allowedModels?: string[]; + allowedCombos?: string[]; + allowedConnections?: string[]; +} + export type { AccessSchedule, RateLimitRule } from "./apiKeys/types"; interface ApiKeyMetadata { @@ -233,9 +240,7 @@ function assertExclusiveLeaseKeyPolicy( allowedConnections: readonly string[] ): void { if (scopes.includes(EXCLUSIVE_LEASE_SCOPE) && allowedConnections.length === 0) { - throw new ApiKeyPolicyInvariantError( - "lease:exclusive requires explicit allowedConnections" - ); + throw new ApiKeyPolicyInvariantError("lease:exclusive requires explicit allowedConnections"); } } @@ -346,7 +351,7 @@ async function getModelPermissionCandidates(modelId: string): Promise providerOrAlias, providerScopedModel, resolveProviderId, - getProviderAlias, + getProviderAlias ); } return Array.from(candidates); @@ -364,7 +369,7 @@ async function getModelPermissionCandidates(modelId: string): Promise } async function getPublishedModelLookupTarget( - modelId: string, + modelId: string ): Promise<{ providerId: string; modelId: string } | null> { const cleanModelId = stripExtendedContextSuffix(modelId.trim()); if (!cleanModelId) return null; @@ -393,7 +398,7 @@ async function getPublishedModelLookupTarget( function ensureApiKeyColumn( db: ApiKeysDbLike, columnNames: Set, - column: (typeof API_KEY_COLUMN_FALLBACKS)[number], + column: (typeof API_KEY_COLUMN_FALLBACKS)[number] ): void { if (columnNames.has(column.name)) return; db.exec(`ALTER TABLE api_keys ADD COLUMN ${column.definition}`); @@ -433,13 +438,13 @@ function getPreparedStatements(db: ApiKeysDbLike): ApiKeysStatements { _stmtGetAllKeys = db.prepare("SELECT * FROM api_keys ORDER BY created_at"); _stmtGetKeyById = db.prepare("SELECT * FROM api_keys WHERE id = ?"); _stmtValidateKey = db.prepare( - "SELECT id, expires_at, revoked_at, is_active, is_banned FROM api_keys WHERE key = ? OR key_hash = ?", + "SELECT id, expires_at, revoked_at, is_active, is_banned FROM api_keys WHERE key = ? OR key_hash = ?" ); _stmtGetKeyMetadata = db.prepare( - "SELECT id, name, machine_id, model_access_mode, allowed_models, blocked_models, allowed_combos, allowed_connections, allowed_quotas, no_log, auto_resolve, is_active, access_schedule, max_requests_per_day, max_requests_per_minute, throttle_delay_ms, max_sessions, revoked_at, expires_at, ip_allowlist, scopes, rate_limits, is_banned, key_hash, allowed_endpoints, stream_default_mode, cache_default_mode, disable_non_public_models, allow_usage_command, usage_limit_enabled, daily_usage_limit_usd, weekly_usage_limit_usd, chaos_mode_enabled, compression_enabled, proxy_id FROM api_keys WHERE key = ? OR key_hash = ?", + "SELECT id, name, machine_id, model_access_mode, allowed_models, blocked_models, allowed_combos, allowed_connections, allowed_quotas, no_log, auto_resolve, is_active, access_schedule, max_requests_per_day, max_requests_per_minute, throttle_delay_ms, max_sessions, revoked_at, expires_at, ip_allowlist, scopes, rate_limits, is_banned, key_hash, allowed_endpoints, stream_default_mode, cache_default_mode, disable_non_public_models, allow_usage_command, usage_limit_enabled, daily_usage_limit_usd, weekly_usage_limit_usd, chaos_mode_enabled, compression_enabled, proxy_id FROM api_keys WHERE key = ? OR key_hash = ?" ); _stmtInsertKey = db.prepare( - "INSERT INTO api_keys (id, name, key, machine_id, allowed_models, allowed_combos, allowed_connections, no_log, created_at, key_prefix, key_hash, scopes) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)" + "INSERT INTO api_keys (id, name, key, machine_id, model_access_mode, allowed_models, allowed_combos, allowed_connections, no_log, created_at, key_prefix, key_hash, scopes) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)" ); _stmtDeleteKey = db.prepare("DELETE FROM api_keys WHERE id = ?"); } @@ -497,7 +502,7 @@ export async function getApiKeys(limit?: number, offset?: number) { camelRow.streamDefaultMode = parseStreamDefaultMode((camelRow as JsonRecord).streamDefaultMode); camelRow.cacheDefaultMode = parseCacheDefaultMode((camelRow as JsonRecord).cacheDefaultMode); camelRow.disableNonPublicModels = parseDisableNonPublicModels( - (camelRow as JsonRecord).disableNonPublicModels, + (camelRow as JsonRecord).disableNonPublicModels ); camelRow.allowUsageCommand = parseAllowUsageCommand((camelRow as JsonRecord).allowUsageCommand); camelRow.chaosModeEnabled = parseChaosModeEnabled((camelRow as JsonRecord).chaosModeEnabled); @@ -556,7 +561,7 @@ export async function getExclusiveLeaseConnectionIds(): Promise> { * inactive, banned, or hard-lease key, and it never widens a key's allowedModels. */ export async function pickApiKeyForInternalUse( - purpose: "combo-health-check" | "cloud-sync-verify" | "internal-probe" = "internal-probe", + purpose: "combo-health-check" | "cloud-sync-verify" | "internal-probe" = "internal-probe" ): Promise { try { const keys = (await getApiKeys()) as Array<{ @@ -579,7 +584,7 @@ export async function pickApiKeyForInternalUse( // 1. Management-scoped key (preferred for any internal probe). const manageKey = keys.find( - (k) => isUsable(k) && Array.isArray(k.scopes) && k.scopes.includes("manage"), + (k) => isUsable(k) && Array.isArray(k.scopes) && k.scopes.includes("manage") ); if (manageKey?.key) return manageKey.key; @@ -634,7 +639,7 @@ export async function getApiKeyById(id: string) { camelRow.streamDefaultMode = parseStreamDefaultMode((camelRow as JsonRecord).streamDefaultMode); camelRow.cacheDefaultMode = parseCacheDefaultMode((camelRow as JsonRecord).cacheDefaultMode); camelRow.disableNonPublicModels = parseDisableNonPublicModels( - (camelRow as JsonRecord).disableNonPublicModels, + (camelRow as JsonRecord).disableNonPublicModels ); camelRow.allowUsageCommand = parseAllowUsageCommand((camelRow as JsonRecord).allowUsageCommand); camelRow.chaosModeEnabled = parseChaosModeEnabled((camelRow as JsonRecord).chaosModeEnabled); @@ -662,12 +667,19 @@ export async function createApiKey( name: string, machineId: string, scopes: string[] = [], - options: { allowedConnections?: string[] } = {} + options: CreateApiKeyOptions = {} ) { if (!machineId) { throw new Error("machineId is required"); } const allowedConnections = options.allowedConnections ?? []; + const modelAccess = normalizeApiKeyPermissionsUpdate({ + modelAccessMode: options.modelAccessMode, + allowedModels: options.allowedModels, + }); + const modelAccessMode = modelAccess.modelAccessMode ?? "all"; + const allowedModels = modelAccess.allowedModels ?? []; + const allowedCombos = options.allowedCombos ?? [ALL_COMBOS_ACCESS_RULE]; assertExclusiveLeaseKeyPolicy(scopes, allowedConnections); const db = getDbInstance() as ApiKeysDbLike; @@ -681,9 +693,9 @@ export async function createApiKey( name: name, key: result.key, machineId: machineId, - modelAccessMode: "all" as const, - allowedModels: [], // Empty array means all models allowed - allowedCombos: [ALL_COMBOS_ACCESS_RULE], // Explicit wildcard means all combos allowed + modelAccessMode, + allowedModels, + allowedCombos, allowedConnections, noLog: false, allowUsageCommand: false, @@ -697,14 +709,15 @@ export async function createApiKey( apiKey.name, apiKey.key, apiKey.machineId, - "[]", + apiKey.modelAccessMode, + JSON.stringify(apiKey.allowedModels), JSON.stringify(apiKey.allowedCombos), JSON.stringify(allowedConnections), 0, apiKey.createdAt, apiKey.key.slice(0, 12), await hashKey(apiKey.key), - JSON.stringify(scopes), + JSON.stringify(scopes) ); setNoLog(apiKey.id, false); @@ -726,7 +739,7 @@ export async function regenerateApiKey(id: string) { // Update in DB const updateStmt = db.prepare( - "UPDATE api_keys SET key = ?, key_hash = ?, key_prefix = ? WHERE id = ?", + "UPDATE api_keys SET key = ?, key_hash = ?, key_prefix = ? WHERE id = ?" ); updateStmt.run(newKey, newHash, newPrefix, id); @@ -747,7 +760,7 @@ export async function regenerateApiKey(id: string) { export async function updateApiKeyPermissions( id: string, - update: string[] | ApiKeyPermissionsUpdate, + update: string[] | ApiKeyPermissionsUpdate ) { const db = getDbInstance() as ApiKeysDbLike; getPreparedStatements(db); @@ -1050,7 +1063,9 @@ export async function updateApiKeyPermissions( return false; } assertExclusiveLeaseKeyPolicy(parseStringList(row.scopes), normalized.allowedConnections); - const upd = db.prepare(`UPDATE api_keys SET ${updates.join(", ")} WHERE id = @id`).run(params); + const upd = db + .prepare(`UPDATE api_keys SET ${updates.join(", ")} WHERE id = @id`) + .run(params); changedRows = upd.changes ?? 0; db.exec("COMMIT"); } catch (err) { @@ -1162,7 +1177,7 @@ export async function revokeApiKey(id: string): Promise { const result = db .prepare( - "UPDATE api_keys SET revoked_at = COALESCE(revoked_at, @ts), is_active = 0 WHERE id = @id", + "UPDATE api_keys SET revoked_at = COALESCE(revoked_at, @ts), is_active = 0 WHERE id = @id" ) .run({ id, ts: new Date().toISOString() }); @@ -1291,7 +1306,7 @@ export async function validateApiKey(key: string | null | undefined) { revokedAt: row.revoked_at, }), "EX", - 3600, // 1 hour cache + 3600 // 1 hour cache ); } } catch { @@ -1308,7 +1323,7 @@ export async function validateApiKey(key: string | null | undefined) { * Get API key metadata with caching for performance */ export async function getApiKeyMetadata( - key: string | null | undefined, + key: string | null | undefined ): Promise { if (!key || typeof key !== "string") return null; @@ -1415,10 +1430,10 @@ export async function getApiKeyMetadata( blockedModels: parseAllowedModels(record.blocked_models ?? record.blockedModels), allowedCombos: parseAllowedCombos(record.allowed_combos ?? record.allowedCombos), allowedConnections: parseAllowedConnections( - record.allowed_connections ?? record.allowedConnections, + record.allowed_connections ?? record.allowedConnections ), allowedQuotas: parseAllowedQuotas( - (record as JsonRecord).allowed_quotas ?? (record as JsonRecord).allowedQuotas, + (record as JsonRecord).allowed_quotas ?? (record as JsonRecord).allowedQuotas ), noLog: parseNoLog(record.no_log ?? record.noLog), autoResolve: parseAutoResolve(record.auto_resolve ?? record.autoResolve), @@ -1440,26 +1455,26 @@ export async function getApiKeyMetadata( proxyId: typeof record.proxy_id === "string" && record.proxy_id.trim() !== "" ? record.proxy_id : null, allowedEndpoints: parseStringList( - (record as JsonRecord).allowed_endpoints ?? (record as JsonRecord).allowedEndpoints, + (record as JsonRecord).allowed_endpoints ?? (record as JsonRecord).allowedEndpoints ), streamDefaultMode: parseStreamDefaultMode( - (record as JsonRecord).stream_default_mode ?? (record as JsonRecord).streamDefaultMode, + (record as JsonRecord).stream_default_mode ?? (record as JsonRecord).streamDefaultMode ), cacheDefaultMode: parseCacheDefaultMode( (record as JsonRecord).cache_default_mode ?? (record as JsonRecord).cacheDefaultMode ), disableNonPublicModels: parseDisableNonPublicModels( (record as JsonRecord).disable_non_public_models ?? - (record as JsonRecord).disableNonPublicModels, + (record as JsonRecord).disableNonPublicModels ), allowUsageCommand: parseAllowUsageCommand( - (record as JsonRecord).allow_usage_command ?? (record as JsonRecord).allowUsageCommand, + (record as JsonRecord).allow_usage_command ?? (record as JsonRecord).allowUsageCommand ), chaosModeEnabled: parseChaosModeEnabled( - (record as JsonRecord).chaos_mode_enabled ?? (record as JsonRecord).chaosModeEnabled, + (record as JsonRecord).chaos_mode_enabled ?? (record as JsonRecord).chaosModeEnabled ), compressionEnabled: parseCompressionEnabled( - (record as JsonRecord).compression_enabled ?? (record as JsonRecord).compressionEnabled, + (record as JsonRecord).compression_enabled ?? (record as JsonRecord).compressionEnabled ), ...parseApiKeyUsageLimitFields(record as JsonRecord), }; @@ -1485,7 +1500,7 @@ export async function getApiKeyMetadata( */ export async function isModelAllowedForKey( key: string | null | undefined, - modelId: string | null | undefined, + modelId: string | null | undefined ) { // If no key provided, allow (request may be using different auth method like JWT) // If no modelId provided, deny (invalid request) diff --git a/src/shared/validation/schemas/keys.ts b/src/shared/validation/schemas/keys.ts index 37741f56a9..1441c8851f 100644 --- a/src/shared/validation/schemas/keys.ts +++ b/src/shared/validation/schemas/keys.ts @@ -33,9 +33,28 @@ const requireExclusiveLeaseConnections = ( }); }; +const requireConsistentModelAccess = ( + value: { + modelAccessMode?: "all" | "restricted"; + allowedModels?: string[]; + }, + ctx: z.RefinementCtx +) => { + if (value.modelAccessMode === "all" && value.allowedModels && value.allowedModels.length > 0) { + ctx.addIssue({ + code: z.ZodIssueCode.custom, + message: "allowedModels must be empty when modelAccessMode is 'all'", + path: ["allowedModels"], + }); + } +}; + export const createKeySchema = z .object({ name: z.string().min(1, "Name is required").max(200), + modelAccessMode: z.enum(["all", "restricted"]).optional(), + allowedModels: z.array(z.string().trim().min(1)).max(1000).optional(), + allowedCombos: z.array(z.string().trim().min(1).max(200)).max(500).optional(), noLog: z.boolean().optional(), allowUsageCommand: z.boolean().optional(), usageLimitEnabled: z.boolean().optional(), @@ -45,7 +64,10 @@ export const createKeySchema = z scopes: z.array(z.string().trim().min(1).max(64)).max(32).optional(), allowedConnections: z.array(z.string().uuid()).min(1).max(100).optional(), }) - .superRefine(requireExclusiveLeaseConnections); + .superRefine((value, ctx) => { + requireConsistentModelAccess(value, ctx); + requireExclusiveLeaseConnections(value, ctx); + }); export const createSyncTokenSchema = z.object({ name: z.string().trim().min(1, "Name is required").max(200), diff --git a/tests/integration/api-keys.test.ts b/tests/integration/api-keys.test.ts index 4e82c15874..b394eee5da 100644 --- a/tests/integration/api-keys.test.ts +++ b/tests/integration/api-keys.test.ts @@ -12,6 +12,8 @@ process.env.CLOUD_URL = "http://cloud.example"; const core = await import("../../src/lib/db/core.ts"); const apiKeysDb = await import("../../src/lib/db/apiKeys.ts"); +const combosDb = await import("../../src/lib/db/combos.ts"); +const apiKeyPolicy = await import("../../src/shared/utils/apiKeyPolicy.ts"); const { updateSettings } = await import("@/lib/db/settings"); const localDb = { updateSettings }; const compliance = await import("../../src/lib/compliance/index.ts"); @@ -134,6 +136,74 @@ test("POST /api/keys creates a key, preserves special characters, and persists n assert.equal(compliance.isNoLog(body.id), true); }); +test("POST /api/keys preserves creation-time ACL and enforces it (#12275)", async () => { + await enableManagementAuth(); + await createManagementKey(); + + const response = await listRoute.POST( + await makeManagementSessionRequest("http://localhost/api/keys", { + method: "POST", + body: { + name: "Restricted Create", + modelAccessMode: "restricted", + allowedModels: ["openai/gpt-4.1-mini"], + allowedCombos: ["focused-chat"], + }, + }) + ); + const body = (await response.json()) as { + id: string; + key: string; + modelAccessMode: string; + allowedModels: string[]; + allowedCombos: string[]; + }; + const stored = await apiKeysDb.getApiKeyById(body.id); + + await combosDb.createCombo({ + name: "focused-chat", + strategy: "priority", + config: { maxRetries: 0, retryDelayMs: 0 }, + models: ["openai/gpt-4.1-mini"], + }); + await combosDb.createCombo({ + name: "blocked-chat", + strategy: "priority", + config: { maxRetries: 0, retryDelayMs: 0 }, + models: ["anthropic/claude-sonnet-4-5"], + }); + + assert.equal(response.status, 201); + assert.equal(body.modelAccessMode, "restricted"); + assert.deepEqual(body.allowedModels, ["openai/gpt-4.1-mini"]); + assert.deepEqual(body.allowedCombos, ["focused-chat"]); + assert.equal(stored?.modelAccessMode, "restricted"); + assert.deepEqual(stored?.allowedModels, ["openai/gpt-4.1-mini"]); + assert.deepEqual(stored?.allowedCombos, ["focused-chat"]); + + const allowed = await apiKeyPolicy.enforceApiKeyPolicy( + makeRequest("http://localhost/api/v1/chat/completions", { token: body.key }), + "openai/gpt-4.1-mini" + ); + const denied = await apiKeyPolicy.enforceApiKeyPolicy( + makeRequest("http://localhost/api/v1/chat/completions", { token: body.key }), + "anthropic/claude-sonnet-4-5" + ); + const allowedCombo = await apiKeyPolicy.enforceApiKeyPolicy( + makeRequest("http://localhost/api/v1/chat/completions", { token: body.key }), + "focused-chat" + ); + const deniedCombo = await apiKeyPolicy.enforceApiKeyPolicy( + makeRequest("http://localhost/api/v1/chat/completions", { token: body.key }), + "blocked-chat" + ); + + assert.equal(allowed.rejection, null); + assert.equal(denied.rejection?.status, 403); + assert.equal(allowedCombo.rejection, null); + assert.equal(deniedCombo.rejection?.status, 403); +}); + test("POST /api/keys validates missing and oversized names", async () => { await enableManagementAuth(); await createManagementKey(); From 9cbc4f118e98c5b481793ab4611bc5bf759a8041 Mon Sep 17 00:00:00 2001 From: Goni Sulaiman <93048973+gonisulaimann@users.noreply.github.com> Date: Fri, 4 Sep 2026 03:27:55 +0100 Subject: [PATCH 078/143] fix(models): publish effort_tiers on Kimi K3 base models only (#12299) (#12371) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 6 PRs destas duas levas sobre o tip de `release/v3.8.51`: os seis boardaram **sem um único conflito**, `typecheck:core` limpo e **54/54** nos 7 arquivos de teste que trazem. O drift de `i18n:check` (`docs/security/GUARDRAILS.md`, `STEALTH_GUIDE.md` — source-changed) foi medido também no tip puro e é idêntico: base-red pré-existente, não desta leva. --- .../fixes/12371-kimi-k3-effort-tiers.md | 1 + src/app/api/v1/models/syncedCapabilities.ts | 30 ++++- tests/unit/kimi-k3-effort-tiers-12299.test.ts | 103 ++++++++++++++++++ ...pabilities-learned-effort-override.test.ts | 3 + 4 files changed, 136 insertions(+), 1 deletion(-) create mode 100644 changelog.d/fixes/12371-kimi-k3-effort-tiers.md create mode 100644 tests/unit/kimi-k3-effort-tiers-12299.test.ts diff --git a/changelog.d/fixes/12371-kimi-k3-effort-tiers.md b/changelog.d/fixes/12371-kimi-k3-effort-tiers.md new file mode 100644 index 0000000000..49a3310761 --- /dev/null +++ b/changelog.d/fixes/12371-kimi-k3-effort-tiers.md @@ -0,0 +1 @@ +- **fix(models):** publish `effort_tiers` on Kimi K3's synced base-model entries (`kmca/k3`, `kmca/k3-256k`) so catalog-only clients (OpenCode, plain SDK pickers) can see and select the reasoning tiers (`low`/`high`/`max`) the synced metadata already carried — the `isSkippedEffortProvider` gate no longer suppresses tier visibility on those base entries, while synthetic `-` variant generation stays prevented and Codex/GLM base models remain excluded unchanged ([#12299](https://github.com/diegosouzapw/OmniRoute/issues/12299)) diff --git a/src/app/api/v1/models/syncedCapabilities.ts b/src/app/api/v1/models/syncedCapabilities.ts index 529753a8d9..aac6f9fd27 100644 --- a/src/app/api/v1/models/syncedCapabilities.ts +++ b/src/app/api/v1/models/syncedCapabilities.ts @@ -21,6 +21,12 @@ * `-` catalog entries (open-sse/utils/syncedEffortVariants.ts) — it * never runs over the base entry's `capabilities`, so it cannot substitute * for this check. Required (not optional) so no call site can silently skip it. + * + * #12299 carve-out: Kimi K3's synced base entries (`k3`, `k3-256k` — the kmca + * catalog's `low`/`high`/`max` vocabulary) are exempted from the exclusion so + * catalog-only clients (OpenCode, plain SDK pickers) can see and select their + * tiers. Model-scoped, never provider-wide: Codex, GLM, and non-K3 kimi models + * keep the full exclusion exactly as before this carve-out. */ // Use the same canonical alias as catalogModelPolicy.ts (l.1) — a relative path from // src/app/api/v1/models/ to open-sse/ would need 5 `../` and silently breaks under @@ -39,8 +45,30 @@ interface SyncedCapabilityFlags { supportedThinkingEfforts?: string[]; } +// Model-id pattern for the Kimi K3 family (#12299): the kmca catalog syncs +// `k3`/`k3-256k` (and prefixed forms such as `kmca/k3`). Same shape the +// executor/translator layers use to recognize K3 elsewhere +// (reasoningContentInjector.ts::K3_AUTHENTIC_REASONING_PATTERN). +const KIMI_K3_MODEL_ID_PATTERN = /(?:^|\/)(?:kimi-)?k3(?:$|-)/i; + +/** + * #12299: only Kimi K3's synced BASE entries are exempt from the + * `isSkippedEffortProvider` exclusion. Model-scoped, never provider-wide — + * the exemption requires a kimi-owned provider AND a K3 model id, so Codex, + * GLM, and non-K3 kimi models keep the exclusion contract from #7694. + */ +function isExemptKimiK3BaseModel(sm: SyncedCapabilityFlags, ownedBy: string): boolean { + return ( + ownedBy.startsWith("kimi") && typeof sm.id === "string" && KIMI_K3_MODEL_ID_PATTERN.test(sm.id) + ); +} + function effectiveEffortTiers(sm: SyncedCapabilityFlags, ownedBy: string): string[] | undefined { - if (isSkippedEffortProvider(ownedBy)) return undefined; + // Exclusion gate (#7694): codex/glm/kimi own a conflicting `-{effort}` suffix + // mechanism — the blind opencode-plugin mapping must never see effort_tiers + // for them, or it double-handles the suffix. #12299 narrows only the kimi K3 + // base-model entries out of that gate; everything else stays excluded. + if (isSkippedEffortProvider(ownedBy) && !isExemptKimiK3BaseModel(sm, ownedBy)) return undefined; const learned = sm.id ? getLearnedReasoningEffortForModel(sm.id) : null; const synced = Array.isArray(sm.supportedThinkingEfforts) && sm.supportedThinkingEfforts.length > 0 diff --git a/tests/unit/kimi-k3-effort-tiers-12299.test.ts b/tests/unit/kimi-k3-effort-tiers-12299.test.ts new file mode 100644 index 0000000000..d4e0619c6b --- /dev/null +++ b/tests/unit/kimi-k3-effort-tiers-12299.test.ts @@ -0,0 +1,103 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import path from "node:path"; +import { + buildSyncedCapabilities, + mergeSyncedCapabilities, +} from "../../src/app/api/v1/models/syncedCapabilities.ts"; +import { + shouldExposeSyncedEffortVariants, + appendSyncedEffortVariants, +} from "../../open-sse/utils/syncedEffortVariants.ts"; + +// #12299: Kimi K3's supportedThinkingEfforts (["low", "high", "max"]) were +// suppressed on the BASE model by isSkippedEffortProvider in +// effectiveEffortTiers(), leaving catalog-only clients with no tiers to copy. +// Fix: publish effort_tiers on the base model while still preventing +// synthetic - variant generation for kimi providers. + +const KIMI_K3_TIERS = ["low", "high", "max"]; + +test("Kimi K3 base model publishes effort_tiers via buildSyncedCapabilities (#12299)", () => { + const caps = buildSyncedCapabilities( + { id: "k3", supportsThinking: true, supportedThinkingEfforts: KIMI_K3_TIERS }, + "kimi-coding-apikey" + ); + assert.ok(caps, "capabilities must be defined for kimi K3"); + assert.deepEqual( + caps.effort_tiers, + KIMI_K3_TIERS, + "kimi K3 base model must publish effort_tiers low/high/max" + ); +}); + +test("Kimi K3-256k base model publishes effort_tiers via buildSyncedCapabilities (#12299)", () => { + const caps = buildSyncedCapabilities( + { id: "k3-256k", supportsThinking: true, supportedThinkingEfforts: KIMI_K3_TIERS }, + "kimi-coding-apikey" + ); + assert.ok(caps, "capabilities must be defined for kimi K3-256k"); + assert.deepEqual( + caps.effort_tiers, + KIMI_K3_TIERS, + "kimi K3-256k base model must publish effort_tiers low/high/max" + ); +}); + +test("Kimi K3 merge path also publishes effort_tiers (#12299)", () => { + const merged = mergeSyncedCapabilities( + { tool_calling: true }, + { id: "k3", supportsThinking: true, supportedThinkingEfforts: KIMI_K3_TIERS }, + "kimi-coding-apikey" + ); + assert.ok(merged, "merged capabilities must be defined"); + assert.deepEqual( + merged.effort_tiers, + KIMI_K3_TIERS, + "merge path must publish kimi K3 effort_tiers" + ); + assert.equal(merged.tool_calling, true, "existing tool_calling must be preserved"); +}); + +test("shouldExposeSyncedEffortVariants still prevents synthetic kimi variants", () => { + // The base model should NOT generate synthetic - entries + assert.equal( + shouldExposeSyncedEffortVariants({ + id: "kimi/k3", + owned_by: "kimi-coding-apikey", + capabilities: { effort_tiers: KIMI_K3_TIERS }, + }), + false, + "must not generate synthetic kimi/k3-low, kimi/k3-high, etc." + ); +}); + +test("appendSyncedEffortVariants does not create kimi variant entries", () => { + const models = [ + { + id: "kimi-coding-apikey/k3", + owned_by: "kimi-coding-apikey", + capabilities: { effort_tiers: KIMI_K3_TIERS }, + }, + ]; + const result = appendSyncedEffortVariants(models); + assert.equal(result.length, 1, "must not add synthetic variant entries for kimi"); + assert.equal(result[0].id, "kimi-coding-apikey/k3", "original entry must be unchanged"); +}); + +test("kimi K3 static registry tiers match synced metadata", () => { + // Verify the static registry in runtime.ts has the correct tiers + const runtimePath = path.join( + path.dirname(fileURLToPath(import.meta.url)), + "../../open-sse/config/providers/registry/kimi/coding/runtime.ts" + ); + const content = readFileSync(runtimePath, "utf8"); + + // Verify the static thinking policies declare the same tiers + assert.ok( + content.includes('"low", "high", "max"'), + "KIMI_CODE_STATIC_THINKING_POLICIES.k3 must declare low/high/max" + ); +}); diff --git a/tests/unit/synced-capabilities-learned-effort-override.test.ts b/tests/unit/synced-capabilities-learned-effort-override.test.ts index f33d08dc56..16f124b05b 100644 --- a/tests/unit/synced-capabilities-learned-effort-override.test.ts +++ b/tests/unit/synced-capabilities-learned-effort-override.test.ts @@ -62,6 +62,9 @@ test("merge path keeps vision AND applies the learned override", () => { // Exclusion gate (#7694): codex/glm/kimi already own a conflicting // `-{effort}` suffix mechanism — the blind opencode-plugin mapping must never // see effort_tiers for them, learned or synced, or it double-handles the suffix. +// #12299 exempts only Kimi K3's BASE model entries (asserted in +// tests/unit/kimi-k3-effort-tiers-12299.test.ts) — non-K3 kimi models such as +// "excluded-model" below stay excluded alongside codex/glm. for (const ownedBy of ["codex", "glm", "glm-cn", "glmt", "kimi", "kimi-coding-apikey"]) { test(`build: excluded provider "${ownedBy}" never gets effort_tiers (synced)`, () => { const caps = buildSyncedCapabilities( From 6e35ad01cca70bf4407aba7216210564c6368049 Mon Sep 17 00:00:00 2001 From: Goni Sulaiman <93048973+gonisulaimann@users.noreply.github.com> Date: Fri, 4 Sep 2026 03:28:12 +0100 Subject: [PATCH 079/143] fix(cli): remove duplicate positional argument in tunnel create command (#12368) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 6 PRs destas duas levas sobre o tip de `release/v3.8.51`: os seis boardaram **sem um único conflito**, `typecheck:core` limpo e **54/54** nos 7 arquivos de teste que trazem. O drift de `i18n:check` (`docs/security/GUARDRAILS.md`, `STEALTH_GUIDE.md` — source-changed) foi medido também no tip puro e é idêntico: base-red pré-existente, não desta leva. --- bin/cli/commands/tunnel.mjs | 2 +- changelog.d/fixes/0000-tunnel-create-crash.md | 1 + .../fixes/12368-tunnel-create-crash.md | 1 + tests/unit/cli-tunnel-create-command.test.ts | 90 +++++++++++++++++++ 4 files changed, 93 insertions(+), 1 deletion(-) create mode 100644 changelog.d/fixes/0000-tunnel-create-crash.md create mode 100644 changelog.d/fixes/12368-tunnel-create-crash.md create mode 100644 tests/unit/cli-tunnel-create-command.test.ts diff --git a/bin/cli/commands/tunnel.mjs b/bin/cli/commands/tunnel.mjs index df07507a66..689a8a7f6b 100644 --- a/bin/cli/commands/tunnel.mjs +++ b/bin/cli/commands/tunnel.mjs @@ -18,7 +18,7 @@ export function registerTunnel(program) { }); tunnel - .command("create [type]") + .command("create") .description(t("tunnel.createDescription")) .addArgument( new Argument("[type]", "Tunnel type").choices(VALID_TUNNEL_TYPES).default("cloudflare") diff --git a/changelog.d/fixes/0000-tunnel-create-crash.md b/changelog.d/fixes/0000-tunnel-create-crash.md new file mode 100644 index 0000000000..5fbfd488c8 --- /dev/null +++ b/changelog.d/fixes/0000-tunnel-create-crash.md @@ -0,0 +1 @@ +- **fix(cli):** `omniroute tunnel create` no longer crashes with `Cannot read properties of undefined (reading optsWithGlobals)` — removed the duplicate positional argument that caused Commander.js to misalign the action callback parameters ([#12295](https://github.com/diegosouzapw/OmniRoute/issues/12295)) diff --git a/changelog.d/fixes/12368-tunnel-create-crash.md b/changelog.d/fixes/12368-tunnel-create-crash.md new file mode 100644 index 0000000000..a1d71d3abb --- /dev/null +++ b/changelog.d/fixes/12368-tunnel-create-crash.md @@ -0,0 +1 @@ +- **fix(cli):** `omniroute tunnel create` no longer crashes with `Cannot read properties of undefined (reading optsWithGlobals)` — removed the duplicate positional argument that caused Commander.js to misalign the action callback parameters ([#12295](https://github.com/diegosouzapw/OmniRoute/issues/12295), [#12368](https://github.com/diegosouzapw/OmniRoute/pull/12368)) diff --git a/tests/unit/cli-tunnel-create-command.test.ts b/tests/unit/cli-tunnel-create-command.test.ts new file mode 100644 index 0000000000..2e733ae063 --- /dev/null +++ b/tests/unit/cli-tunnel-create-command.test.ts @@ -0,0 +1,90 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +// #12295: `omniroute tunnel create ` crashed with +// "Cannot read properties of undefined (reading optsWithGlobals)" because +// `.command("create [type]")` + `.addArgument(new Argument("[type]", ...))` +// registered TWO positional arguments. Commander then passed +// (type, type2, opts, command) to the action callback, but the handler +// destructured only (type, opts, cmd) — so `cmd` was bound to the real opts +// object and `cmd.parent` was undefined. + +test("tunnel create subcommand declares exactly one positional argument (#12295)", async () => { + const { Command } = await import("commander"); + const { registerTunnel } = await import("../../bin/cli/commands/tunnel.mjs"); + + const program = new Command(); + registerTunnel(program); + + const tunnelCmd = program.commands.find((c) => c.name() === "tunnel"); + assert.ok(tunnelCmd, "tunnel subcommand must exist"); + + const createCmd = tunnelCmd.commands.find((c) => c.name() === "create"); + assert.ok(createCmd, "create subcommand must exist"); + + // Before the fix, Commander registered two arguments named "type" because + // both .command("create [type]") and .addArgument(...) contributed one. + // After the fix, only .addArgument(...) defines the positional. + // Commander stores positional arguments in _args; the public args getter + // returns only required args, so optional args (like [type]) only appear in _args. + assert.equal( + createCmd._args.length, + 1, + `create subcommand must have exactly 1 positional argument, got ${createCmd._args.length}` + ); +}); + +test("tunnel create subcommand action handler accesses parent via Command instance (#12295)", async () => { + const { Command } = await import("commander"); + const { registerTunnel } = await import("../../bin/cli/commands/tunnel.mjs"); + + const program = new Command(); + registerTunnel(program); + + const tunnelCmd = program.commands.find((c) => c.name() === "tunnel"); + const createCmd = tunnelCmd.commands.find((c) => c.name() === "create"); + + assert.ok(createCmd._actionHandler, "create subcommand must have an action handler"); + + // Before the fix, the action callback received (type, type2, opts, command) + // because of the double positional. Destructuring (type, opts, cmd) then + // bound cmd to the opts object, making cmd.parent undefined and + // cmd.parent.optsWithGlobals() throw. + // After the fix, there's only one positional, so (type, opts, cmd) correctly + // binds cmd to the Command instance where cmd.parent === tunnelCmd. + // We can verify this by checking the parent chain on the createCmd itself: + assert.equal( + createCmd.parent, + tunnelCmd, + "create subcommand's parent must be the tunnel command" + ); + assert.equal( + createCmd.parent.parent, + program, + "tunnel command's parent must be the root program" + ); +}); + +test("tunnel create subcommand accepts valid tunnel type choices", async () => { + const { Command } = await import("commander"); + const { registerTunnel } = await import("../../bin/cli/commands/tunnel.mjs"); + + const program = new Command(); + registerTunnel(program); + + const tunnelCmd = program.commands.find((c) => c.name() === "tunnel"); + const createCmd = tunnelCmd.commands.find((c) => c.name() === "create"); + + // The addArgument with choices should still be registered. + assert.equal(createCmd._args.length, 1, "must have exactly one positional after addArgument"); + + // Verify choices are present on the argument. + const arg = createCmd._args[0]; + assert.ok(arg, "first argument must exist"); + assert.deepEqual( + arg.argChoices, + ["cloudflare", "tailscale", "ngrok"], + "argument choices must match VALID_TUNNEL_TYPES" + ); + assert.equal(arg.defaultValue, "cloudflare", "default type must be cloudflare"); +}); From 57d9357d88102144eea09c5039bd111bad073cee Mon Sep 17 00:00:00 2001 From: Goni Sulaiman <93048973+gonisulaimann@users.noreply.github.com> Date: Fri, 4 Sep 2026 03:28:33 +0100 Subject: [PATCH 080/143] fix(i18n): wrap ccOnboardingKeyPlaceholder in ICU single quotes across all 43 locales (#12369) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 6 PRs destas duas levas sobre o tip de `release/v3.8.51`: os seis boardaram **sem um único conflito**, `typecheck:core` limpo e **54/54** nos 7 arquivos de teste que trazem. O drift de `i18n:check` (`docs/security/GUARDRAILS.md`, `STEALTH_GUIDE.md` — source-changed) foi medido também no tip puro e é idêntico: base-red pré-existente, não desta leva. --- .../12369-i18n-cc-onboarding-placeholder.md | 1 + src/i18n/messages/ar.json | 2 +- src/i18n/messages/az.json | 2 +- src/i18n/messages/bg.json | 2 +- src/i18n/messages/bn.json | 2 +- src/i18n/messages/cs.json | 2 +- src/i18n/messages/da.json | 2 +- src/i18n/messages/de.json | 2 +- src/i18n/messages/en.json | 2 +- src/i18n/messages/es.json | 2 +- src/i18n/messages/fa.json | 2 +- src/i18n/messages/fi.json | 2 +- src/i18n/messages/fr.json | 2 +- src/i18n/messages/gu.json | 2 +- src/i18n/messages/he.json | 2 +- src/i18n/messages/hi.json | 2 +- src/i18n/messages/hu.json | 2 +- src/i18n/messages/id.json | 2 +- src/i18n/messages/it.json | 2 +- src/i18n/messages/ja.json | 2 +- src/i18n/messages/ko.json | 2 +- src/i18n/messages/mr.json | 2 +- src/i18n/messages/ms.json | 2 +- src/i18n/messages/nl.json | 2 +- src/i18n/messages/no.json | 2 +- src/i18n/messages/phi.json | 2 +- src/i18n/messages/pl.json | 2 +- src/i18n/messages/pt-BR.json | 2 +- src/i18n/messages/pt.json | 2 +- src/i18n/messages/ro.json | 2 +- src/i18n/messages/ru.json | 2 +- src/i18n/messages/sk.json | 2 +- src/i18n/messages/sv.json | 2 +- src/i18n/messages/sw.json | 2 +- src/i18n/messages/ta.json | 2 +- src/i18n/messages/te.json | 2 +- src/i18n/messages/th.json | 2 +- src/i18n/messages/tr.json | 2 +- src/i18n/messages/uk-UA.json | 2 +- src/i18n/messages/ur.json | 2 +- src/i18n/messages/vi.json | 2 +- src/i18n/messages/zh-CN.json | 2 +- src/i18n/messages/zh-TW.json | 2 +- ...8n-cc-onboarding-placeholder-12302.test.ts | 90 +++++++++++++++++++ 44 files changed, 133 insertions(+), 42 deletions(-) create mode 100644 changelog.d/fixes/12369-i18n-cc-onboarding-placeholder.md create mode 100644 tests/unit/i18n-cc-onboarding-placeholder-12302.test.ts diff --git a/changelog.d/fixes/12369-i18n-cc-onboarding-placeholder.md b/changelog.d/fixes/12369-i18n-cc-onboarding-placeholder.md new file mode 100644 index 0000000000..fac5d9349b --- /dev/null +++ b/changelog.d/fixes/12369-i18n-cc-onboarding-placeholder.md @@ -0,0 +1 @@ +- **fix(i18n):** wrap `ccOnboardingKeyPlaceholder` in ICU single quotes across all 43 locales so angle brackets render literally instead of being parsed as rich-text tags, which crashed the Claude Code onboarding block with `INVALID_MESSAGE: INVALID_TAG` ([#12302](https://github.com/diegosouzapw/OmniRoute/issues/12302)) diff --git a/src/i18n/messages/ar.json b/src/i18n/messages/ar.json index f470c997b5..dcaaed82f8 100644 --- a/src/i18n/messages/ar.json +++ b/src/i18n/messages/ar.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json لاكتشاف نموذج البوابة", "ccOnboardingCopy": "نسخ", "ccOnboardingCopied": "تم النسخ", - "ccOnboardingKeyPlaceholder": "<مفتاح واجهة برمجة تطبيقات OmniRoute الخاص بك>", + "ccOnboardingKeyPlaceholder": "'<مفتاح واجهة برمجة تطبيقات OmniRoute الخاص بك>'", "ccOnboardingWindowNote": "يفترض Claude Code وجود نافذة سياق تبلغ 200K لأي معرف نموذج لا يتعرف عليه. بالنسبة لنموذج له نافذة حقيقية مختلفة، أضف CLAUDE_CODE_AUTO_COMPACT_WINDOW أسفلها مباشرة حتى لا يتم تشغيل الضغط التلقائي في وقت مبكر جدًا.", "failedSave": "فشل الحفظ", "profileSyncTitle": "المزامنة التلقائية لملفات تعريف CLI", diff --git a/src/i18n/messages/az.json b/src/i18n/messages/az.json index 10cd89ab9d..e13bd45903 100644 --- a/src/i18n/messages/az.json +++ b/src/i18n/messages/az.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "gateway modeli kəşfi üçün settings.json", "ccOnboardingCopy": "Kopyala", "ccOnboardingCopied": "Kopyalandı", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code tanımadığı hər hansı model id üçün 200K kontekst pəncərəsi qəbul edir. Fərqli real pəncərəyə malik bir model üçün, avtomatik sıxılmanın çox tez başlamaması üçün onun altına CLAUDE_CODE_AUTO_COMPACT_WINDOW əlavə edin.", "failedSave": "Yadda saxlamaq mümkün olmadı", "profileSyncTitle": "CLI profilinin avtomatik sinxronizasiyası", diff --git a/src/i18n/messages/bg.json b/src/i18n/messages/bg.json index a46d545e1c..f8662e5db3 100644 --- a/src/i18n/messages/bg.json +++ b/src/i18n/messages/bg.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json за откриване на модел на шлюз", "ccOnboardingCopy": "Копирай", "ccOnboardingCopied": "Копирано", - "ccOnboardingKeyPlaceholder": "<вашият ключ за OmniRoute API>", + "ccOnboardingKeyPlaceholder": "'<вашият ключ за OmniRoute API>'", "ccOnboardingWindowNote": "Claude Code предполагае контекстен прозорец от 200K за всяко идентификатор на модел, който не разпознава. За модел с различен реален прозорец, добавете CLAUDE_CODE_AUTO_COMPACT_WINDOW точно под него, за да не се задейства автоматичното компресиране твърде рано.", "failedSave": "Неуспешно запазване", "profileSyncTitle": "Автоматично синхронизиране на CLI профили", diff --git a/src/i18n/messages/bn.json b/src/i18n/messages/bn.json index 20f15cafa2..0b768593ef 100644 --- a/src/i18n/messages/bn.json +++ b/src/i18n/messages/bn.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "gateway মডেল আবিষ্কারের জন্য settings.json", "ccOnboardingCopy": "কপি করুন", "ccOnboardingCopied": "কপি করা হয়েছে", - "ccOnboardingKeyPlaceholder": "<আপনার OmniRoute API কী>", + "ccOnboardingKeyPlaceholder": "'<আপনার OmniRoute API কী>'", "ccOnboardingWindowNote": "Claude Code একটি 200K প্রসঙ্গ উইন্ডো ধারণ করে যেকোন মডেল আইডির জন্য যা এটি চিনতে পারে না। একটি ভিন্ন বাস্তব উইন্ডো সহ মডেলের জন্য, এর ঠিক নিচে CLAUDE_CODE_AUTO_COMPACT_WINDOW যোগ করুন যাতে স্বয়ংক্রিয় সংকোচন খুব তাড়াতাড়ি শুরু না হয়।", "failedSave": "সংরক্ষণ করতে ব্যর্থ হয়েছে", "profileSyncTitle": "CLI প্রোফাইল অটো-সিঙ্ক", diff --git a/src/i18n/messages/cs.json b/src/i18n/messages/cs.json index 427e6d2ef6..ec3e8ce094 100644 --- a/src/i18n/messages/cs.json +++ b/src/i18n/messages/cs.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json pro objevování modelu brány", "ccOnboardingCopy": "Kopírovat", "ccOnboardingCopied": "Zkopírováno", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code předpokládá kontextové okno 200K pro jakýkoli model id, který nepozná. Pro model s jiným skutečným oknem přidejte CLAUDE_CODE_AUTO_COMPACT_WINDOW těsně pod něj, aby automatická komprese nenastala příliš brzy.", "failedSave": "Nepodařilo se uložit", "profileSyncTitle": "Automatická synchronizace profilů CLI", diff --git a/src/i18n/messages/da.json b/src/i18n/messages/da.json index fa3c227a46..9587febd10 100644 --- a/src/i18n/messages/da.json +++ b/src/i18n/messages/da.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json til gateway model opdagelse", "ccOnboardingCopy": "Kopier", "ccOnboardingCopied": "Kopieret", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code antager et 200K kontekstvindue for enhver model-id, den ikke genkender. For en model med et andet reelt vindue, tilføj CLAUDE_CODE_AUTO_COMPACT_WINDOW lige under den, så auto-komprimering ikke aktiveres for tidligt.", "failedSave": "Kunne ikke gemme", "profileSyncTitle": "Automatisk synkronisering af CLI-profil", diff --git a/src/i18n/messages/de.json b/src/i18n/messages/de.json index 16e1370d91..d233f0dba4 100644 --- a/src/i18n/messages/de.json +++ b/src/i18n/messages/de.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json für die Entdeckung des Gateway-Modells", "ccOnboardingCopy": "Kopieren", "ccOnboardingCopied": "Kopiert", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code geht von einem 200K-Kontextfenster für jede Modell-ID aus, die es nicht erkennt. Für ein Modell mit einem anderen tatsächlichen Fenster fügen Sie CLAUDE_CODE_AUTO_COMPACT_WINDOW direkt darunter hinzu, damit die automatische Komprimierung nicht zu früh ausgelöst wird.", "failedSave": "Speichern fehlgeschlagen", "profileSyncTitle": "Automatische Synchronisierung von CLI-Profilen", diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index 163a1fe581..88ba6a9bdc 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json for gateway model discovery", "ccOnboardingCopy": "Copy", "ccOnboardingCopied": "Copied", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code assumes a 200K context window for any model id it does not recognize. For a model with a different real window, add CLAUDE_CODE_AUTO_COMPACT_WINDOW just under it so auto-compaction does not fire too early.", "failedSave": "Failed to save", "profileSyncTitle": "CLI profile auto-sync", diff --git a/src/i18n/messages/es.json b/src/i18n/messages/es.json index 73dd31e3b8..24dd42291a 100644 --- a/src/i18n/messages/es.json +++ b/src/i18n/messages/es.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json para el descubrimiento del modelo de gateway", "ccOnboardingCopy": "Copiar", "ccOnboardingCopied": "Copiado", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code asume una ventana de contexto de 200K para cualquier ID de modelo que no reconozca. Para un modelo con una ventana real diferente, añade CLAUDE_CODE_AUTO_COMPACT_WINDOW justo debajo para que la auto-compresión no se active demasiado pronto.", "failedSave": "Failed to save", "profileSyncTitle": "CLI profile auto-sync", diff --git a/src/i18n/messages/fa.json b/src/i18n/messages/fa.json index f6507ed49e..1b643eed99 100644 --- a/src/i18n/messages/fa.json +++ b/src/i18n/messages/fa.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json برای کشف مدل دروازه", "ccOnboardingCopy": "کپی", "ccOnboardingCopied": "کپی شد", - "ccOnboardingKeyPlaceholder": "<کلید API OmniRoute شما>", + "ccOnboardingKeyPlaceholder": "'<کلید API OmniRoute شما>'", "ccOnboardingWindowNote": "Claude Code فرض می‌کند که یک پنجره متنی 200K برای هر شناسه مدلی که شناسایی نمی‌کند وجود دارد. برای مدلی با پنجره واقعی متفاوت، CLAUDE_CODE_AUTO_COMPACT_WINDOW را درست زیر آن اضافه کنید تا فشرده‌سازی خودکار خیلی زود فعال نشود.", "failedSave": "ذخیره‌سازی ناموفق بود", "profileSyncTitle": "همگام‌سازی خودکار پروفایل CLI", diff --git a/src/i18n/messages/fi.json b/src/i18n/messages/fi.json index 253e58687f..ea6c058f9a 100644 --- a/src/i18n/messages/fi.json +++ b/src/i18n/messages/fi.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json portin mallin löytämiseksi", "ccOnboardingCopy": "Kopioi", "ccOnboardingCopied": "Kopioitu", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code olettaa 200K kontekstin ikkunan kaikille mallin tunnuksille, joita se ei tunnista. Mallille, jolla on eri todellinen ikkuna, lisää CLAUDE_CODE_AUTO_COMPACT_WINDOW sen alle, jotta automaattinen tiivistys ei käynnisty liian aikaisin.", "failedSave": "Tallennus epäonnistui", "profileSyncTitle": "CLI-profiilien automaattinen synkronointi", diff --git a/src/i18n/messages/fr.json b/src/i18n/messages/fr.json index 9870e03616..b1dc36a641 100644 --- a/src/i18n/messages/fr.json +++ b/src/i18n/messages/fr.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json pour la découverte des modèles de la passerelle", "ccOnboardingCopy": "Copier", "ccOnboardingCopied": "Copié", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code suppose une fenêtre de contexte de 200K pour tout identifiant de modèle qu'il ne reconnaît pas. Pour un modèle dont la fenêtre réelle est différente, ajoutez CLAUDE_CODE_AUTO_COMPACT_WINDOW juste en dessous afin que le compactage automatique ne se déclenche pas trop tôt.", "failedSave": "Échec de l'enregistrement", "profileSyncTitle": "Synchronisation automatique des profils CLI", diff --git a/src/i18n/messages/gu.json b/src/i18n/messages/gu.json index e6e20f58c2..b8f03d0f23 100644 --- a/src/i18n/messages/gu.json +++ b/src/i18n/messages/gu.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "ગેટવે મોડલ શોધ માટે settings.json", "ccOnboardingCopy": "કોપી", "ccOnboardingCopied": "કોપી કરેલ", - "ccOnboardingKeyPlaceholder": "<તમારો OmniRoute API કી>", + "ccOnboardingKeyPlaceholder": "'<તમારો OmniRoute API કી>'", "ccOnboardingWindowNote": "Claude Code એ કોઈપણ મોડેલ આઈડી માટે 200K સંદર્ભ વિન્ડો માન્ય રાખે છે જે તે ઓળખતું નથી. જુદા રિયલ વિન્ડો ધરાવતા મોડેલ માટે, નીચે CLAUDE_CODE_AUTO_COMPACT_WINDOW ઉમેરો જેથી ઓટો-કમ્પેક્શન ખૂબ જ વહેલા ન ફાયર થાય.", "failedSave": "સાચવવામાં નિષ્ફળ", "profileSyncTitle": "CLI પ્રોફાઇલ ઓટો-સિંક", diff --git a/src/i18n/messages/he.json b/src/i18n/messages/he.json index 654be87cc9..cb5619f767 100644 --- a/src/i18n/messages/he.json +++ b/src/i18n/messages/he.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json עבור גילוי מודל שער", "ccOnboardingCopy": "העתק", "ccOnboardingCopied": "הועתק", - "ccOnboardingKeyPlaceholder": "<מפתח ה-API של OmniRoute שלך>", + "ccOnboardingKeyPlaceholder": "'<מפתח ה-API של OmniRoute שלך>'", "ccOnboardingWindowNote": "Claude Code מניח חלון הקשר של 200K עבור כל מזהה מודל שהוא לא מזהה. עבור מודל עם חלון אמיתי שונה, הוסף CLAUDE_CODE_AUTO_COMPACT_WINDOW מיד מתחתיו כך שהדחיסה האוטומטית לא תתבצע מוקדם מדי.", "failedSave": "השמירה נכשלה", "profileSyncTitle": "סנכרון אוטומטי של פרופילי CLI", diff --git a/src/i18n/messages/hi.json b/src/i18n/messages/hi.json index f02e4c2e5c..3cc6c5038a 100644 --- a/src/i18n/messages/hi.json +++ b/src/i18n/messages/hi.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "गेटवे मॉडल खोज के लिए settings.json", "ccOnboardingCopy": "कॉपी", "ccOnboardingCopied": "कॉपी किया गया", - "ccOnboardingKeyPlaceholder": "<आपकी OmniRoute API कुंजी>", + "ccOnboardingKeyPlaceholder": "'<आपकी OmniRoute API कुंजी>'", "ccOnboardingWindowNote": "Claude Code किसी भी मॉडल आईडी के लिए 200K संदर्भ विंडो मानता है जिसे वह पहचानता नहीं है। यदि किसी मॉडल की वास्तविक विंडो अलग है, तो इसे उसके ठीक नीचे CLAUDE_CODE_AUTO_COMPACT_WINDOW जोड़ें ताकि ऑटो-कम्पैक्शन बहुत जल्दी न चले।", "failedSave": "सहेजने में विफल", "profileSyncTitle": "CLI प्रोफ़ाइल ऑटो-सिंक", diff --git a/src/i18n/messages/hu.json b/src/i18n/messages/hu.json index 7f30948405..abcd906753 100644 --- a/src/i18n/messages/hu.json +++ b/src/i18n/messages/hu.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json a gateway modell felfedezéshez", "ccOnboardingCopy": "Másolás", "ccOnboardingCopied": "Másolt", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "A Claude Code 200K kontextusablakot feltételez bármely olyan modellazonosító esetén, amelyet nem ismer fel. Ha a modellnek eltérő valós ablaka van, add hozzá a CLAUDE_CODE_AUTO_COMPACT_WINDOW-t közvetlenül alá, hogy az automatikus tömörítés ne lépjen működésbe túl korán.", "failedSave": "Nem sikerült menteni", "profileSyncTitle": "CLI-profil automatikus szinkronizálása", diff --git a/src/i18n/messages/id.json b/src/i18n/messages/id.json index d92469fca9..62a822f6f0 100644 --- a/src/i18n/messages/id.json +++ b/src/i18n/messages/id.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json untuk penemuan model gateway", "ccOnboardingCopy": "Salin", "ccOnboardingCopied": "Disalin", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code mengasumsikan jendela konteks 200K untuk setiap ID model yang tidak dikenalnya. Untuk model dengan jendela nyata yang berbeda, tambahkan CLAUDE_CODE_AUTO_COMPACT_WINDOW tepat di bawahnya agar kompak otomatis tidak aktif terlalu awal.", "failedSave": "Gagal menyimpan", "profileSyncTitle": "Sinkronisasi otomatis profil CLI", diff --git a/src/i18n/messages/it.json b/src/i18n/messages/it.json index 0b1d244f02..8c8d28641c 100644 --- a/src/i18n/messages/it.json +++ b/src/i18n/messages/it.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json per la scoperta modelli gateway", "ccOnboardingCopy": "Copia", "ccOnboardingCopied": "Copiato", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code assume una finestra di contesto di 200K per qualsiasi id modello non riconosciuto. Per un modello con una finestra reale diversa, aggiungi CLAUDE_CODE_AUTO_COMPACT_WINDOW subito sotto in modo che l'auto-compattazione non si attivi troppo presto.", "failedSave": "Impossibile salvare", "profileSyncTitle": "Sincronizzazione automatica profili CLI", diff --git a/src/i18n/messages/ja.json b/src/i18n/messages/ja.json index 61f4d19d6e..2281cdb29e 100644 --- a/src/i18n/messages/ja.json +++ b/src/i18n/messages/ja.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "gatewayモデル発見のためのsettings.json", "ccOnboardingCopy": "コピー", "ccOnboardingCopied": "コピーしました", - "ccOnboardingKeyPlaceholder": "<あなたのOmniRoute APIキー>", + "ccOnboardingKeyPlaceholder": "'<あなたのOmniRoute APIキー>'", "ccOnboardingWindowNote": "Claude Codeは、認識できないモデルIDに対して200Kのコンテキストウィンドウを仮定します。異なる実際のウィンドウを持つモデルの場合は、CLAUDE_CODE_AUTO_COMPACT_WINDOWをその直下に追加して、自動圧縮が早すぎることがないようにします。", "failedSave": "保存に失敗しました", "profileSyncTitle": "CLIプロファイルの自動同期", diff --git a/src/i18n/messages/ko.json b/src/i18n/messages/ko.json index c4b71001df..8fe59e947e 100644 --- a/src/i18n/messages/ko.json +++ b/src/i18n/messages/ko.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "gateway 모델 발견을 위한 settings.json", "ccOnboardingCopy": "복사", "ccOnboardingCopied": "복사됨", - "ccOnboardingKeyPlaceholder": "<귀하의 OmniRoute API 키>", + "ccOnboardingKeyPlaceholder": "'<귀하의 OmniRoute API 키>'", "ccOnboardingWindowNote": "Claude Code는 인식하지 못하는 모델 ID에 대해 200K 컨텍스트 창을 가정합니다. 다른 실제 창을 가진 모델의 경우, 자동 압축이 너무 일찍 발생하지 않도록 그 아래에 CLAUDE_CODE_AUTO_COMPACT_WINDOW를 추가하세요.", "failedSave": "저장하지 못했습니다", "profileSyncTitle": "CLI 프로필 자동 동기화", diff --git a/src/i18n/messages/mr.json b/src/i18n/messages/mr.json index eabefd5464..0e41c94813 100644 --- a/src/i18n/messages/mr.json +++ b/src/i18n/messages/mr.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "gateway मॉडेल शोधासाठी settings.json", "ccOnboardingCopy": "कॉपी", "ccOnboardingCopied": "कॉपी केलेले", - "ccOnboardingKeyPlaceholder": "<तुमचा OmniRoute API की>", + "ccOnboardingKeyPlaceholder": "'<तुमचा OmniRoute API की>'", "ccOnboardingWindowNote": "Claude Code कोणत्याही ओळखत नसलेल्या मॉडेल आयडीसाठी 200K संदर्भ विंडो गृहीत धरतो. भिन्न वास्तविक विंडो असलेल्या मॉडेलसाठी, CLAUDE_CODE_AUTO_COMPACT_WINDOW त्याच्या खालीच जोडा जेणेकरून ऑटो-कम्पॅक्शन लवकर सुरू होणार नाही.", "failedSave": "सेव्ह करण्यात अयशस्वी", "profileSyncTitle": "CLI प्रोफाइल ऑटो-सिंक", diff --git a/src/i18n/messages/ms.json b/src/i18n/messages/ms.json index 98a8e04778..23b56c84a1 100644 --- a/src/i18n/messages/ms.json +++ b/src/i18n/messages/ms.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json untuk penemuan model gateway", "ccOnboardingCopy": "Salin", "ccOnboardingCopied": "Disalin", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code menganggap tetingkap konteks 200K untuk mana-mana ID model yang tidak dikenali. Untuk model dengan tetingkap sebenar yang berbeza, tambah CLAUDE_CODE_AUTO_COMPACT_WINDOW tepat di bawahnya supaya pemampatan automatik tidak berlaku terlalu awal.", "failedSave": "Gagal menyimpan", "profileSyncTitle": "Penyinkronan automatik profil CLI", diff --git a/src/i18n/messages/nl.json b/src/i18n/messages/nl.json index dc251e05db..0f2f81bbea 100644 --- a/src/i18n/messages/nl.json +++ b/src/i18n/messages/nl.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json voor gateway model ontdekking", "ccOnboardingCopy": "Kopiëren", "ccOnboardingCopied": "Gekopieerd", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code gaat uit van een contextvenster van 200K voor elk model-id dat het niet herkent. Voor een model met een ander echt venster, voeg CLAUDE_CODE_AUTO_COMPACT_WINDOW net eronder toe zodat automatische compactie niet te vroeg wordt geactiveerd.", "failedSave": "Opslaan mislukt", "profileSyncTitle": "Automatische synchronisatie van CLI-profielen", diff --git a/src/i18n/messages/no.json b/src/i18n/messages/no.json index 0051f48f05..371319f17a 100644 --- a/src/i18n/messages/no.json +++ b/src/i18n/messages/no.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json for gateway modelloppdagelse", "ccOnboardingCopy": "Kopier", "ccOnboardingCopied": "Kopiert", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code antar et 200K kontekstvindu for enhver modell-ID den ikke gjenkjenner. For en modell med et annet reelt vindu, legg til CLAUDE_CODE_AUTO_COMPACT_WINDOW rett under den, slik at automatisk komprimering ikke aktiveres for tidlig.", "failedSave": "Kunne ikke lagre", "profileSyncTitle": "Automatisk synkronisering av CLI-profiler", diff --git a/src/i18n/messages/phi.json b/src/i18n/messages/phi.json index 6263e50e01..04fd631ba1 100644 --- a/src/i18n/messages/phi.json +++ b/src/i18n/messages/phi.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json para sa pagtuklas ng modelo ng gateway", "ccOnboardingCopy": "Kopyahin", "ccOnboardingCopied": "Nakopya", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Ipinapalagay ng Claude Code ang 200K na konteksto para sa anumang model id na hindi nito nakikilala. Para sa isang modelo na may ibang tunay na bintana, idagdag ang CLAUDE_CODE_AUTO_COMPACT_WINDOW sa ilalim nito upang hindi masyadong maaga ang pag-activate ng auto-compaction.", "failedSave": "Nabigong i-save", "profileSyncTitle": "Auto-sync ng profile ng CLI", diff --git a/src/i18n/messages/pl.json b/src/i18n/messages/pl.json index 29fc91d100..9b0c208058 100644 --- a/src/i18n/messages/pl.json +++ b/src/i18n/messages/pl.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json dla odkrywania modelu bramy", "ccOnboardingCopy": "Kopiuj", "ccOnboardingCopied": "Skopiowano", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code zakłada okno kontekstowe 200K dla każdego identyfikatora modelu, którego nie rozpoznaje. Dla modelu z innym rzeczywistym oknem, dodaj CLAUDE_CODE_AUTO_COMPACT_WINDOW tuż pod nim, aby automatyczna kompresja nie uruchomiła się zbyt wcześnie.", "failedSave": "Nie udało się zapisać", "profileSyncTitle": "Automatyczna synchronizacja profili CLI", diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index 15b6f6df86..840a2d5656 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -2711,7 +2711,7 @@ "ccOnboardingTitle": "settings.json para descoberta de modelos pelo gateway", "ccOnboardingCopy": "Copiar", "ccOnboardingCopied": "Copiado", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "O Claude Code assume uma janela de contexto de 200K para qualquer id de modelo que ele não reconhece. Para um modelo com janela real diferente, acrescente CLAUDE_CODE_AUTO_COMPACT_WINDOW logo abaixo para a compactação automática não disparar cedo demais.", "failedSave": "Falha ao salvar", "profileSyncTitle": "Sincronização automática de perfis de CLI", diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index 3ecb98b353..baa6d1fe1c 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json para descoberta do modelo de gateway", "ccOnboardingCopy": "Copiar", "ccOnboardingCopied": "Copiado", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code assume uma janela de contexto de 200K para qualquer ID de modelo que não reconhece. Para um modelo com uma janela real diferente, adicione CLAUDE_CODE_AUTO_COMPACT_WINDOW logo abaixo para que a auto-compacção não seja acionada muito cedo.", "failedSave": "Falha ao guardar", "profileSyncTitle": "Sincronização automática de perfis da CLI", diff --git a/src/i18n/messages/ro.json b/src/i18n/messages/ro.json index 124b1a610d..65c1129bed 100644 --- a/src/i18n/messages/ro.json +++ b/src/i18n/messages/ro.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json pentru descoperirea modelului gateway", "ccOnboardingCopy": "Copiază", "ccOnboardingCopied": "Copiat", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code presupune o fereastră de context de 200K pentru orice ID de model pe care nu îl recunoaște. Pentru un model cu o fereastră reală diferită, adăugați CLAUDE_CODE_AUTO_COMPACT_WINDOW imediat sub acesta, astfel încât auto-compacția să nu se activeze prea devreme.", "failedSave": "Eroare la salvare", "profileSyncTitle": "Sincronizare automată profil CLI", diff --git a/src/i18n/messages/ru.json b/src/i18n/messages/ru.json index 7b65d8eafa..2d011d4c1d 100644 --- a/src/i18n/messages/ru.json +++ b/src/i18n/messages/ru.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json для обнаружения gateway моделей", "ccOnboardingCopy": "Копировать", "ccOnboardingCopied": "Скопировано", - "ccOnboardingKeyPlaceholder": "<ваш API ключ OmniRoute>", + "ccOnboardingKeyPlaceholder": "'<ваш API ключ OmniRoute>'", "ccOnboardingWindowNote": "Claude Code предполагает окно контекста в 200K для любой незнакомой модели. Если у модели другой реальный размер окна, добавьте CLAUDE_CODE_AUTO_COMPACT_WINDOW прямо под ней, чтобы авто-компактизация не срабатывала слишком рано.", "failedSave": "Не удалось сохранить", "profileSyncTitle": "Автосинхронизация профилей CLI", diff --git a/src/i18n/messages/sk.json b/src/i18n/messages/sk.json index 2f8c814357..db24a48116 100644 --- a/src/i18n/messages/sk.json +++ b/src/i18n/messages/sk.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json pre objavovanie modelu brány", "ccOnboardingCopy": "Kopírovať", "ccOnboardingCopied": "Skopírované", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code predpokladá kontextové okno 200K pre akékoľvek ID modelu, ktoré nerozpoznáva. Pre model s iným skutočným oknom pridajte CLAUDE_CODE_AUTO_COMPACT_WINDOW tesne pod ním, aby sa automatická kompresia nespustila príliš skoro.", "failedSave": "Nepodarilo sa uložiť", "profileSyncTitle": "Automatická synchronizácia profilov CLI", diff --git a/src/i18n/messages/sv.json b/src/i18n/messages/sv.json index 73a2cea1dd..0ee1fc3362 100644 --- a/src/i18n/messages/sv.json +++ b/src/i18n/messages/sv.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json för gateway-modellupptäckten", "ccOnboardingCopy": "Kopiera", "ccOnboardingCopied": "Kopierad", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code förutsätter ett 200K kontextfönster för alla modell-ID:n som det inte känner igen. För en modell med ett annat verkligt fönster, lägg till CLAUDE_CODE_AUTO_COMPACT_WINDOW precis under den så att automatisk komprimering inte aktiveras för tidigt.", "failedSave": "Misslyckades med att spara", "profileSyncTitle": "Automatisk synkning av CLI-profil", diff --git a/src/i18n/messages/sw.json b/src/i18n/messages/sw.json index d1974ce945..72e87fd1ef 100644 --- a/src/i18n/messages/sw.json +++ b/src/i18n/messages/sw.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json kwa ugunduzi wa mfano wa lango", "ccOnboardingCopy": "Nakili", "ccOnboardingCopied": "Imepigwa nakala", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code inadhania dirisha la muktadha la 200K kwa kitambulisho chochote cha mfano ambacho hakitambui. Kwa mfano wenye dirisha halisi tofauti, ongeza CLAUDE_CODE_AUTO_COMPACT_WINDOW chini yake ili auto-compaction isifanye kazi mapema sana.", "failedSave": "Imeshindwa kuhifadhi", "profileSyncTitle": "Ulandanishaji wa kiotomatiki wa wasifu wa CLI", diff --git a/src/i18n/messages/ta.json b/src/i18n/messages/ta.json index 1d12b14de6..8ba8a84e87 100644 --- a/src/i18n/messages/ta.json +++ b/src/i18n/messages/ta.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "gateway மாதிரி கண்டுபிடிப்புக்கு settings.json", "ccOnboardingCopy": "பதிப்பேற்று", "ccOnboardingCopied": "பிரதி எடுக்கப்பட்டது", - "ccOnboardingKeyPlaceholder": "<உங்கள் OmniRoute API விசை>", + "ccOnboardingKeyPlaceholder": "'<உங்கள் OmniRoute API விசை>'", "ccOnboardingWindowNote": "Claude Code எந்த மாதிரி அடையாளத்தை அடையாளம் காணவில்லை என்றால் 200K சூழல் ஜன்னலைக் கருதுகிறது. வேறு உண்மையான ஜன்னலுடன் உள்ள மாதிரிக்கு, CLAUDE_CODE_AUTO_COMPACT_WINDOW ஐ அதன் கீழே சேர்க்கவும், எனவே தானியங்கி சுருக்கம் மிகவும் விரைவாக செயல்படாது.", "failedSave": "சேமிப்பதில் தோல்வி", "profileSyncTitle": "CLI சுயவிவர தானியங்கு ஒத்திசைவு", diff --git a/src/i18n/messages/te.json b/src/i18n/messages/te.json index dc5998cee4..f13ab45bde 100644 --- a/src/i18n/messages/te.json +++ b/src/i18n/messages/te.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "gateway మోడల్ డిస్కవరీ కోసం settings.json", "ccOnboardingCopy": "కాపీ", "ccOnboardingCopied": "కాపీ చేయబడింది", - "ccOnboardingKeyPlaceholder": "<మీ OmniRoute API కీ>", + "ccOnboardingKeyPlaceholder": "'<మీ OmniRoute API కీ>'", "ccOnboardingWindowNote": "Claude Code గుర్తించని ఏ మోడల్ ఐడికి 200K సందర్భం కిటికీని అనుకుంటుంది. వేరే వాస్తవ కిటికీ ఉన్న మోడల్ కోసం, ఆటో-కంపాక్షన్ చాలా త్వరగా జరగకుండా ఉండటానికి దాని కింద CLAUDE_CODE_AUTO_COMPACT_WINDOWని జోడించండి.", "failedSave": "సేవ్ చేయడం విఫలమైంది", "profileSyncTitle": "CLI ప్రొఫైల్ ఆటో-సింక్", diff --git a/src/i18n/messages/th.json b/src/i18n/messages/th.json index 13b002b810..1e4b2d4356 100644 --- a/src/i18n/messages/th.json +++ b/src/i18n/messages/th.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json สำหรับการค้นพบโมเดลเกตเวย์", "ccOnboardingCopy": "คัดลอก", "ccOnboardingCopied": "คัดลอกแล้ว", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code สมมติว่ามีหน้าต่างบริบท 200K สำหรับรหัสโมเดลใด ๆ ที่มันไม่รู้จัก สำหรับโมเดลที่มีหน้าต่างจริงที่แตกต่างกัน ให้เพิ่ม CLAUDE_CODE_AUTO_COMPACT_WINDOW ลงไปใต้โมเดลนั้นเพื่อไม่ให้การบีบอัดอัตโนมัติทำงานเร็วเกินไป", "failedSave": "บันทึกไม่สำเร็จ", "profileSyncTitle": "การซิงค์โปรไฟล์ CLI อัตโนมัติ", diff --git a/src/i18n/messages/tr.json b/src/i18n/messages/tr.json index 3b90a96aac..35bec39cfa 100644 --- a/src/i18n/messages/tr.json +++ b/src/i18n/messages/tr.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "gateway modeli keşfi için settings.json", "ccOnboardingCopy": "Kopyala", "ccOnboardingCopied": "Kopyalandı", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code, tanımadığı herhangi bir model kimliği için 200K bağlam penceresi varsayıyor. Farklı bir gerçek pencereye sahip bir model için, otomatik sıkıştırmanın çok erken başlamaması için hemen altına CLAUDE_CODE_AUTO_COMPACT_WINDOW ekleyin.", "failedSave": "Kaydedilemedi", "profileSyncTitle": "CLI profili otomatik senkronizasyonu", diff --git a/src/i18n/messages/uk-UA.json b/src/i18n/messages/uk-UA.json index a84e6b9b77..ca35b7c104 100644 --- a/src/i18n/messages/uk-UA.json +++ b/src/i18n/messages/uk-UA.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "settings.json для виявлення моделі шлюзу", "ccOnboardingCopy": "Копіювати", "ccOnboardingCopied": "Скопійовано", - "ccOnboardingKeyPlaceholder": "<ваш ключ API OmniRoute>", + "ccOnboardingKeyPlaceholder": "'<ваш ключ API OmniRoute>'", "ccOnboardingWindowNote": "Claude Code припускає вікно контексту 200K для будь-якого ідентифікатора моделі, який він не розпізнає. Для моделі з іншим реальним вікном додайте CLAUDE_CODE_AUTO_COMPACT_WINDOW безпосередньо під ним, щоб автоматичне стиснення не спрацьовувало занадто рано.", "failedSave": "Не вдалося зберегти", "profileSyncTitle": "Автосинхронізація профілів CLI", diff --git a/src/i18n/messages/ur.json b/src/i18n/messages/ur.json index fa38b24d3e..98a41c76ae 100644 --- a/src/i18n/messages/ur.json +++ b/src/i18n/messages/ur.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "gateway ماڈل کی دریافت کے لیے settings.json", "ccOnboardingCopy": "کاپی", "ccOnboardingCopied": "نقل کیا گیا", - "ccOnboardingKeyPlaceholder": "<آپ کا OmniRoute API کلید>", + "ccOnboardingKeyPlaceholder": "'<آپ کا OmniRoute API کلید>'", "ccOnboardingWindowNote": "Claude Code کسی بھی ماڈل ID کے لیے 200K سیاق و سباق کی ونڈو فرض کرتا ہے جسے یہ نہیں پہچانتا۔ اگر کسی ماڈل کی حقیقی ونڈو مختلف ہو تو اس کے نیچے CLAUDE_CODE_AUTO_COMPACT_WINDOW شامل کریں تاکہ خودکار کمپیکشن بہت جلد نہ ہو۔", "failedSave": "محفوظ کرنے میں ناکامی", "profileSyncTitle": "CLI پروفائل کی خودکار مطابقت پذیری", diff --git a/src/i18n/messages/vi.json b/src/i18n/messages/vi.json index cb5dc787d4..0f775b637c 100644 --- a/src/i18n/messages/vi.json +++ b/src/i18n/messages/vi.json @@ -2711,7 +2711,7 @@ "ccOnboardingTitle": "settings.json cho việc khám phá mô hình qua gateway", "ccOnboardingCopy": "Sao chép", "ccOnboardingCopied": "Đã sao chép", - "ccOnboardingKeyPlaceholder": "", + "ccOnboardingKeyPlaceholder": "''", "ccOnboardingWindowNote": "Claude Code mặc định coi mọi id mô hình mà nó không nhận ra là có cửa sổ ngữ cảnh 200K. Với mô hình có cửa sổ thực tế khác, hãy thêm CLAUDE_CODE_AUTO_COMPACT_WINDOW ngay bên dưới để việc nén ngữ cảnh tự động không kích hoạt quá sớm.", "failedSave": "Không thể lưu", "profileSyncTitle": "Tự động đồng bộ hồ sơ CLI", diff --git a/src/i18n/messages/zh-CN.json b/src/i18n/messages/zh-CN.json index 14413d9130..7283008b91 100644 --- a/src/i18n/messages/zh-CN.json +++ b/src/i18n/messages/zh-CN.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "网关模型发现的 settings.json", "ccOnboardingCopy": "复制", "ccOnboardingCopied": "已复制", - "ccOnboardingKeyPlaceholder": "<你的 OmniRoute API 密钥>", + "ccOnboardingKeyPlaceholder": "'<你的 OmniRoute API 密钥>'", "ccOnboardingWindowNote": "Claude Code 对所有不认识的模型 ID 都假设 200K 上下文窗口。如果模型的实际窗口不同,请在下方添加 CLAUDE_CODE_AUTO_COMPACT_WINDOW,这样自动压缩就不会过早触发。", "failedSave": "保存失败", "profileSyncTitle": "CLI 配置文件自动同步", diff --git a/src/i18n/messages/zh-TW.json b/src/i18n/messages/zh-TW.json index 77932434c9..426b3f2544 100644 --- a/src/i18n/messages/zh-TW.json +++ b/src/i18n/messages/zh-TW.json @@ -2710,7 +2710,7 @@ "ccOnboardingTitle": "gateway 模型發現的 settings.json", "ccOnboardingCopy": "複製", "ccOnboardingCopied": "已複製", - "ccOnboardingKeyPlaceholder": "<您的 OmniRoute API 金鑰>", + "ccOnboardingKeyPlaceholder": "'<您的 OmniRoute API 金鑰>'", "ccOnboardingWindowNote": "Claude Code 假設對於任何它不認識的模型 ID,使用 200K 的上下文窗口。對於具有不同實際窗口的模型,請在其下方添加 CLAUDE_CODE_AUTO_COMPACT_WINDOW,以便自動壓縮不會過早觸發。", "failedSave": "無法儲存", "profileSyncTitle": "CLI 設定檔自動同步", diff --git a/tests/unit/i18n-cc-onboarding-placeholder-12302.test.ts b/tests/unit/i18n-cc-onboarding-placeholder-12302.test.ts new file mode 100644 index 0000000000..68fe24268b --- /dev/null +++ b/tests/unit/i18n-cc-onboarding-placeholder-12302.test.ts @@ -0,0 +1,90 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync, readdirSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import path from "node:path"; + +// #12302: ccOnboardingKeyPlaceholder used raw angle brackets () in all 43 locale files. next-intl's IntlMessageFormat parser treated +// these as rich-text tags and threw INVALID_MESSAGE: INVALID_TAG, crashing the +// Claude Code onboarding block. The fix wraps values in ICU single quotes so +// angle brackets render literally. + +const messagesDir = path.join( + path.dirname(fileURLToPath(import.meta.url)), + "../../src/i18n/messages" +); + +function findNested(obj: unknown, key: string): string | undefined { + if (obj === null || typeof obj !== "object") return undefined; + for (const [k, v] of Object.entries(obj as Record)) { + if (k === key && typeof v === "string") return v; + if (v !== null && typeof v === "object") { + const found = findNested(v, key); + if (found !== undefined) return found; + } + } + return undefined; +} + +const localeFiles = readdirSync(messagesDir) + .filter((f) => f.endsWith(".json")) + .sort(); + +test("ccOnboardingKeyPlaceholder exists in all locale files", () => { + for (const file of localeFiles) { + const messages = JSON.parse(readFileSync(path.join(messagesDir, file), "utf8")); + const value = findNested(messages, "ccOnboardingKeyPlaceholder"); + assert.ok(value, `ccOnboardingKeyPlaceholder must exist in ${file}`); + } +}); + +test("ccOnboardingKeyPlaceholder compiles without INVALID_TAG in all locales (#12302)", async () => { + const { IntlMessageFormat } = await import("intl-messageformat"); + + for (const file of localeFiles) { + const messages = JSON.parse(readFileSync(path.join(messagesDir, file), "utf8")); + const value = findNested(messages, "ccOnboardingKeyPlaceholder"); + assert.ok(value, `ccOnboardingKeyPlaceholder must exist in ${file}`); + + const locale = file.replace(".json", ""); + let threw = false; + try { + const fmt = new IntlMessageFormat(value, locale); + fmt.format(); + } catch (err) { + threw = true; + assert.fail(`ccOnboardingKeyPlaceholder in ${file} threw during compilation: ${err}`); + } + assert.ok(!threw, `ccOnboardingKeyPlaceholder in ${file} must not throw`); + } +}); + +test("ccOnboardingKeyPlaceholder renders literal angle brackets in all locales", async () => { + const { IntlMessageFormat } = await import("intl-messageformat"); + + for (const file of localeFiles) { + const messages = JSON.parse(readFileSync(path.join(messagesDir, file), "utf8")); + const value = findNested(messages, "ccOnboardingKeyPlaceholder"); + assert.ok(value, `ccOnboardingKeyPlaceholder must exist in ${file}`); + + const locale = file.replace(".json", ""); + const fmt = new IntlMessageFormat(value, locale); + const result = String(fmt.format()); + + assert.ok( + result.includes("<"), + `ccOnboardingKeyPlaceholder in ${file} must render literal '<', got: ${result}` + ); + assert.ok( + result.includes(">"), + `ccOnboardingKeyPlaceholder in ${file} must render literal '>', got: ${result}` + ); + // Must NOT be treated as a tag — the output should NOT contain "INVALID_TAG" + // or empty output (which happens when tags are stripped). + assert.ok( + result.length > 0, + `ccOnboardingKeyPlaceholder in ${file} must not render empty string` + ); + } +}); From 04ba19fa62199573422612e41420654202237edb Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 23:29:56 -0300 Subject: [PATCH 081/143] chore(quality): rebaseline apiKeys.ts for #12352's preserved ACL (#12673) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rebaseline medido no tip com o #12352 mergeado. Única violação de file-size do tip e inteiramente crescimento próprio daquele PR. --- changelog.d/maintenance/12352-apikeys-filesize.md | 1 + config/quality/file-size-baseline.json | 5 +++-- 2 files changed, 4 insertions(+), 2 deletions(-) create mode 100644 changelog.d/maintenance/12352-apikeys-filesize.md diff --git a/changelog.d/maintenance/12352-apikeys-filesize.md b/changelog.d/maintenance/12352-apikeys-filesize.md new file mode 100644 index 0000000000..6dfbfddc4f --- /dev/null +++ b/changelog.d/maintenance/12352-apikeys-filesize.md @@ -0,0 +1 @@ +- **chore(quality):** rebaseline `src/lib/db/apiKeys.ts` for the ACL the key-creation path now preserves ([#12352](https://github.com/diegosouzapw/OmniRoute/pull/12352)) diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index 5ed9877db2..cc292a2f3b 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -447,7 +447,7 @@ "src/app/api/providers/[id]/test/route.ts": 1252, "src/app/api/v1/models/catalog.ts": 2075, "src/app/docs/lib/openapi.generated.ts": 1347, - "src/lib/db/apiKeys.ts": 1610, + "src/lib/db/apiKeys.ts": 1625, "src/lib/db/core.ts": 1745, "src/lib/db/migrationRunner.ts": 1206, "src/lib/tailscaleTunnel.ts": 1208, @@ -638,5 +638,6 @@ "_rebaseline_2026_09_03_houminxi_batch_stacked": "Crescimento medido DEPOIS que os 9 PRs da leva HouMinXi entraram, quando cada um empilhou sobre o rebaseline do anterior: providers/page.tsx 2007->2025 (+18 = feedback de erro por linha do import CSV do #12504 somado a busca por nome/baseUrl do #12495, ambos no mesmo painel de conexoes); chatCore.ts 5981->5984 (+3 = o #12325 invalida o cache generico de quota no 429 upstream, ao lado do ramo Codex ja existente); accountFallback.ts 2461->2467 (+6 = o #12566 empilha a carve-out de familia Antigravity sobre o rebaseline 2422->2461 que o #12590 registrou para o carve-out credits_exhausted da Moonshot; os dois tocam checkFallbackError). Cada PR mediu certo isoladamente, mas nenhum enxergava o empilhamento. Fiacao em chokepoints existentes. NAO cobre codex.ts nem stream.ts, que ja violavam no tip antes desta leva (drift da base).", "_rebaseline_2026_09_03_12604_claude_code_2_1_258": "PR #12604 (bump da wire identity do Claude Code 2.1.220->2.1.258, commits do @ggiak vindos do #12402) crescimento proprio: src/app/(dashboard)/dashboard/settings/components/RoutingTab.tsx 1606->1607 (+1, a linha do seletor que acompanha a nova versao de identidade). Uma linha num painel de settings ja existente; nao ha o que extrair. Coberto por client-identity-profiles e claude-codex-identity-version-sync (138/138 focados).", "_rebaseline_2026_09_03_hartmark_batch": "Leva hartmark (#12293 #12355 #12447 #12445 #12446 #12460 #12461 #12338 #12448) crescimento proprio, medido no tip com os nove mergeados: src/app/(dashboard)/dashboard/combos/page.tsx 5012->5018 (+6, #12355 impede que a falha de bundling do tiktoken de um provider sem relacao derrube /api/providers, e o painel passa a lidar com o estado degradado); open-sse/services/combo.ts 4023->4036 (+13, #12338 nos fixes do universal-handoff: nota de bare-fallback, escopo por mesma requisicao e log da falha silenciosa). Fiacao em chokepoints existentes do roteamento de combo. NAO cobre codex.ts nem stream.ts, ja violando no tip antes desta leva (drift da base).", - "_rebaseline_2026_09_03_error_boundary_campaign": "Campanha de error-boundary (#12431 #12438 #12444 #12454 #12455 #12456 #12457 #12458 #12459 #12465 #12466 #12467 #12469 #12435), medido no tip com os 14 mergeados. open-sse/executors/codex.ts 1499->1505: os primeiros 4 (1499->1503) sao DRIFT ANTERIOR a esta campanha, ja presente no tip antes dela; os 2 ultimos (1503->1505) sao do #12444, que fecha o boundary de falha da resposta do Codex. Absorver o drift junto foi inevitavel porque o cap e um numero so, mas fica registrado aqui que 4 das 6 linhas nao sao desta leva. open-sse/vendor/codex-chatgpt-web/bridge.ts 1322->1335 (+13): tambem do #12444, no mesmo caminho de falha. NAO cobre open-sse/utils/stream.ts, que segue violando por drift anterior e independente." + "_rebaseline_2026_09_03_error_boundary_campaign": "Campanha de error-boundary (#12431 #12438 #12444 #12454 #12455 #12456 #12457 #12458 #12459 #12465 #12466 #12467 #12469 #12435), medido no tip com os 14 mergeados. open-sse/executors/codex.ts 1499->1505: os primeiros 4 (1499->1503) sao DRIFT ANTERIOR a esta campanha, ja presente no tip antes dela; os 2 ultimos (1503->1505) sao do #12444, que fecha o boundary de falha da resposta do Codex. Absorver o drift junto foi inevitavel porque o cap e um numero so, mas fica registrado aqui que 4 das 6 linhas nao sao desta leva. open-sse/vendor/codex-chatgpt-web/bridge.ts 1322->1335 (+13): tambem do #12444, no mesmo caminho de falha. NAO cobre open-sse/utils/stream.ts, que segue violando por drift anterior e independente.", + "_rebaseline_2026_09_03_12352_apikey_acl": "PR #12352 (fix/api-key-create-acl-12275) crescimento proprio: src/lib/db/apiKeys.ts 1610->1625 (+15). A criacao de API key descartava a ACL enviada no payload; preservar essa ACL exige carregar e persistir o conjunto no mesmo chokepoint de INSERT do modulo de dominio, sem extracao possivel sem partir a funcao de criacao ao meio. Coberto pelos testes do proprio PR (54/54 focados na leva)." } From bb8e75a00da5c14aeaf3f7fa2d288c9896fe0662 Mon Sep 17 00:00:00 2001 From: Ravi Tharuma <25951435+RaviTharuma@users.noreply.github.com> Date: Fri, 4 Sep 2026 04:37:47 +0200 Subject: [PATCH 082/143] fix(resilience): surface Responses failed.error.message in 502s (#12472) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 10 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check-file-size` OK e **241/242** nos 29 arquivos de teste que os PRs tocam. A única "falha" não é falha: `tests/unit/autoCombo/strict-zero-cost-filter.test.ts` é um teste em estilo Vitest que eu incluí por engano na invocação do runner nativo do Node — ele quebra no import (`@vitest/runner`), não numa asserção. Ao investigar, descobri que esse arquivo não roda em nenhum dos dois runners hoje (o glob do `test:unit` não lista `autoCombo` e o `include` do Vitest só pega `.tsx` nessa pasta); é um problema pré-existente do repositório, sem relação com esta leva, e vou registrá-lo separadamente. O #12636 conflitava apenas na lista de testes do `@omniroute/opencode-plugin/package.json`, de forma aditiva: o tip já tinha `models-fetcher.test.ts` (do #12607, irmão desta mesma leva) e o #12636 acrescenta `telemetry.test.ts`. Fiz a união dos dois lados (25 arquivos contra 24 de cada) em vez de escolher um, o que teria removido um arquivo da suíte do plugin em silêncio. Obrigado, @RaviTharuma. --- open-sse/utils/diagnostics.ts | 8 +++++++- tests/unit/diagnostics.test.ts | 16 ++++++++++++++++ 2 files changed, 23 insertions(+), 1 deletion(-) diff --git a/open-sse/utils/diagnostics.ts b/open-sse/utils/diagnostics.ts index 06f7a0a0b6..ebee72e1e8 100644 --- a/open-sse/utils/diagnostics.ts +++ b/open-sse/utils/diagnostics.ts @@ -323,8 +323,14 @@ export function describeMalformedNonStream( ): { message: string; code: string; type: string } { const body = resp && typeof resp === "object" ? (resp as Record) : null; if (body?.object === "response" && body.status === "failed") { + const err = body.error && typeof body.error === "object" ? (body.error as Record) : null; + const rawMessage = + typeof err?.message === "string" && err.message.trim().length > 0 ? err.message.trim() : null; return { - message: "upstream reported a failed response without usable output", + // Trim only here; buildErrorBody (chatCore) does the single sanitization pass. + message: rawMessage + ? `upstream reported a failed response: ${rawMessage}` + : "upstream reported a failed response without usable output", code: "upstream_response_failed", type: "upstream_response_error", }; diff --git a/tests/unit/diagnostics.test.ts b/tests/unit/diagnostics.test.ts index dbcb643480..62da56fc0c 100644 --- a/tests/unit/diagnostics.test.ts +++ b/tests/unit/diagnostics.test.ts @@ -105,6 +105,22 @@ test("failed Responses API body gets a request-scoped machine-readable classific }); }); +test("failed Responses API body surfaces the upstream error message when present", () => { + const failed = { + object: "response", + status: "failed", + output: [], + error: { code: "server_error", message: " Gemini 503: overloaded " }, + }; + const reason = detectMalformedNonStream(failed); + assert.equal(reason, "empty_choices"); + assert.deepEqual(describeMalformedNonStream(failed, reason), { + message: "upstream reported a failed response: Gemini 503: overloaded", + code: "upstream_response_failed", + type: "upstream_response_error", + }); +}); + test("detectMalformedNonStream returns 'empty_choices' when choice message has no content", () => { const body = { choices: [{ message: { content: "", tool_calls: null }, finish_reason: "stop" }], From b0557543b88018eaf3ef52d41ef706de7a38a52a Mon Sep 17 00:00:00 2001 From: Ravi Tharuma <25951435+RaviTharuma@users.noreply.github.com> Date: Fri, 4 Sep 2026 04:38:04 +0200 Subject: [PATCH 083/143] fix(opencode-plugin): lengthen /v1/models timeout and attach HTTP status (#12607) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 10 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check-file-size` OK e **241/242** nos 29 arquivos de teste que os PRs tocam. A única "falha" não é falha: `tests/unit/autoCombo/strict-zero-cost-filter.test.ts` é um teste em estilo Vitest que eu incluí por engano na invocação do runner nativo do Node — ele quebra no import (`@vitest/runner`), não numa asserção. Ao investigar, descobri que esse arquivo não roda em nenhum dos dois runners hoje (o glob do `test:unit` não lista `autoCombo` e o `include` do Vitest só pega `.tsx` nessa pasta); é um problema pré-existente do repositório, sem relação com esta leva, e vou registrá-lo separadamente. O #12636 conflitava apenas na lista de testes do `@omniroute/opencode-plugin/package.json`, de forma aditiva: o tip já tinha `models-fetcher.test.ts` (do #12607, irmão desta mesma leva) e o #12636 acrescenta `telemetry.test.ts`. Fiz a união dos dois lados (25 arquivos contra 24 de cada) em vez de escolher um, o que teria removido um arquivo da suíte do plugin em silêncio. Obrigado, @RaviTharuma. --- @omniroute/opencode-plugin/package.json | 2 +- @omniroute/opencode-plugin/src/index.ts | 11 +++-- .../tests/models-fetcher.test.ts | 45 +++++++++++++++++++ ...2-opencode-plugin-models-timeout-status.md | 1 + config/quality/eslint-suppressions.json | 3 ++ 5 files changed, 57 insertions(+), 5 deletions(-) create mode 100644 @omniroute/opencode-plugin/tests/models-fetcher.test.ts create mode 100644 changelog.d/fixes/12602-opencode-plugin-models-timeout-status.md diff --git a/@omniroute/opencode-plugin/package.json b/@omniroute/opencode-plugin/package.json index 527ff4555e..705429fc7b 100644 --- a/@omniroute/opencode-plugin/package.json +++ b/@omniroute/opencode-plugin/package.json @@ -23,7 +23,7 @@ "scripts": { "build": "tsup", "clean": "rm -rf dist", - "test": "node --import tsx/esm --test tests/scaffold.test.ts tests/auth.test.ts tests/options-schema.test.ts tests/multi-instance.test.ts tests/fetch-interceptor.test.ts tests/provider.test.ts tests/gemini-sanitize.test.ts tests/combos.test.ts tests/config-shim.test.ts tests/features.test.ts tests/feature-defaults.test.ts tests/usable-combo.test.ts tests/disk-snapshot-perms.test.ts tests/fork-features.test.ts tests/auto-combo-context.test.ts tests/provider-id-routing.test.ts tests/management-read-token.test.ts tests/auto-sync.test.ts tests/model-allowlist.test.ts tests/log-level.test.ts tests/effort-tier-variants.test.ts tests/naming.test.ts tests/free-budget-magnitude.test.ts", + "test": "node --import tsx/esm --test tests/scaffold.test.ts tests/auth.test.ts tests/options-schema.test.ts tests/multi-instance.test.ts tests/fetch-interceptor.test.ts tests/provider.test.ts tests/gemini-sanitize.test.ts tests/combos.test.ts tests/config-shim.test.ts tests/features.test.ts tests/feature-defaults.test.ts tests/usable-combo.test.ts tests/disk-snapshot-perms.test.ts tests/fork-features.test.ts tests/auto-combo-context.test.ts tests/provider-id-routing.test.ts tests/management-read-token.test.ts tests/auto-sync.test.ts tests/model-allowlist.test.ts tests/log-level.test.ts tests/effort-tier-variants.test.ts tests/naming.test.ts tests/models-fetcher.test.ts tests/free-budget-magnitude.test.ts", "prepublishOnly": "npm run clean && npm run build && npm test" }, "keywords": [ diff --git a/@omniroute/opencode-plugin/src/index.ts b/@omniroute/opencode-plugin/src/index.ts index 18545bfece..3ed215f341 100644 --- a/@omniroute/opencode-plugin/src/index.ts +++ b/@omniroute/opencode-plugin/src/index.ts @@ -1199,7 +1199,7 @@ export type OmniRouteModelsFetcher = ( export const defaultOmniRouteModelsFetcher: OmniRouteModelsFetcher = async ( baseURL, apiKey, - timeoutMs = 10_000 + timeoutMs = 30_000 ) => { if (!apiKey) throw new Error("@omniroute/opencode-plugin: apiKey required to fetch /v1/models"); if (!baseURL) throw new Error("@omniroute/opencode-plugin: baseURL required to fetch /v1/models"); @@ -1221,9 +1221,12 @@ export const defaultOmniRouteModelsFetcher: OmniRouteModelsFetcher = async ( signal: controller.signal, }); if (!res.ok) { - throw new Error( + const err = new Error( `@omniroute/opencode-plugin: GET ${url} failed: ${res.status} ${res.statusText}` - ); + ) as Error & { statusCode: number; status: number }; + err.statusCode = res.status; + err.status = res.status; + throw err; } const body = (await res.json()) as unknown; const rawList: unknown[] = Array.isArray(body) @@ -5398,7 +5401,7 @@ export function createOmniRouteConfigHook( // exact warn message so per-endpoint fallbacks are preserved. const doModels = async (): Promise => { try { - localRawModels = await fetcher(baseURL, apiKey, 10_000); + localRawModels = await fetcher(baseURL, apiKey, 30_000); } catch (err) { logAt( "error", diff --git a/@omniroute/opencode-plugin/tests/models-fetcher.test.ts b/@omniroute/opencode-plugin/tests/models-fetcher.test.ts new file mode 100644 index 0000000000..c33b37b1e0 --- /dev/null +++ b/@omniroute/opencode-plugin/tests/models-fetcher.test.ts @@ -0,0 +1,45 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { defaultOmniRouteModelsFetcher } from "../src/index.js"; + +test("defaultOmniRouteModelsFetcher attaches statusCode on HTTP 401", async () => { + const original = globalThis.fetch; + globalThis.fetch = (async () => + new Response(JSON.stringify({ error: "authentication expired" }), { + status: 401, + statusText: "Unauthorized", + })) as typeof fetch; + try { + await assert.rejects( + () => defaultOmniRouteModelsFetcher("https://gateway.example/v1", "test-key"), + (err: unknown) => { + assert.ok(err instanceof Error); + const rec = err as Error & { statusCode?: number; status?: number }; + assert.equal(rec.statusCode, 401); + assert.equal(rec.status, 401); + assert.match(rec.message, /401/); + return true; + }, + ); + } finally { + globalThis.fetch = original; + } +}); + +test("defaultOmniRouteModelsFetcher default timeout is 30s", async () => { + const original = globalThis.fetch; + let signal: AbortSignal | undefined; + globalThis.fetch = (async (_input, init) => { + signal = init?.signal ?? undefined; + return new Response(JSON.stringify({ object: "list", data: [] }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + }) as typeof fetch; + try { + await defaultOmniRouteModelsFetcher("https://gateway.example/v1", "test-key"); + assert.equal(signal instanceof AbortSignal, true); + } finally { + globalThis.fetch = original; + } +}); diff --git a/changelog.d/fixes/12602-opencode-plugin-models-timeout-status.md b/changelog.d/fixes/12602-opencode-plugin-models-timeout-status.md new file mode 100644 index 0000000000..e86ff8c78d --- /dev/null +++ b/changelog.d/fixes/12602-opencode-plugin-models-timeout-status.md @@ -0,0 +1 @@ +- OpenCode plugin `/v1/models` catalog fetch now waits 30s by default and attaches HTTP `statusCode` on 401/5xx so host fallback plugins can hop instead of seeing an untyped AbortError/UnknownError. diff --git a/config/quality/eslint-suppressions.json b/config/quality/eslint-suppressions.json index e10d1f6551..fe963b547c 100644 --- a/config/quality/eslint-suppressions.json +++ b/config/quality/eslint-suppressions.json @@ -854,6 +854,9 @@ "src/app/(dashboard)/dashboard/combos/page.tsx": { "@typescript-eslint/no-unused-vars": { "count": 6 + }, + "react-hooks/set-state-in-effect": { + "count": 1 } }, "src/app/(dashboard)/dashboard/costs/CostOverviewTab.tsx": { From 8c8d23a98f03ef1635f440a1f410cc28856e3d8c Mon Sep 17 00:00:00 2001 From: Ravi Tharuma <25951435+RaviTharuma@users.noreply.github.com> Date: Fri, 4 Sep 2026 04:38:20 +0200 Subject: [PATCH 084/143] fix(api): bound hung GET /v1/models catalog rebuilds (#12628) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 10 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check-file-size` OK e **241/242** nos 29 arquivos de teste que os PRs tocam. A única "falha" não é falha: `tests/unit/autoCombo/strict-zero-cost-filter.test.ts` é um teste em estilo Vitest que eu incluí por engano na invocação do runner nativo do Node — ele quebra no import (`@vitest/runner`), não numa asserção. Ao investigar, descobri que esse arquivo não roda em nenhum dos dois runners hoje (o glob do `test:unit` não lista `autoCombo` e o `include` do Vitest só pega `.tsx` nessa pasta); é um problema pré-existente do repositório, sem relação com esta leva, e vou registrá-lo separadamente. O #12636 conflitava apenas na lista de testes do `@omniroute/opencode-plugin/package.json`, de forma aditiva: o tip já tinha `models-fetcher.test.ts` (do #12607, irmão desta mesma leva) e o #12636 acrescenta `telemetry.test.ts`. Fiz a união dos dois lados (25 arquivos contra 24 de cada) em vez de escolher um, o que teria removido um arquivo da suíte do plugin em silêncio. Obrigado, @RaviTharuma. --- .env.example | 5 ++ .../fixes/12627-catalog-inflight-timeout.md | 1 + docs/reference/ENVIRONMENT.md | 1 + src/app/api/v1/models/catalogCache.ts | 83 ++++++++++++------- .../12627-catalog-inflight-timeout.test.ts | 61 ++++++++++++++ 5 files changed, 120 insertions(+), 31 deletions(-) create mode 100644 changelog.d/fixes/12627-catalog-inflight-timeout.md create mode 100644 tests/unit/12627-catalog-inflight-timeout.test.ts diff --git a/.env.example b/.env.example index 470a6107d6..aa1c588b8f 100644 --- a/.env.example +++ b/.env.example @@ -1914,6 +1914,11 @@ APP_LOG_TO_FILE=true # Default: true # MODEL_CATALOG_INCLUDE_NAMES=true +# Cold-path wait bound for a coalesced GET /v1/models catalog rebuild (#12627). +# Used by: src/app/api/v1/models/catalogCache.ts +# Default: 8000 (8 seconds). On timeout, a last-good 200 is served when available. +# CATALOG_BUILD_TIMEOUT_MS=8000 + # ── NanoBanana (Image Generation) ── # Polling config for async image generation jobs. # Used by: open-sse/handlers/imageGeneration.ts diff --git a/changelog.d/fixes/12627-catalog-inflight-timeout.md b/changelog.d/fixes/12627-catalog-inflight-timeout.md new file mode 100644 index 0000000000..b50416b2c0 --- /dev/null +++ b/changelog.d/fixes/12627-catalog-inflight-timeout.md @@ -0,0 +1 @@ +- **fix(api):** GET /v1/models no longer waits forever on a hung coalesced catalog rebuild; cold-path waits are bounded (`CATALOG_BUILD_TIMEOUT_MS`, default 8s) and a last-good 200 is served when the rebuild times out ([#12627](https://github.com/diegosouzapw/OmniRoute/issues/12627)). diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index 27bf863d21..ae5abf8781 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -1041,6 +1041,7 @@ desktop install. | ---------------------------------------------- | -------------------------------------- | ---------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------- | | `OPENROUTER_CATALOG_TTL_MS` | `86400000` (24h) | `src/lib/catalog/openrouterCatalog.ts` | OpenRouter model catalog cache TTL. | | `MODEL_CATALOG_INCLUDE_NAMES` | `true` | `src/shared/constants/featureFlagDefinitions.ts` | Include display-friendly `name` fields in `/v1/models` responses. Disable for clients that expect IDs only. | +| `CATALOG_BUILD_TIMEOUT_MS` | `8000` (8s) | `src/app/api/v1/models/catalogCache.ts` | Cold-path wait bound for a coalesced `GET /v1/models` catalog rebuild (#12627). On timeout, a last-good 200 is served when one exists. | | `NANOBANANA_POLL_TIMEOUT_MS` | `120000` | `open-sse/handlers/imageGeneration.ts` | Max wait for NanoBanana image generation jobs. | | `NANOBANANA_POLL_INTERVAL_MS` | `2500` | `open-sse/handlers/imageGeneration.ts` | NanoBanana job polling frequency. | | `ADOBE_FIREFLY_SUBMIT_BASE_DELAY_MS` | `8000` | `open-sse/services/adobeFireflyUpscale.ts` | Base delay for the Adobe Firefly upscale submit-retry exponential backoff. | diff --git a/src/app/api/v1/models/catalogCache.ts b/src/app/api/v1/models/catalogCache.ts index c5a9509397..a7f8d0aa51 100644 --- a/src/app/api/v1/models/catalogCache.ts +++ b/src/app/api/v1/models/catalogCache.ts @@ -135,35 +135,27 @@ export type CatalogCacheOptions = { */ export const CATALOG_CACHE_TTL_MS_DEFAULT = 60_000; -/** - * Per-call knobs for {@link resolveCachedCatalogResponse}. - * - * `hideAutoCombos` / `hideNoThinkVariants` are catalog-shape dimensions folded into - * the cache key. `getStaleWhileRevalidateMs` and `scheduleBackgroundRefresh` are the - * injection points restored in #11551: the route wires Next's `after()` so the - * background refresh runs only once the response has been flushed to the client. - */ +/** Cold-path wait bound for a coalesced catalog rebuild (#12627). Override with CATALOG_BUILD_TIMEOUT_MS. */ +export const CATALOG_BUILD_TIMEOUT_MS_DEFAULT = 8_000; -/** Defers `task` until it is safe to run without delaying the current response. */ +function catalogBuildTimeoutMs(): number { + const raw = process.env.CATALOG_BUILD_TIMEOUT_MS; + if (!raw) return CATALOG_BUILD_TIMEOUT_MS_DEFAULT; + const n = Number.parseInt(raw, 10); + return Number.isFinite(n) && n > 0 ? n : CATALOG_BUILD_TIMEOUT_MS_DEFAULT; +} -/** - * Default scheduler (#8728 / #11551). - * - * Next's `after()` runs the task once the response has been flushed, which is the - * whole point of the stale-while-revalidate path: the builder is overwhelmingly - * synchronous under the single-threaded App Router, so running it before the flush - * pins the event loop and the "served immediately" stale body only reaches the - * client after the rebuild finishes. - * - * `after()` requires a Next request scope. Callers outside one (instrumentation - * warm-up, direct unit-test imports) fall back to a macrotask, which preserves the - * "hand the response back first" ordering within the same process. - */ +const catalogLastGood = new Map(); -type CatalogInFlight = { - version: number; - promise: Promise; -}; +function withTimeout(promise: Promise, ms: number, label: string): Promise { + return new Promise((resolve, reject) => { + const timer = setTimeout(() => reject(new Error(label)), ms); + promise.then( + (value) => { clearTimeout(timer); resolve(value); }, + (err) => { clearTimeout(timer); reject(err); } + ); + }); +} const catalogCache = new Map(); @@ -251,6 +243,7 @@ function storePayload( if (buildGeneration === getModelCatalogCacheVersion()) { catalogCache.set(cacheKey, entry); } + if (entry.status === 200) catalogLastGood.set(cacheKey, entry); return entry; } @@ -318,6 +311,37 @@ function runBuilder( return buildPayload(request); } +async function awaitCatalogInFlight( + cacheKey: string, + inflight: InFlightBuild, + corsHeaders: Record, + diagnosticHeaders: Record +): Promise { + let payload: CachedCatalog; + try { + payload = await withTimeout(inflight.promise, catalogBuildTimeoutMs(), "catalog_build_timeout"); + } catch (err) { + const msg = err instanceof Error ? err.message : String(err); + if (catalogInFlight.get(cacheKey)?.promise === inflight.promise) { + catalogInFlight.delete(cacheKey); + } + const lastGood = catalogLastGood.get(cacheKey); + if (msg === "catalog_build_timeout" && lastGood) { + return new Response(lastGood.body, { + status: lastGood.status, + headers: mergeCatalogHeaders(corsHeaders, lastGood.headers, diagnosticHeaders, { + "x-omniroute-catalog": "last-good", + }), + }); + } + throw err; + } + return new Response(payload.body, { + status: payload.status, + headers: mergeCatalogHeaders(corsHeaders, payload.headers, diagnosticHeaders), + }); +} + /** * Resolve the cached catalog response for `request`, building it through * `buildPayload` when there is nothing fresh to serve. @@ -382,11 +406,7 @@ export async function resolveCachedCatalogResponse( }); } - const payload = await inflight.promise; - return new Response(payload.body, { - status: payload.status, - headers: mergeCatalogHeaders(corsHeaders, payload.headers, diagnosticHeaders), - }); + return awaitCatalogInFlight(cacheKey, inflight, corsHeaders, diagnosticHeaders); } // ── Test hooks ─────────────────────────────────────────────────────────────── @@ -397,6 +417,7 @@ export function __resetCatalogBuilderRunsForTest(): void { _catalogBuilderRuns = 0; catalogCache.clear(); catalogInFlight.clear(); + catalogLastGood.clear(); lastSeenCatalogCacheVersion = getModelCatalogCacheVersion(); } diff --git a/tests/unit/12627-catalog-inflight-timeout.test.ts b/tests/unit/12627-catalog-inflight-timeout.test.ts new file mode 100644 index 0000000000..62abeb4282 --- /dev/null +++ b/tests/unit/12627-catalog-inflight-timeout.test.ts @@ -0,0 +1,61 @@ +/** + * #12627 — a hung coalesced catalog rebuild must not pin later GET /v1/models clients. + */ +import assert from "node:assert/strict"; +import test from "node:test"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-12627-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const catalogCache = await import("../../src/app/api/v1/models/catalogCache.ts"); + +function request() { + return new Request("http://localhost/v1/models"); +} + +function payload(body: string): catalogCache.CatalogPayload { + return { body, headers: { "content-type": "application/json" }, status: 200, cacheTTL: 60_000 }; +} + +const neverResolves = () => new Promise(() => {}); + +test.beforeEach(() => { + catalogCache.__resetCatalogBuilderRunsForTest(); + process.env.CATALOG_BUILD_TIMEOUT_MS = "40"; +}); + +test.afterEach(() => { + delete process.env.CATALOG_BUILD_TIMEOUT_MS; +}); + +test("#12627 cold hung rebuild times out instead of waiting forever", async () => { + await assert.rejects( + catalogCache.resolveCachedCatalogResponse( + request(), + { corsHeaders: {}, diagnosticHeaders: {} }, + neverResolves as (req: Request) => Promise + ), + /catalog_build_timeout/ + ); +}); + +test("#12627 timeout serves last-good 200 when a prior build succeeded", async () => { + const first = await catalogCache.resolveCachedCatalogResponse( + request(), + { corsHeaders: {}, diagnosticHeaders: {} }, + async () => payload("good") + ); + assert.equal(await first.text(), "good"); + catalogCache.__expireCatalogCacheForTest(60_000); + + const second = await catalogCache.resolveCachedCatalogResponse( + request(), + { corsHeaders: {}, diagnosticHeaders: {} }, + neverResolves as (req: Request) => Promise + ); + assert.equal(second.status, 200); + assert.equal(await second.text(), "good"); + assert.equal(second.headers.get("x-omniroute-catalog"), "last-good"); +}); From 3d2bcc9f1271d27b66558796ad884d3317bd5f8e Mon Sep 17 00:00:00 2001 From: Ravi Tharuma <25951435+RaviTharuma@users.noreply.github.com> Date: Fri, 4 Sep 2026 04:38:37 +0200 Subject: [PATCH 085/143] feat(api): emit gateway-measured tokens-per-second excluding TTFT (#12631) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 10 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check-file-size` OK e **241/242** nos 29 arquivos de teste que os PRs tocam. A única "falha" não é falha: `tests/unit/autoCombo/strict-zero-cost-filter.test.ts` é um teste em estilo Vitest que eu incluí por engano na invocação do runner nativo do Node — ele quebra no import (`@vitest/runner`), não numa asserção. Ao investigar, descobri que esse arquivo não roda em nenhum dos dois runners hoje (o glob do `test:unit` não lista `autoCombo` e o `include` do Vitest só pega `.tsx` nessa pasta); é um problema pré-existente do repositório, sem relação com esta leva, e vou registrá-lo separadamente. O #12636 conflitava apenas na lista de testes do `@omniroute/opencode-plugin/package.json`, de forma aditiva: o tip já tinha `models-fetcher.test.ts` (do #12607, irmão desta mesma leva) e o #12636 acrescenta `telemetry.test.ts`. Fiz a união dos dois lados (25 arquivos contra 24 de cada) em vez de escolher um, o que teria removido um arquivo da suíte do plugin em silêncio. Obrigado, @RaviTharuma. --- .../features/12616-tokens-per-second.md | 1 + open-sse/utils/generationThroughput.ts | 44 ++++++++++ open-sse/utils/stream.ts | 12 +-- open-sse/utils/streamTiming.ts | 9 ++ open-sse/utils/usageTracking.ts | 19 ++++- src/domain/omnirouteResponseMeta.ts | 20 +++++ src/shared/constants/headers.ts | 1 + tests/unit/generation-throughput.test.ts | 84 +++++++++++++++++++ 8 files changed, 182 insertions(+), 8 deletions(-) create mode 100644 changelog.d/features/12616-tokens-per-second.md create mode 100644 open-sse/utils/generationThroughput.ts create mode 100644 tests/unit/generation-throughput.test.ts diff --git a/changelog.d/features/12616-tokens-per-second.md b/changelog.d/features/12616-tokens-per-second.md new file mode 100644 index 0000000000..164161da99 --- /dev/null +++ b/changelog.d/features/12616-tokens-per-second.md @@ -0,0 +1 @@ +- **feat(api):** Emit gateway-measured `tokens_per_second` (TTFT excluded) on streaming usage and `X-OmniRoute-Tokens-Per-Second` when first-token latency is known ([#12616](https://github.com/diegosouzapw/OmniRoute/issues/12616)) diff --git a/open-sse/utils/generationThroughput.ts b/open-sse/utils/generationThroughput.ts new file mode 100644 index 0000000000..54d48ad1e3 --- /dev/null +++ b/open-sse/utils/generationThroughput.ts @@ -0,0 +1,44 @@ +/** + * Gateway-measured generation throughput (#12616). + * + * tok/s MUST exclude TTFT. `output_tokens / total_latency` includes queueing and + * first-token wait and is not generation speed. When TTFT is unknown (typical + * non-streaming JSON), omit the field rather than guessing. + */ +export function generationDurationMs( + totalMs: number, + ttftMs: number | null | undefined +): number | null { + if (!Number.isFinite(totalMs) || totalMs <= 0) return null; + if (ttftMs == null || !Number.isFinite(ttftMs) || ttftMs < 0) return null; + const generationMs = totalMs - ttftMs; + return generationMs > 0 ? generationMs : null; +} + +export function tokensPerSecond( + outputTokens: number, + generationMs: number | null | undefined +): number | null { + if (generationMs == null || !Number.isFinite(generationMs) || generationMs <= 0) return null; + if (!Number.isFinite(outputTokens) || outputTokens <= 0) return null; + return outputTokens / (generationMs / 1000); +} + +function outputTokenCount(usage: Record): number { + const raw = + usage.completion_tokens ?? + usage.output_tokens ?? + usage.candidatesTokenCount ?? + usage.outputTokens ?? + usage.completionTokens; + const n = typeof raw === "number" ? raw : typeof raw === "string" ? Number(raw) : NaN; + return Number.isFinite(n) ? n : 0; +} + +/** Attach `tokens_per_second` when generation duration (excluding TTFT) is known. */ +export function attachTokensPerSecond(usage: T, generationMs: number | null | undefined): T { + if (!usage || typeof usage !== "object" || Array.isArray(usage)) return usage; + const tps = tokensPerSecond(outputTokenCount(usage as Record), generationMs); + if (tps == null) return usage; + return { ...(usage as Record), tokens_per_second: Number(tps.toFixed(3)) } as T; +} diff --git a/open-sse/utils/stream.ts b/open-sse/utils/stream.ts index f7b01b1464..6b3c44cfcc 100644 --- a/open-sse/utils/stream.ts +++ b/open-sse/utils/stream.ts @@ -1036,11 +1036,11 @@ export function createSSEStream(options: StreamOptions = {}) { totalContentLength > 0 ) { const estimated = estimateUsage(body, totalContentLength, sourceFormat); - itemSanitized.usage = filterUsageForFormat(estimated, sourceFormat); + itemSanitized.usage = timing.withTps(filterUsageForFormat(estimated, sourceFormat)); state.usage = estimated; } else if (state?.finishReason && isFinishChunk && state.usage) { const buffered = addBufferToUsage(state.usage); - itemSanitized.usage = filterUsageForFormat(buffered, sourceFormat); + itemSanitized.usage = timing.withTps(filterUsageForFormat(buffered, sourceFormat)); } if ( @@ -1079,8 +1079,8 @@ export function createSSEStream(options: StreamOptions = {}) { model, cacheHit: false, latencyMs: Date.now() - streamStartedAt, - usage: finalUsage, - costUsd, + usage: timing.withTps(finalUsage), + costUsd, ttftMs: timing.ttftMs(), }); if (!comment) return; reqLogger?.appendConvertedChunk?.(comment); @@ -2046,7 +2046,7 @@ export function createSSEStream(options: StreamOptions = {}) { // estimate is now emitted in flush(), only when the upstream stayed silent. if (isFinishChunk && hasValidUsage(usage) && !passthroughForwardedUsage) { const buffered = addBufferToUsage(usage); - parsed.usage = filterUsageForFormat(buffered, sourceFormat || FORMATS.OPENAI); + parsed.usage = timing.withTps(filterUsageForFormat(buffered, sourceFormat || FORMATS.OPENAI)); output = `data: ${JSON.stringify(parsed)}\n\n`; passthroughForwardedUsage = true; injectedUsage = true; @@ -2571,7 +2571,7 @@ export function createSSEStream(options: StreamOptions = {}) { created: Math.floor(Date.now() / 1000), model, choices: [], - usage: filterUsageForFormat(usage, sourceFormat || FORMATS.OPENAI), + usage: timing.withTps(filterUsageForFormat(usage, sourceFormat || FORMATS.OPENAI)), }; const usageOutput = `data: ${JSON.stringify(usageOnlyChunk)}\n\n`; reqLogger?.appendConvertedChunk?.(usageOutput); diff --git a/open-sse/utils/streamTiming.ts b/open-sse/utils/streamTiming.ts index ee5385ebb3..28b1329a7c 100644 --- a/open-sse/utils/streamTiming.ts +++ b/open-sse/utils/streamTiming.ts @@ -30,6 +30,8 @@ * instances as absolute times — the same convention `earlyStreamKeepalive.ts` * already follows on this streaming path. */ +import { attachTokensPerSecond, generationDurationMs } from "./generationThroughput.ts"; + export interface StreamTiming { startedAt: number; firstByteAt: number | null; @@ -48,6 +50,10 @@ export interface StreamTiming { avgItlMs(): number | null; /** Time from stream start to completion (ms). */ totalMs(): number; + /** + * Attach gateway-measured tok/s (TTFT excluded). No-op when TTFT is unknown. + */ + withTps(usage: T): T; } /** Max number of inter-chunk samples kept (bounds memory). */ @@ -88,6 +94,9 @@ export function createStreamTiming(): StreamTiming { totalMs() { return performance.now() - this.startedAt; }, + withTps(usage) { + return attachTokensPerSecond(usage, generationDurationMs(this.totalMs(), this.ttftMs())); + }, }; return timing; } diff --git a/open-sse/utils/usageTracking.ts b/open-sse/utils/usageTracking.ts index 25a1a20d8b..d71b858cc8 100644 --- a/open-sse/utils/usageTracking.ts +++ b/open-sse/utils/usageTracking.ts @@ -293,6 +293,7 @@ export function filterUsageForFormat(usage: UsageLike | null | undefined, target "cache_read_input_tokens", "cache_creation_input_tokens", "estimated", + "tokens_per_second", ], [FORMATS.GEMINI]: [ "promptTokenCount", @@ -301,6 +302,7 @@ export function filterUsageForFormat(usage: UsageLike | null | undefined, target "cachedContentTokenCount", "thoughtsTokenCount", "estimated", + "tokens_per_second", ], [FORMATS.OPENAI_RESPONSES]: [ "input_tokens", @@ -312,6 +314,7 @@ export function filterUsageForFormat(usage: UsageLike | null | undefined, target "cost_in_usd_ticks", "server_side_tool_usage_details", "server_side_tool_usage", + "tokens_per_second", ], // OpenAI format (default for OPENAI, CODEX, KIRO, etc.) default: [ @@ -327,6 +330,7 @@ export function filterUsageForFormat(usage: UsageLike | null | undefined, target "cache_read_input_tokens", "cache_creation_input_tokens", "estimated", + "tokens_per_second", ], }; @@ -671,9 +675,20 @@ export function hasValidUsage(usage: UsageLike | null | undefined) { export function isEmptyUsage(usage: unknown): boolean { if (!usage || typeof usage !== "object" || Array.isArray(usage)) return true; const u = usage as Record; - for (const k of ["prompt_tokens","completion_tokens","total_tokens","input_tokens","output_tokens","promptTokenCount","candidatesTokenCount","totalTokenCount"]) { + for (const k of [ + "prompt_tokens", + "completion_tokens", + "total_tokens", + "input_tokens", + "output_tokens", + "promptTokenCount", + "candidatesTokenCount", + "totalTokenCount", + ]) { const v = u[k]; - if (typeof v === "number" && Number.isFinite(v)) { if (v > 0) return false; } + if (typeof v === "number" && Number.isFinite(v)) { + if (v > 0) return false; + } } return true; } diff --git a/src/domain/omnirouteResponseMeta.ts b/src/domain/omnirouteResponseMeta.ts index 6a6a05e958..c9a3d33c14 100644 --- a/src/domain/omnirouteResponseMeta.ts +++ b/src/domain/omnirouteResponseMeta.ts @@ -1,6 +1,10 @@ import { getProviderAlias } from "@/shared/constants/providers"; import { OMNIROUTE_RESPONSE_HEADERS } from "@/shared/constants/headers"; import { APP_CONFIG } from "@/shared/constants/appConfig"; +import { + generationDurationMs, + tokensPerSecond, +} from "@omniroute/open-sse/utils/generationThroughput"; type UsageLike = Record | null | undefined; @@ -123,6 +127,7 @@ export function buildOmniRouteResponseMetaHeaders({ requestId = null, strategy = null, usage = null, + ttftMs = null, }: { cacheHit?: boolean; costUsd?: unknown; @@ -145,6 +150,12 @@ export function buildOmniRouteResponseMetaHeaders({ */ strategy?: string | null; usage?: UsageLike; + /** + * First-token latency in ms. Required to emit tok/s: generation speed is + * `output_tokens / (latencyMs - ttftMs)` and MUST omit the field when TTFT + * is unknown so plugins do not treat `tokens / total_latency` as speed. + */ + ttftMs?: number | null; }): Record { const tokens = getOmniRouteTokenCounts(usage); const headers: Record = { @@ -186,6 +197,15 @@ export function buildOmniRouteResponseMetaHeaders({ headers[OMNIROUTE_RESPONSE_HEADERS.decision] = decisionValue; } + let tps = tokensPerSecond(tokens.output, generationDurationMs(toFiniteNumber(latencyMs), ttftMs)); + if (tps == null && usage && typeof usage === "object") { + const fromUsage = toFiniteNumber((usage as Record).tokens_per_second); + if (fromUsage > 0) tps = fromUsage; + } + if (tps != null) { + headers[OMNIROUTE_RESPONSE_HEADERS.tokensPerSecond] = toHeaderValue(tps.toFixed(3)); + } + return headers; } diff --git a/src/shared/constants/headers.ts b/src/shared/constants/headers.ts index a4b6b50ff2..136be95bf2 100644 --- a/src/shared/constants/headers.ts +++ b/src/shared/constants/headers.ts @@ -14,5 +14,6 @@ export const OMNIROUTE_RESPONSE_HEADERS = { responseCost: "X-OmniRoute-Response-Cost", tokensIn: "X-OmniRoute-Tokens-In", tokensOut: "X-OmniRoute-Tokens-Out", + tokensPerSecond: "X-OmniRoute-Tokens-Per-Second", version: "X-OmniRoute-Version", } as const; diff --git a/tests/unit/generation-throughput.test.ts b/tests/unit/generation-throughput.test.ts new file mode 100644 index 0000000000..bf71f68efb --- /dev/null +++ b/tests/unit/generation-throughput.test.ts @@ -0,0 +1,84 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { + attachTokensPerSecond, + generationDurationMs, + tokensPerSecond, +} from "../../open-sse/utils/generationThroughput.ts"; +import { filterUsageForFormat } from "../../open-sse/utils/usageTracking.ts"; +import { FORMATS } from "../../open-sse/translator/formats.ts"; +import { createStreamTiming } from "../../open-sse/utils/streamTiming.ts"; +import { + buildOmniRouteResponseMetaHeaders, + buildOmniRouteSseMetadataComment, +} from "../../src/domain/omnirouteResponseMeta.ts"; +import { OMNIROUTE_RESPONSE_HEADERS } from "../../src/shared/constants/headers.ts"; + +test("#12616 tok/s excludes TTFT (200 tokens over 2s generation after 3s TTFT)", () => { + const generationMs = generationDurationMs(5000, 3000); + assert.equal(generationMs, 2000); + assert.equal(tokensPerSecond(200, generationMs), 100); +}); + +test("#12616 tok/s is omitted when TTFT is unknown (do not use tokens/total_latency)", () => { + assert.equal(generationDurationMs(5000, null), null); + assert.equal(tokensPerSecond(200, null), null); + const usage = attachTokensPerSecond({ prompt_tokens: 10, completion_tokens: 200 }, null); + assert.equal((usage as { tokens_per_second?: number }).tokens_per_second, undefined); +}); + +test("#12616 tok/s is omitted when generation duration is not positive", () => { + assert.equal(generationDurationMs(3000, 3000), null); + assert.equal(generationDurationMs(2000, 3000), null); + assert.equal(tokensPerSecond(0, 2000), null); +}); + +test("#12616 filterUsageForFormat keeps tokens_per_second for OpenAI and Claude", () => { + const usage = { prompt_tokens: 10, completion_tokens: 20, tokens_per_second: 42.5 }; + const openai = filterUsageForFormat(usage, FORMATS.OPENAI) as Record; + const claude = filterUsageForFormat( + { input_tokens: 10, output_tokens: 20, tokens_per_second: 42.5 }, + FORMATS.CLAUDE + ) as Record; + assert.equal(openai.tokens_per_second, 42.5); + assert.equal(claude.tokens_per_second, 42.5); +}); + +test("#12616 headers omit tok/s without ttftMs and emit it when TTFT is known", () => { + const without = buildOmniRouteResponseMetaHeaders({ + provider: "openai", + model: "gpt-4o-mini", + latencyMs: 5000, + usage: { prompt_tokens: 11, completion_tokens: 200 }, + }); + assert.equal(without[OMNIROUTE_RESPONSE_HEADERS.tokensPerSecond], undefined); + + const withTtft = buildOmniRouteResponseMetaHeaders({ + provider: "openai", + model: "gpt-4o-mini", + latencyMs: 5000, + ttftMs: 3000, + usage: { prompt_tokens: 11, completion_tokens: 200 }, + }); + assert.equal(withTtft[OMNIROUTE_RESPONSE_HEADERS.tokensPerSecond], "100.000"); +}); + +test("#12616 SSE comment carries tok/s from usage.tokens_per_second when TTFT is unknown", () => { + const comment = buildOmniRouteSseMetadataComment({ + provider: "openai", + model: "gpt-4o-mini", + latencyMs: 50, + usage: { prompt_tokens: 4, completion_tokens: 2, tokens_per_second: 12.5 }, + }); + assert.match(comment, /^: x-omniroute-tokens-per-second=12.500/m); +}); + +test("#12616 StreamTiming.withTps attaches tok/s after first forward", async () => { + const t = createStreamTiming(); + t.markForward(); + await new Promise((r) => setTimeout(r, 25)); + const usage = t.withTps({ prompt_tokens: 1, completion_tokens: 100 }); + const tps = (usage as { tokens_per_second?: number }).tokens_per_second; + assert.equal(typeof tps, "number"); + assert.ok(tps! > 0); +}); From 8c1dfc416d1c458e4c5295d4e07b9041c8123a94 Mon Sep 17 00:00:00 2001 From: Ravi Tharuma <25951435+RaviTharuma@users.noreply.github.com> Date: Fri, 4 Sep 2026 04:38:57 +0200 Subject: [PATCH 086/143] fix(combo): do not treat credits-exhausted 401 as auth skip (#12449) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 10 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check-file-size` OK e **241/242** nos 29 arquivos de teste que os PRs tocam. A única "falha" não é falha: `tests/unit/autoCombo/strict-zero-cost-filter.test.ts` é um teste em estilo Vitest que eu incluí por engano na invocação do runner nativo do Node — ele quebra no import (`@vitest/runner`), não numa asserção. Ao investigar, descobri que esse arquivo não roda em nenhum dos dois runners hoje (o glob do `test:unit` não lista `autoCombo` e o `include` do Vitest só pega `.tsx` nessa pasta); é um problema pré-existente do repositório, sem relação com esta leva, e vou registrá-lo separadamente. O #12636 conflitava apenas na lista de testes do `@omniroute/opencode-plugin/package.json`, de forma aditiva: o tip já tinha `models-fetcher.test.ts` (do #12607, irmão desta mesma leva) e o #12636 acrescenta `telemetry.test.ts`. Fiz a união dos dois lados (25 arquivos contra 24 de cada) em vez de escolher um, o que teria removido um arquivo da suíte do plugin em silêncio. Obrigado, @RaviTharuma. --- open-sse/services/combo/targetExhaustion.ts | 21 +++ src/lib/embeddings/service.ts | 7 +- src/sse/handlers/chatHelpers.ts | 15 +- .../unit/12441-embeddings-credits-402.test.ts | 41 +++++ tests/unit/12441-quota-not-auth-skip.test.ts | 174 ++++++++++++++++++ tests/unit/chat-helpers.test.ts | 51 +++-- 6 files changed, 290 insertions(+), 19 deletions(-) create mode 100644 tests/unit/12441-embeddings-credits-402.test.ts create mode 100644 tests/unit/12441-quota-not-auth-skip.test.ts diff --git a/open-sse/services/combo/targetExhaustion.ts b/open-sse/services/combo/targetExhaustion.ts index 0325b64b97..11e8efa654 100644 --- a/open-sse/services/combo/targetExhaustion.ts +++ b/open-sse/services/combo/targetExhaustion.ts @@ -56,6 +56,25 @@ function isEmptyContentFailure(status: number, errorText: string): boolean { return status === 502 && (/empty content/i.test(errorText) || /empty response/i.test(errorText)); } +/** #12441 — quota/credits bodies must not take the 401/403 auth-skip path. */ +export function isQuotaOrCreditsError( + errorText: string, + structuredError?: { code?: string; type?: string; message?: string } +): boolean { + const blobs = [ + errorText, + structuredError?.type, + structuredError?.message, + structuredError?.code, + ].filter((value): value is string => Boolean(value)); + const joined = blobs.join(" "); + if (/credits exhausted/i.test(joined)) return true; + if (/quota exhausted/i.test(joined) && !/authentication expired/i.test(joined)) return true; + // Classify each candidate independently. A non-quota structuredError.code must + // not hide quota wording in errorText or structuredError.message. + return blobs.some((blob) => classifyErrorText(blob) === RateLimitReason.QUOTA_EXHAUSTED); +} + export type ComboExhaustionSets = { exhaustedProviders: Set; exhaustedConnections: Set; @@ -173,12 +192,14 @@ export function applyComboTargetExhaustion( .filter(Boolean) .join(" ") ); + const quotaMisclassifiedAsAuth = isQuotaOrCreditsError(errorText, structuredError); if ( AUTH_LEVEL_ERROR_STATUSES.includes(result.status) && // Cloudflare 1010 is a 403-ONLY fingerprint rejection. A 401 that merely happens to // mention "1010" or "fingerprint_rejection" in a port/count/model token must NOT skip // auth-level exhaustion — only a 403 carrying the Cloudflare fingerprint signal does. !(result.status === 403 && (fingerprintToken || fingerprintText)) && + !quotaMisclassifiedAsAuth && provider && provider !== "unknown" ) { diff --git a/src/lib/embeddings/service.ts b/src/lib/embeddings/service.ts index 7d6ee77727..7bf961f2af 100644 --- a/src/lib/embeddings/service.ts +++ b/src/lib/embeddings/service.ts @@ -320,9 +320,12 @@ export async function createEmbeddingResponse( ); } if ("allExpired" in credentials && credentials.allExpired) { + const expiredStatus = (credentials as { expiredStatus?: string }).expiredStatus; + const quota = expiredStatus === "credits_exhausted"; + const reason = quota ? "credits exhausted" : "authentication expired"; return errorResponse( - HTTP_STATUS.UNAUTHORIZED, - `[${provider}] All ${credentials.expiredCount || 1} connection(s) authentication expired — please reconnect in the dashboard` + quota ? HTTP_STATUS.PAYMENT_REQUIRED : HTTP_STATUS.UNAUTHORIZED, + `[${provider}] All ${credentials.expiredCount || 1} connection(s) ${reason} — please reconnect in the dashboard` ); } } else if (provider === "ollama-local" || provider === "lmstudio") { diff --git a/src/sse/handlers/chatHelpers.ts b/src/sse/handlers/chatHelpers.ts index ac53afc178..b98ee96542 100644 --- a/src/sse/handlers/chatHelpers.ts +++ b/src/sse/handlers/chatHelpers.ts @@ -777,9 +777,11 @@ export function handleNoCredentials( } if (credentials?.allExpired) { // Every connection for this provider is in a terminal state (expired, - // banned, or credits_exhausted). Surface as 401 with a re-auth hint - // instead of the generic 400 "No credentials", so dashboards/CLIs can - // distinguish "never configured" from "needs to reconnect". + // banned, or credits_exhausted). Surface expired/banned as 401 with a + // re-auth hint instead of the generic 400 "No credentials", so + // dashboards/CLIs can distinguish "never configured" from "needs to + // reconnect". credits_exhausted is quota (HTTP 402), not invalid + // credentials — see #12441. const status = credentials.expiredStatus || "expired"; const count = credentials.expiredCount || 1; const reason = @@ -790,7 +792,12 @@ export function handleNoCredentials( : "authentication expired"; const message = `[${provider}] All ${count} connection(s) ${reason} — please reconnect in the dashboard`; log.warn("CHAT", message); - return errorResponse(HTTP_STATUS.UNAUTHORIZED, message); + // #12441: credits_exhausted is quota, not invalid credentials. Combo + // dispatch treats 401 as AUTH_LEVEL skip (#8133). Surface 402 so quota + // exhaustion follows the #1731 path instead of "authentication expired". + const httpStatus = + status === "credits_exhausted" ? HTTP_STATUS.PAYMENT_REQUIRED : HTTP_STATUS.UNAUTHORIZED; + return errorResponse(httpStatus, message); } if (!excludeConnectionId) { // Ported from upstream decolua/9router#336 (Ibrahim Ryan): surface as 404 diff --git a/tests/unit/12441-embeddings-credits-402.test.ts b/tests/unit/12441-embeddings-credits-402.test.ts new file mode 100644 index 0000000000..b3eed15ec4 --- /dev/null +++ b/tests/unit/12441-embeddings-credits-402.test.ts @@ -0,0 +1,41 @@ +/** + * #12441 — embeddings credential exhaustion with credits_exhausted must + * surface HTTP 402, matching handleNoCredentials. + */ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-embed-credits-402-")); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.API_KEY_SECRET = process.env.API_KEY_SECRET || "embed-credits-402-test-secret"; + +const core = await import("../../src/lib/db/core.ts"); +const providersDb = await import("../../src/lib/db/providers.ts"); +const { createEmbeddingResponse } = await import("../../src/lib/embeddings/service.ts"); + +test.after(() => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("createEmbeddingResponse maps allExpired+credits_exhausted to HTTP 402", async () => { + await providersDb.createProviderConnection({ + provider: "mistral", + authType: "apikey", + apiKey: "mistral-exhausted-key", + isActive: true, + testStatus: "credits_exhausted", + }); + + const res = await createEmbeddingResponse({ + model: "mistral/mistral-embed", + input: "hello", + }); + const body = (await res.json()) as { error?: { message?: string } }; + + assert.equal(res.status, 402); + assert.match(String(body.error?.message || ""), /credits exhausted/i); +}); diff --git a/tests/unit/12441-quota-not-auth-skip.test.ts b/tests/unit/12441-quota-not-auth-skip.test.ts new file mode 100644 index 0000000000..c3b976fa35 --- /dev/null +++ b/tests/unit/12441-quota-not-auth-skip.test.ts @@ -0,0 +1,174 @@ +/** + * #12441 — credits-exhausted / quota bodies restated as HTTP 401 must not + * take the combo AUTH_LEVEL skip path (#8133 authentication expired). + */ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + applyComboTargetExhaustion, + isQuotaOrCreditsError, + type ComboExhaustionSets, +} from "../../open-sse/services/combo/targetExhaustion.ts"; + +function emptySets(): ComboExhaustionSets { + return { + exhaustedProviders: new Set(), + exhaustedConnections: new Set(), + transientRateLimitedProviders: new Set(), + }; +} + +const log = { info() {}, warn() {}, error() {}, debug() {} }; + +function chutesTarget() { + return { + kind: "model", + executionKey: "ek", + modelStr: "chutes/moonshotai/Kimi-K3-TEE", + provider: "chutes", + providerId: null, + connectionId: "conn-chutes-1", + } as Parameters[0]; +} + +test("#12441 isQuotaOrCreditsError detects credits-exhausted 401 bodies", () => { + assert.equal( + isQuotaOrCreditsError( + "[chutes] All 3 connection(s) credits exhausted — please reconnect in the dashboard" + ), + true + ); + assert.equal( + isQuotaOrCreditsError( + "[claude] All 1 connection(s) authentication expired — please reconnect in the dashboard" + ), + false + ); +}); + +test("#12441 isQuotaOrCreditsError still matches quota text when structuredError.code is non-quota", () => { + assert.equal( + isQuotaOrCreditsError("generic upstream failure", { + code: "invalid_request", + type: "api_error", + message: "You've reached your usage limit for this billing cycle", + }), + true + ); + assert.equal( + isQuotaOrCreditsError( + "[chutes] All 3 connection(s) credits exhausted — please reconnect in the dashboard", + { code: "unauthorized", type: "auth_error" } + ), + true + ); + assert.equal( + isQuotaOrCreditsError("generic upstream failure", { + code: "unauthorized", + message: "authentication expired", + }), + false + ); +}); + +test("#12441 credits-exhausted HTTP 401 does not mark auth-level connection skip", () => { + const s = emptySets(); + const exhausted = applyComboTargetExhaustion(chutesTarget(), { + result: { status: 401 }, + fallbackResult: {}, + errorText: "[chutes] All 3 connection(s) credits exhausted — please reconnect in the dashboard", + rawModel: "moonshotai/Kimi-K3-TEE", + isTokenLimitBreach: false, + allAccountsRateLimited: false, + requestScopedFailure: false, + sets: s, + log, + tag: "COMBO", + exhaustedLogLevel: "info", + }); + assert.equal(exhausted, false); + assert.equal(s.exhaustedConnections.has("chutes:conn-chutes-1"), false); + assert.equal(s.exhaustedProviders.has("chutes"), false); +}); + +test("#12441 structuredError quota message with a non-quota code does not auth-skip", () => { + const s = emptySets(); + const exhausted = applyComboTargetExhaustion(chutesTarget(), { + result: { status: 401 }, + fallbackResult: {}, + errorText: "generic upstream failure", + rawModel: "moonshotai/Kimi-K3-TEE", + isTokenLimitBreach: false, + allAccountsRateLimited: false, + requestScopedFailure: false, + sets: s, + log, + tag: "COMBO", + exhaustedLogLevel: "info", + structuredError: { + code: "invalid_request", + message: "You've reached your usage limit for this billing cycle", + }, + }); + assert.equal(exhausted, false); + assert.equal(s.exhaustedConnections.has("chutes:conn-chutes-1"), false); +}); + +test("#12441 real authentication expired 401 still marks the connection", () => { + const s = emptySets(); + const exhausted = applyComboTargetExhaustion(chutesTarget(), { + result: { status: 401 }, + fallbackResult: {}, + errorText: + "[claude] All 1 connection(s) authentication expired — please reconnect in the dashboard", + rawModel: "claude-opus-4-8", + isTokenLimitBreach: false, + allAccountsRateLimited: false, + requestScopedFailure: false, + sets: s, + log, + tag: "COMBO", + exhaustedLogLevel: "info", + }); + assert.equal(exhausted, true); + assert.equal(s.exhaustedConnections.has("chutes:conn-chutes-1"), true); +}); + +function quotaTarget() { + return { + kind: "model", + executionKey: "ek", + modelStr: "test-dedup-provider/m1", + provider: "test-dedup-provider", + providerId: null, + connectionId: "conn-1", + } as Parameters[0]; +} + +test("#12441 credits-exhausted 401 on a non-passthrough provider takes quota skip (#1731) not auth skip (#8133)", () => { + const s = emptySets(); + const exhausted = applyComboTargetExhaustion(quotaTarget(), { + result: { status: 401 }, + fallbackResult: {}, + errorText: "[chutes] All 3 connection(s) credits exhausted — please reconnect in the dashboard", + rawModel: "m1", + isTokenLimitBreach: false, + allAccountsRateLimited: false, + requestScopedFailure: false, + sets: s, + log, + tag: "COMBO", + exhaustedLogLevel: "info", + }); + assert.equal(exhausted, true); + assert.equal( + s.exhaustedConnections.has("test-dedup-provider:conn-1"), + false, + "must not take the #8133 auth-level connection skip" + ); + assert.equal( + s.exhaustedProviders.has("test-dedup-provider"), + true, + "quota body restated as 401 must still follow #1731 provider skip" + ); +}); diff --git a/tests/unit/chat-helpers.test.ts b/tests/unit/chat-helpers.test.ts index a971101891..710bb6f21c 100644 --- a/tests/unit/chat-helpers.test.ts +++ b/tests/unit/chat-helpers.test.ts @@ -25,6 +25,16 @@ const { getCircuitBreaker, resetAllCircuitBreakers, STATE } = // DATA_DIR must be fixed before these modules load; keep this test seam dynamic. const { setTlsClientForTest } = await import("../../open-sse/utils/proxyFetch.ts"); +type ApiErrorJson = { + error?: { + message?: string; + code?: string; + type?: string; + model?: string; + reset_seconds?: number; + }; +}; + async function resetStorage() { resetAllCircuitBreakers(); core.resetDbInstance(); @@ -85,7 +95,7 @@ test("resolveModelOrError rejects unknown built-in auto catalog ids", async () = assert.ok(result.error); assert.equal(result.error.status, 400); - const json = (await result.error.json()) as any; + const json = (await result.error.json()) as ApiErrorJson; assert.match(json.error.message, /Unknown built-in auto combo/i); }); @@ -120,7 +130,7 @@ test("resolveModelOrError rejects ambiguous aliases without a provider prefix", assert.ok(result.error); assert.equal(result.error.status, 400); - const json = (await result.error.json()) as any; + const json = (await result.error.json()) as ApiErrorJson; assert.match(json.error.message, /Ambiguous model/i); }); @@ -133,7 +143,7 @@ test("resolveModelOrError rejects ambiguous slashful canonical ids instead of mi assert.ok(result.error); assert.equal(result.error.status, 400); - const json = (await result.error.json()) as any; + const json = (await result.error.json()) as ApiErrorJson; assert.match(json.error.message, /Ambiguous model/i); assert.match(json.error.message, /openai\/gpt-oss-120b/i); }); @@ -147,7 +157,7 @@ test("resolveModelOrError rejects malformed model strings", async () => { assert.ok(result.error); assert.equal(result.error.status, 400); - const json = (await result.error.json()) as any; + const json = (await result.error.json()) as ApiErrorJson; assert.match(json.error.message, /Invalid model format/i); }); @@ -261,7 +271,7 @@ test("checkPipelineGates blocks providers with an open circuit breaker", async ( resetTimeoutMs: 5_000, }, }); - const json = (await response.json()) as any; + const json = (await response.json()) as ApiErrorJson; const retryAfter = Number(response.headers.get("Retry-After")); assert.equal(response.status, 503); @@ -329,8 +339,8 @@ test("handleNoCredentials reports missing provider credentials and exhausted acc 500 ); - const missingJson = (await missing.json()) as any; - const exhaustedJson = (await exhausted.json()) as any; + const missingJson = (await missing.json()) as ApiErrorJson; + const exhaustedJson = (await exhausted.json()) as ApiErrorJson; assert.equal(missing.status, 404); assert.match(missingJson.error.message, /No active credentials for provider: openai/); @@ -413,7 +423,7 @@ test("handleNoCredentials returns Retry-After when every account is rate limited null, null ); - const json = (await response.json()) as any; + const json = (await response.json()) as ApiErrorJson; assert.equal(response.status, 429); assert.ok(Number(response.headers.get("Retry-After")) >= 1); @@ -438,7 +448,7 @@ test("handleNoCredentials returns structured model_cooldown when every credentia null, null ); - const json = (await response.json()) as any; + const json = (await response.json()) as ApiErrorJson; assert.equal(response.status, 429); assert.equal(Number(response.headers.get("Retry-After")) >= 1, true); @@ -461,7 +471,7 @@ test("handleNoCredentials returns 401 with re-auth hint when every connection is null, null ); - const json = (await response.json()) as any; + const json = (await response.json()) as ApiErrorJson; assert.equal(response.status, 401); assert.match(json.error.message, /\[kiro\]/); @@ -478,12 +488,27 @@ test("handleNoCredentials maps allExpired status='expired' to the 'authenticatio null, null ); - const json = (await response.json()) as any; + const json = (await response.json()) as ApiErrorJson; assert.equal(response.status, 401); assert.match(json.error.message, /3 connection\(s\) authentication expired/); }); +test("handleNoCredentials maps credits_exhausted to HTTP 402 not 401 (#12441)", async () => { + const response = handleNoCredentials( + { allExpired: true, expiredCount: 3, expiredStatus: "credits_exhausted" }, + null, + "chutes", + "moonshotai/Kimi-K3-TEE", + null, + null + ); + const json = (await response.json()) as ApiErrorJson; + + assert.equal(response.status, 402); + assert.match(json.error.message, /3 connection\(s\) credits exhausted/); +}); + test("handleNoCredentials preserves lastError over allExpired after a failed attempt", async () => { const response = handleNoCredentials( { allExpired: true, expiredCount: 1, expiredStatus: "credits_exhausted" }, @@ -501,7 +526,7 @@ test("handleNoCredentials preserves lastError over allExpired after a failed att test("safeResolveProxy returns the direct route when no proxy config is present", async () => { const connection = await seedConnection("openai", { apiKey: "sk-openai-direct" }); - const resolved = await safeResolveProxy((connection as any).id); + const resolved = await safeResolveProxy((connection as { id: string }).id); assert.deepEqual(resolved, { proxy: null, @@ -693,7 +718,7 @@ test("resolveModelOrError returns model_not_found error for unrecognised bare mo assert.ok(result.error); assert.equal(result.error.status, 400); - const json = (await result.error.json()) as any; + const json = (await result.error.json()) as ApiErrorJson; assert.match(json.error.message, /Unable to determine provider/i); assert.match(json.error.message, /completely-unknown-model-xyz/i); }); From 85b8d128eb4a1788121496ea1adc701f9cb3207d Mon Sep 17 00:00:00 2001 From: Ravi Tharuma <25951435+RaviTharuma@users.noreply.github.com> Date: Fri, 4 Sep 2026 04:39:22 +0200 Subject: [PATCH 087/143] fix(auth): do not park healthy quota accounts as expired (#12452) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 10 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check-file-size` OK e **241/242** nos 29 arquivos de teste que os PRs tocam. A única "falha" não é falha: `tests/unit/autoCombo/strict-zero-cost-filter.test.ts` é um teste em estilo Vitest que eu incluí por engano na invocação do runner nativo do Node — ele quebra no import (`@vitest/runner`), não numa asserção. Ao investigar, descobri que esse arquivo não roda em nenhum dos dois runners hoje (o glob do `test:unit` não lista `autoCombo` e o `include` do Vitest só pega `.tsx` nessa pasta); é um problema pré-existente do repositório, sem relação com esta leva, e vou registrá-lo separadamente. O #12636 conflitava apenas na lista de testes do `@omniroute/opencode-plugin/package.json`, de forma aditiva: o tip já tinha `models-fetcher.test.ts` (do #12607, irmão desta mesma leva) e o #12636 acrescenta `telemetry.test.ts`. Fiz a união dos dois lados (25 arquivos contra 24 de cada) em vez de escolher um, o que teria removido um arquivo da suíte do plugin em silêncio. Obrigado, @RaviTharuma. --- open-sse/services/errorClassifier.ts | 2 +- src/lib/quota/connectionRecovery.ts | 36 ++++++- src/lib/tokenHealthCheck.ts | 20 ++-- src/sse/services/auth.ts | 81 +++------------ src/sse/services/authTerminalStatus.ts | 93 +++++++++++++++++ stryker.conf.json | 1 + tests/unit/false-terminal-401-quota.test.ts | 100 +++++++++++++++++++ tests/unit/quota-connection-recovery.test.ts | 34 +++++++ tests/unit/token-health-check-cursor.test.ts | 6 +- 9 files changed, 289 insertions(+), 84 deletions(-) create mode 100644 src/sse/services/authTerminalStatus.ts create mode 100644 tests/unit/false-terminal-401-quota.test.ts diff --git a/open-sse/services/errorClassifier.ts b/open-sse/services/errorClassifier.ts index 2bdbbbc8c2..5e60017662 100644 --- a/open-sse/services/errorClassifier.ts +++ b/open-sse/services/errorClassifier.ts @@ -256,7 +256,7 @@ export function classifyProviderError( const oauthInvalid = isOAuthInvalidToken(bodyStr); const preserveQuota429 = shouldPreserveQuotaSignalsFor429(provider); - if ((creditsExhausted || subscriptionQuotaExhausted) && [400, 402, 403].includes(statusCode)) { + if ((creditsExhausted || subscriptionQuotaExhausted) && [400, 401, 402, 403].includes(statusCode)) { return PROVIDER_ERROR_TYPES.QUOTA_EXHAUSTED; } diff --git a/src/lib/quota/connectionRecovery.ts b/src/lib/quota/connectionRecovery.ts index e7c09be1bb..26952a262d 100644 --- a/src/lib/quota/connectionRecovery.ts +++ b/src/lib/quota/connectionRecovery.ts @@ -63,6 +63,7 @@ export interface RecoverableConnectionInput { testStatus?: string | null; rateLimitedUntil?: string | null; lastErrorAt?: string | null; + lastErrorType?: string | null; } function normalizeStatus(value: string | null | undefined): string { @@ -158,6 +159,37 @@ export function isRecoverableCooldownConnection( * * Pure — `nowMs` and `reprobeMs` are injected so callers/tests control the clock. */ + +const EXPIRED_REPROBE_BLOCKLIST = new Set([ + "account_deactivated", + "invalid_grant", + "unrecoverable_refresh_error", + "provider_deprecated", + "no_refresh_token", +]); + +/** + * Re-probe `expired` after the same window as credits_exhausted. + * API-key 401s and OAuth races were persisted as expired and then never + * retried (combo pre-skip + health-check skip). Do not reopen a real + * deactivation / invalid_grant. + */ +export function isExpiredReprobeCandidate( + connection: RecoverableConnectionInput | null | undefined, + nowMs: number, + reprobeMs: number = DEFAULT_CREDITS_REPROBE_MS +): boolean { + if (!connection || typeof connection.id !== "string" || connection.id.length === 0) { + return false; + } + if (normalizeStatus(connection.testStatus) !== "expired") return false; + const err = (connection.lastErrorType || "").trim().toLowerCase(); + if (EXPIRED_REPROBE_BLOCKLIST.has(err)) return false; + const sinceMs = cooldownUntilMs(connection.lastErrorAt || connection.rateLimitedUntil || ""); + if (!Number.isFinite(sinceMs) || sinceMs <= 0) return true; + return nowMs - sinceMs >= reprobeMs; +} + export function isCreditsExhaustedReprobeCandidate( connection: RecoverableConnectionInput | null | undefined, nowMs: number, @@ -188,7 +220,8 @@ export function selectRecoverableConnections isRecoverableCooldownConnection(connection, nowMs) || - isCreditsExhaustedReprobeCandidate(connection, nowMs) + isCreditsExhaustedReprobeCandidate(connection, nowMs) || + isExpiredReprobeCandidate(connection, nowMs) ); } @@ -245,6 +278,7 @@ export async function runConnectionRecoveryTick( testStatus: typeof row.testStatus === "string" ? row.testStatus : null, rateLimitedUntil: typeof row.rateLimitedUntil === "string" ? row.rateLimitedUntil : null, lastErrorAt: typeof row.lastErrorAt === "string" ? row.lastErrorAt : null, + lastErrorType: typeof row.lastErrorType === "string" ? row.lastErrorType : null, })); }); connections = await load(); diff --git a/src/lib/tokenHealthCheck.ts b/src/lib/tokenHealthCheck.ts index 2820a28e23..eff72dfd9c 100644 --- a/src/lib/tokenHealthCheck.ts +++ b/src/lib/tokenHealthCheck.ts @@ -564,18 +564,11 @@ export async function checkConnection(conn) { } } - // #8182: skip terminal connections (credits_exhausted / banned / expired). - // These can never self-heal via a token refresh — probing them wastes - // CPU and network on every sweep cycle. Mirrors isTerminalConnectionStatus - // in src/sse/services/auth.ts and TERMINAL_CONNECTION_STATUSES in - // src/lib/quota/connectionRecovery.ts. - // - // #5326 exception: a GitHub Copilot access-token-only connection parked in - // "expired" with errorCode "no_refresh_token" is NOT actually terminal — it's - // the exact target of the self-heal below (canClearGitHubNoRefreshTokenState), - // which clears that stale status back to "active" once the Copilot sub-token - // proves usable. Treating it as terminal here made that self-heal unreachable, - // leaving healthy Copilot connections stuck at "expired" forever. + // #8182: skip banned/expired (dead credentials). credits_exhausted is a + // renewing window — keep sweeping so OAuth refresh can clear a false mark. + // #5326: GitHub Copilot access-token-only "expired" + no_refresh_token is + // the self-heal target below (canClearGitHubNoRefreshTokenState). Treating + // it as terminal made that heal unreachable and stuck healthy Copilot rows. const isRecoverableGithubCopilotNoRefresh = conn.testStatus === "expired" && conn.errorCode === "no_refresh_token" && @@ -596,7 +589,8 @@ export async function checkConnection(conn) { conn.testStatus === "expired" && conn.lastErrorType !== "account_deactivated" && getExpiredRetryCount(conn) < EXPIRED_RETRY_MAX; - const terminalStatuses = new Set(["credits_exhausted", "banned", "expired"]); + // Skip only banned/expired. Combo pre-skip still hides exhausted rows. + const terminalStatuses = new Set(["banned", "expired"]); if ( typeof conn.testStatus === "string" && terminalStatuses.has(conn.testStatus.toLowerCase()) && diff --git a/src/sse/services/auth.ts b/src/sse/services/auth.ts index 37029593af..9cdd16a6e3 100644 --- a/src/sse/services/auth.ts +++ b/src/sse/services/auth.ts @@ -94,6 +94,7 @@ import { classifyProviderError, PROVIDER_ERROR_TYPES, } from "@omniroute/open-sse/services/errorClassifier.ts"; +import { resolveTerminalConnectionStatus } from "./authTerminalStatus.ts"; import { ALIBABA_FREE_DRAINED_LOCK_MS, getAlibabaBillingMode, @@ -326,71 +327,6 @@ function isTerminalConnectionStatusForModel( return true; } -// #8200: cookie-auth providers (perplexity-web, grok-web, ...) use a rotating browser -// session, not a static API key — a 401 means "session needs a refresh", not "dead". -function isRecoverableCookieAuth401( - provider: string | null, - providerErrorType: string | null -): boolean { - return ( - providerErrorType !== PROVIDER_ERROR_TYPES.ACCOUNT_DEACTIVATED && - provider != null && - resolveProviderId(provider) in WEB_COOKIE_PROVIDERS - ); -} -// #12242 (402 variant of #3027): a bare 402 on a passthrough/gateway -// provider that multiplexes many models behind one credential -// (kilo-gateway, ollama-cloud, etc.) is a PER-MODEL billing signal, not -// proof the credential itself is dead — free models on the same connection -// remain perfectly usable. Only terminalize the whole connection for a 402 -// when the provider is NOT a per-model-quota provider; the caller lets it -// fall through to the per-model lockout branch instead. -// `result.creditsExhausted` is a provider's own explicit classification -// (independent of HTTP status) and stays unconditionally terminal — it is -// not scoped by this check. -function isConnectionWideCreditsExhausted( - status: number, - result: { permanent?: boolean; creditsExhausted?: boolean }, - isPerModelQuotaProvider: boolean -): boolean { - return result.creditsExhausted || (status === 402 && !isPerModelQuotaProvider); -} -function resolveTerminalConnectionStatus( - status: number, - result: { permanent?: boolean; creditsExhausted?: boolean }, - providerErrorType: string | null = null, - provider: string | null = null, - isPerModelQuotaProvider = false -): string | null { - if (isConnectionWideCreditsExhausted(status, result, isPerModelQuotaProvider)) { - return "credits_exhausted"; - } - if ( - providerErrorType === PROVIDER_ERROR_TYPES.PROJECT_ROUTE_ERROR || - providerErrorType === PROVIDER_ERROR_TYPES.GEO_BLOCKED || - providerErrorType === PROVIDER_ERROR_TYPES.OAUTH_INVALID_TOKEN || - // #1010: Cloudflare fingerprint rejection is the CDN refusing the CLIENT's - // signature, not the account's credentials — never a terminal account state. - // A different client on the same key succeeds (measured 2026-08-08: curl 200, - // urllib 403 on byte-identical body), so banning the account here would flip a - // healthy free pool to ALL_ACCOUNTS_INACTIVE after two such calls. - providerErrorType === PROVIDER_ERROR_TYPES.FINGERPRINT_REJECTION - ) { - return null; - } - if (result.permanent || providerErrorType === PROVIDER_ERROR_TYPES.FORBIDDEN) { - return "banned"; - } - if ( - (providerErrorType === PROVIDER_ERROR_TYPES.ACCOUNT_DEACTIVATED || - providerErrorType === PROVIDER_ERROR_TYPES.UNAUTHORIZED || - status === 401) && - !isRecoverableCookieAuth401(provider, providerErrorType) - ) { - return "expired"; - } - return null; -} export function resolveQuotaLimitPolicy( provider: string, providerSpecificData: JsonRecord @@ -3038,13 +2974,24 @@ export async function markAccountUnavailable( return { shouldFallback: true, cooldownMs: lockout.cooldownMs }; } - const terminalStatus = resolveTerminalConnectionStatus( + let terminalStatus = resolveTerminalConnectionStatus( status, result as { permanent?: boolean; creditsExhausted?: boolean }, providerErrorType, provider, - isPerModelQuotaProvider + isPerModelQuotaProvider, + errorText ); + // A still-valid access token after a successful refresh is not "expired". + // A follow-up 401 (timeout, hop, race) must cooldown, not park the account. + const tokenExpiryMs = Date.parse(String(conn?.tokenExpiresAt || conn?.expiresAt || "")); + if ( + terminalStatus === "expired" && + Number.isFinite(tokenExpiryMs) && + tokenExpiryMs > Date.now() + 60_000 + ) { + terminalStatus = null; + } const cachedQuotaResetAt = providerErrorType === PROVIDER_ERROR_TYPES.QUOTA_EXHAUSTED || reason === RateLimitReason.QUOTA_EXHAUSTED diff --git a/src/sse/services/authTerminalStatus.ts b/src/sse/services/authTerminalStatus.ts new file mode 100644 index 0000000000..b8afed537b --- /dev/null +++ b/src/sse/services/authTerminalStatus.ts @@ -0,0 +1,93 @@ +import { PROVIDER_ERROR_TYPES } from "@omniroute/open-sse/services/errorClassifier.ts"; +import { isCreditsExhausted } from "@omniroute/open-sse/services/accountFallback.ts"; +import { resolveProviderId, WEB_COOKIE_PROVIDERS } from "@/shared/constants/providers"; + +// #8200: cookie-auth providers (perplexity-web, grok-web, ...) use a rotating browser +// session, not a static API key — a 401 means "session needs a refresh", not "dead". +export function isRecoverableCookieAuth401( + provider: string | null, + providerErrorType: string | null +): boolean { + return ( + providerErrorType !== PROVIDER_ERROR_TYPES.ACCOUNT_DEACTIVATED && + provider != null && + resolveProviderId(provider) in WEB_COOKIE_PROVIDERS + ); +} +// #12242 (402 variant of #3027): a bare 402 on a passthrough/gateway +// provider that multiplexes many models behind one credential +// (kilo-gateway, ollama-cloud, etc.) is a PER-MODEL billing signal, not +// proof the credential itself is dead — free models on the same connection +// remain perfectly usable. Only terminalize the whole connection for a 402 +// when the provider is NOT a per-model-quota provider; the caller lets it +// fall through to the per-model lockout branch instead. +// `result.creditsExhausted` is a provider's own explicit classification +// (independent of HTTP status) and stays unconditionally terminal — it is +// not scoped by this check. +export function isConnectionWideCreditsExhausted( + status: number, + result: { permanent?: boolean; creditsExhausted?: boolean }, + isPerModelQuotaProvider: boolean +): boolean { + return result.creditsExhausted || (status === 402 && !isPerModelQuotaProvider); +} + +/** Credits-depleted bodies park; renewing billing-cycle quota does not. */ +export function shouldParkCreditsExhausted( + status: number, + result: { permanent?: boolean; creditsExhausted?: boolean }, + isPerModelQuotaProvider: boolean, + errorText: string +): boolean { + return ( + isConnectionWideCreditsExhausted(status, result, isPerModelQuotaProvider) || + (!isPerModelQuotaProvider && isCreditsExhausted(errorText)) + ); +} + +function isNonTerminalProviderError(providerErrorType: string | null): boolean { + return ( + providerErrorType === PROVIDER_ERROR_TYPES.PROJECT_ROUTE_ERROR || + providerErrorType === PROVIDER_ERROR_TYPES.GEO_BLOCKED || + providerErrorType === PROVIDER_ERROR_TYPES.OAUTH_INVALID_TOKEN || + // #1010: Cloudflare fingerprint rejection is the CDN refusing the CLIENT's + // signature, not the account's credentials — never a terminal account state. + providerErrorType === PROVIDER_ERROR_TYPES.FINGERPRINT_REJECTION + ); +} + +function isExpiredAuthFailure( + status: number, + providerErrorType: string | null, + provider: string | null +): boolean { + return ( + (providerErrorType === PROVIDER_ERROR_TYPES.ACCOUNT_DEACTIVATED || + providerErrorType === PROVIDER_ERROR_TYPES.UNAUTHORIZED || + status === 401) && + !isRecoverableCookieAuth401(provider, providerErrorType) + ); +} + +export function resolveTerminalConnectionStatus( + status: number, + result: { permanent?: boolean; creditsExhausted?: boolean }, + providerErrorType: string | null = null, + provider: string | null = null, + isPerModelQuotaProvider = false, + errorText: string = "" +): string | null { + if (shouldParkCreditsExhausted(status, result, isPerModelQuotaProvider, errorText)) { + return "credits_exhausted"; + } + if (isNonTerminalProviderError(providerErrorType)) { + return null; + } + if (result.permanent || providerErrorType === PROVIDER_ERROR_TYPES.FORBIDDEN) { + return "banned"; + } + if (isExpiredAuthFailure(status, providerErrorType, provider)) { + return "expired"; + } + return null; +} diff --git a/stryker.conf.json b/stryker.conf.json index 27e2825662..c2986bcb28 100644 --- a/stryker.conf.json +++ b/stryker.conf.json @@ -251,6 +251,7 @@ "tests/unit/executor-contract-violation-terminal.test.ts", "tests/unit/executor-devin-cli-agentic-acp.test.ts", "tests/unit/executor-web-cookie-sweep.test.ts", + "tests/unit/false-terminal-401-quota.test.ts", "tests/unit/format-provider-error-cause.test.ts", "tests/unit/forwarded-header-budget.test.ts", "tests/unit/fusion-vision-panel-3378.test.ts", diff --git a/tests/unit/false-terminal-401-quota.test.ts b/tests/unit/false-terminal-401-quota.test.ts new file mode 100644 index 0000000000..393e2fdd68 --- /dev/null +++ b/tests/unit/false-terminal-401-quota.test.ts @@ -0,0 +1,100 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-false-terminal-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const providersDb = await import("../../src/lib/db/providers.ts"); +const auth = await import("../../src/sse/services/auth.ts"); + +async function resetStorage() { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +test.after(() => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("401 credits-exhausted body is credits_exhausted, not expired", async () => { + await resetStorage(); + const conn = await providersDb.createProviderConnection({ + provider: "chutes", + authType: "apikey", + apiKey: "sk-chutes-live", + isActive: true, + testStatus: "active", + }); + const connId = String(conn.id); + await auth.markAccountUnavailable( + connId, + 401, + "[chutes] All 3 connection(s) credits exhausted — please reconnect in the dashboard", + "chutes", + "moonshotai/Kimi-K3-TEE" + ); + const after = await providersDb.getProviderConnectionById(connId); + assert.equal(after.testStatus, "credits_exhausted"); + assert.notEqual(after.testStatus, "expired"); +}); + +test("billing-cycle quota 403 stays unavailable until the cached reset", async () => { + await resetStorage(); + const quotaCache = await import("../../src/domain/quotaCache.ts"); + const conn = await providersDb.createProviderConnection({ + provider: "kimi-coding", + authType: "oauth", + accessToken: "kimi-access-token", + refreshToken: "kimi-refresh-token", + isActive: true, + testStatus: "active", + }); + const connId = String(conn.id); + const resetAt = new Date(Date.now() + 30 * 60 * 1000).toISOString(); + quotaCache.setQuotaCache(connId, "kimi-coding", { + Ratelimit: { remainingPercentage: 0, resetAt }, + Weekly: { + remainingPercentage: 62, + resetAt: new Date(Date.now() + 5 * 24 * 60 * 60 * 1000).toISOString(), + }, + }); + const result = await auth.markAccountUnavailable( + connId, + 403, + "You've reached your usage limit for this billing cycle. Your quota will be refreshed in the next cycle.", + "kimi-coding", + "kimi-for-coding" + ); + const after = await providersDb.getProviderConnectionById(connId); + assert.equal(result.shouldFallback, true); + assert.ok(Math.abs(result.cooldownMs - 30 * 60 * 1000) < 2_000); + assert.equal(after.testStatus, "unavailable"); + assert.notEqual(after.testStatus, "credits_exhausted"); + assert.equal(after.lastErrorType, "quota_exhausted"); + quotaCache.__clearForTests(); +}); + +test("401 with a still-valid access token does not expire the connection", async () => { + await resetStorage(); + const conn = await providersDb.createProviderConnection({ + provider: "claude", + authType: "oauth", + accessToken: "sk-ant-fresh", + refreshToken: "rt-fresh", + isActive: true, + testStatus: "active", + tokenExpiresAt: new Date(Date.now() + 8 * 60 * 60 * 1000).toISOString(), + expiresAt: new Date(Date.now() + 8 * 60 * 60 * 1000).toISOString(), + }); + const connId = String(conn.id); + await auth.markAccountUnavailable(connId, 401, "unauthorized", "claude", "claude-opus-4-8"); + const after = await providersDb.getProviderConnectionById(connId); + assert.notEqual(after.testStatus, "expired"); + assert.equal(after.isActive, true); +}); diff --git a/tests/unit/quota-connection-recovery.test.ts b/tests/unit/quota-connection-recovery.test.ts index fa96ae8f93..ddd535c4e1 100644 --- a/tests/unit/quota-connection-recovery.test.ts +++ b/tests/unit/quota-connection-recovery.test.ts @@ -3,6 +3,7 @@ import assert from "node:assert/strict"; import { CREDITS_EXHAUSTED_STATUS, isCreditsExhaustedReprobeCandidate, + isExpiredReprobeCandidate, isRecoverableCooldownConnection, selectRecoverableConnections, runConnectionRecoveryTick, @@ -216,3 +217,36 @@ describe("connectionRecovery — mixed timestamp encodings", () => { assert.equal(isRecoverableCooldownConnection(conn, nowMs), false); }); }); + +describe("connectionRecovery — expired reprobe", () => { + const nowMs = 1_700_000_000_000; + const thirtyMinMs = 30 * 60 * 1000; + + it("should reprobe expired after 30m unless lastErrorType is a real deactivation", () => { + const old = { + id: "e-1", + testStatus: "expired", + lastErrorAt: new Date(nowMs - thirtyMinMs - 1000).toISOString(), + lastErrorType: "unauthorized", + }; + assert.equal(isExpiredReprobeCandidate(old, nowMs), true); + assert.equal( + isExpiredReprobeCandidate({ ...old, lastErrorType: "invalid_grant" }, nowMs), + false + ); + }); + + it("selectRecoverableConnections includes stale expired rows", () => { + const selected = selectRecoverableConnections( + [ + { + id: "e-1", + testStatus: "expired", + lastErrorAt: new Date(nowMs - thirtyMinMs - 1000).toISOString(), + }, + ], + nowMs + ); + assert.deepEqual(selected.map((c) => c.id), ["e-1"]); + }); +}); diff --git a/tests/unit/token-health-check-cursor.test.ts b/tests/unit/token-health-check-cursor.test.ts index f05933e2bb..12284bc3ae 100644 --- a/tests/unit/token-health-check-cursor.test.ts +++ b/tests/unit/token-health-check-cursor.test.ts @@ -430,7 +430,7 @@ test("checkConnection: a banned Cursor connection stays skipped regardless of la }); }); -test("checkConnection: a credits_exhausted Cursor connection stays skipped regardless of lastErrorType", async () => { +test("checkConnection: a credits_exhausted Cursor connection is still swept", async () => { await resetStorage(); await withCursorEnv(async () => { const id = await createCursorConnection({ @@ -444,7 +444,9 @@ test("checkConnection: a credits_exhausted Cursor connection stays skipped regar await tokenHealthCheck.checkConnection(before); const after = await freshConn(id); - assert.deepEqual(after, before); + assert.notEqual(after.testStatus, "credits_exhausted"); + assert.ok(after.lastHealthCheckAt); + assert.notEqual(after.updatedAt, before.updatedAt); }); }); From 6aa3690dea6d8f62474bad737071a33c219b39d9 Mon Sep 17 00:00:00 2001 From: Ravi Tharuma <25951435+RaviTharuma@users.noreply.github.com> Date: Fri, 4 Sep 2026 04:39:43 +0200 Subject: [PATCH 088/143] feat(providers): filter GitHub combo members against live catalog (#12473) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 10 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check-file-size` OK e **241/242** nos 29 arquivos de teste que os PRs tocam. A única "falha" não é falha: `tests/unit/autoCombo/strict-zero-cost-filter.test.ts` é um teste em estilo Vitest que eu incluí por engano na invocação do runner nativo do Node — ele quebra no import (`@vitest/runner`), não numa asserção. Ao investigar, descobri que esse arquivo não roda em nenhum dos dois runners hoje (o glob do `test:unit` não lista `autoCombo` e o `include` do Vitest só pega `.tsx` nessa pasta); é um problema pré-existente do repositório, sem relação com esta leva, e vou registrá-lo separadamente. O #12636 conflitava apenas na lista de testes do `@omniroute/opencode-plugin/package.json`, de forma aditiva: o tip já tinha `models-fetcher.test.ts` (do #12607, irmão desta mesma leva) e o #12636 acrescenta `telemetry.test.ts`. Fiz a união dos dois lados (25 arquivos contra 24 de cada) em vez de escolher um, o que teria removido um arquivo da suíte do plugin em silêncio. Obrigado, @RaviTharuma. --- .../12473-github-live-catalog-filter.md | 1 + open-sse/services/githubCopilotModels.ts | 9 +- src/lib/db/models/activeSyncedCatalog.ts | 18 ++++ src/sse/handlers/chat.ts | 15 +--- .../handlers/chat/githubLiveCatalogFilter.ts | 71 +++++++++++++++ .../github-copilot-model-discovery.test.ts | 16 ++++ .../github-live-catalog-combo-12137.test.ts | 86 +++++++++++++++++++ 7 files changed, 204 insertions(+), 12 deletions(-) create mode 100644 changelog.d/features/12473-github-live-catalog-filter.md create mode 100644 src/sse/handlers/chat/githubLiveCatalogFilter.ts create mode 100644 tests/unit/github-live-catalog-combo-12137.test.ts diff --git a/changelog.d/features/12473-github-live-catalog-filter.md b/changelog.d/features/12473-github-live-catalog-filter.md new file mode 100644 index 0000000000..8eed69412e --- /dev/null +++ b/changelog.d/features/12473-github-live-catalog-filter.md @@ -0,0 +1 @@ +- **feat(providers):** skip GitHub combo members missing from the live synced catalog, and drop Copilot models that are policy-disabled or hidden from the model picker ([#12473](https://github.com/diegosouzapw/OmniRoute/pull/12473)) — thanks @RaviTharuma diff --git a/open-sse/services/githubCopilotModels.ts b/open-sse/services/githubCopilotModels.ts index b7a87ffbf2..39d112f014 100644 --- a/open-sse/services/githubCopilotModels.ts +++ b/open-sse/services/githubCopilotModels.ts @@ -92,8 +92,15 @@ function toNonEmptyString(value: unknown): string | null { // (rename-robust) rather than an id allowlist: any model the account is entitled // to whose capabilities.type is "chat" (or that carries a chat-shaped // supported_endpoints) is kept, so a newly-entitled model shows up with no code -// change. Only explicitly non-chat rows (embeddings / completion) are dropped. +// change. Also filters out rows when policy.state is set and != "enabled", or +// when model_picker_enabled=false. Explicitly non-chat rows (embeddings / +// completion) are dropped as well. function isRoutableChatModel(item: RawRecord): boolean { + const policy = asRecord(item.policy); + const policyState = toNonEmptyString(policy.state); + if (policyState && policyState !== "enabled") return false; + if (item.model_picker_enabled === false) return false; + const capabilities = asRecord(item.capabilities); const capType = toNonEmptyString(capabilities.type); if (capType) return capType === "chat"; diff --git a/src/lib/db/models/activeSyncedCatalog.ts b/src/lib/db/models/activeSyncedCatalog.ts index 18ade8260b..d20e361695 100644 --- a/src/lib/db/models/activeSyncedCatalog.ts +++ b/src/lib/db/models/activeSyncedCatalog.ts @@ -13,6 +13,24 @@ export type ActiveSyncedCatalog = { models: SyncedAvailableModel[]; }; +/** + * Fail-open membership check for explicit combo members against a live catalog. + * `null` means no authoritative catalog is synced yet (unchanged behavior). + */ +export function catalogContainsModel( + catalog: ActiveSyncedCatalog, + modelId: string +): boolean | null { + if (!catalog.authoritative) return null; + const trimmed = modelId.trim(); + if (!trimmed) return false; + const ids = new Set(catalog.models.map((model) => model.id)); + if (ids.has(trimmed)) return true; + const slash = trimmed.indexOf("/"); + if (slash > 0 && ids.has(trimmed.slice(slash + 1))) return true; + return false; +} + export type ProviderCatalogReconciliation = { providers: string[]; excludedProviders: string[]; diff --git a/src/sse/handlers/chat.ts b/src/sse/handlers/chat.ts index 7ad2131076..f72b074b31 100644 --- a/src/sse/handlers/chat.ts +++ b/src/sse/handlers/chat.ts @@ -78,6 +78,7 @@ import { } from "@/lib/db/sessionAccountAffinity"; import { dispatchChatWithAffinityEviction } from "./chatDispatch"; import { getCachedSettings, getCombosCacheVersion } from "@/lib/db/readCache"; +import { comboCheckProvider, ghComboGate } from "./chat/githubLiveCatalogFilter.ts"; import { getCombos } from "@/lib/db/combos"; import { resolveModelLockoutSettings } from "@/lib/resilience/modelLockoutSettings"; import { @@ -1025,18 +1026,10 @@ async function handleChatImplementation( if (isCommonChatGptWebRetirementError(error)) return false; throw error; } - // Apply the same prefix-override guard as handleSingleModelChat: - // if providerId is just the prefix already in the model string, use - // the fully-resolved modelInfo.provider for a precise credential check. - const provider = (() => { - if (!target?.providerId) return modelInfo.provider; - if (target.providerId === modelInfo.provider) return modelInfo.provider; - if (modelString.startsWith(target.providerId + "/")) return modelInfo.provider; - return target.providerId; - })(); - if (!provider) return true; // can't determine provider, let it try - + const provider = comboCheckProvider(modelString, modelInfo, target?.providerId); const resolvedModel = modelInfo.model || modelString; + const githubGate = await ghComboGate(comboPreselectedCredentials, provider, resolvedModel); + if (githubGate !== null) return githubGate; const hasForcedConnection = typeof target?.connectionId === "string" && target.connectionId.trim().length > 0; let allowedConnections = intersectAllowedConnectionIds( diff --git a/src/sse/handlers/chat/githubLiveCatalogFilter.ts b/src/sse/handlers/chat/githubLiveCatalogFilter.ts new file mode 100644 index 0000000000..4112e815fc --- /dev/null +++ b/src/sse/handlers/chat/githubLiveCatalogFilter.ts @@ -0,0 +1,71 @@ +/** + * GitHub live-catalog combo gate (#12137). + * + * Extracted from chat.ts so the frozen handler does not grow. Explicit combo + * members missing from an authoritative GitHub catalog are skipped; a catalog + * that is not synced yet fails open (same pattern as providerWildcard). + * + * The catalog promise is memoized per request-scope object so combo candidates + * share one getActiveSyncedCatalog fetch instead of re-hitting the DB. + */ +import { + catalogContainsModel, + getActiveSyncedCatalog, + type ActiveSyncedCatalog, +} from "@/lib/db/models/activeSyncedCatalog"; + +const catalogByScope = new WeakMap>>(); + +/** Prefix-override guard used by combo pre-check (same as handleSingleModelChat). */ +export function comboCheckProvider( + modelString: string, + modelInfo: { provider?: string }, + providerId?: string | null +): string | undefined { + if (!providerId) return modelInfo.provider; + if (providerId === modelInfo.provider) return modelInfo.provider; + if (modelString.startsWith(providerId + "/")) return modelInfo.provider; + return providerId; +} + +function loadGithubLiveCatalog( + scope: object, + providerId: string, + loadCatalog: (id: string) => Promise = getActiveSyncedCatalog +): Promise { + let byProvider = catalogByScope.get(scope); + if (!byProvider) { + byProvider = new Map(); + catalogByScope.set(scope, byProvider); + } + let pending = byProvider.get(providerId); + if (!pending) { + pending = loadCatalog(providerId); + byProvider.set(providerId, pending); + } + return pending; +} + +/** + * Combo pre-check for GitHub live-catalog membership. + * + * Returns: + * - `true` — allow immediately (provider could not be determined) + * - `false` — skip this combo member + * - `null` — not a GitHub skip; continue the remaining credential checks + */ +export async function ghComboGate( + scope: object, + provider: string | null | undefined, + resolvedModel: string, + loadCatalog?: (id: string) => Promise +): Promise { + if (!provider) return true; + if (provider !== "github" && provider !== "gh") return null; + const inLiveCatalog = catalogContainsModel( + await loadGithubLiveCatalog(scope, provider, loadCatalog), + resolvedModel + ); + if (inLiveCatalog === false) return false; + return null; +} diff --git a/tests/unit/github-copilot-model-discovery.test.ts b/tests/unit/github-copilot-model-discovery.test.ts index 77c5b11a91..83ff7ee24e 100644 --- a/tests/unit/github-copilot-model-discovery.test.ts +++ b/tests/unit/github-copilot-model-discovery.test.ts @@ -38,6 +38,20 @@ const MOCK_COPILOT_MODELS_RESPONSE = { policy: { state: "enabled" }, capabilities: { type: "chat", limits: { max_context_window_tokens: 128000 } }, }, + { + id: "disabled-by-policy", + name: "Disabled by policy", + model_picker_enabled: true, + policy: { state: "disabled" }, + capabilities: { type: "chat" }, + }, + { + id: "hidden-from-picker", + name: "Hidden from picker", + model_picker_enabled: false, + policy: { state: "enabled" }, + capabilities: { type: "chat" }, + }, { id: "claude-sonnet-4.5", name: "Claude Sonnet 4.5", @@ -80,6 +94,8 @@ test("#3120 parseGitHubCopilotModels keeps every entitled CHAT model (capability assert.equal(gpt.owned_by, "github"); assert.ok(!ids.includes("text-embedding-3-small"), "embeddings models are skipped"); assert.ok(!ids.includes("gpt-41-copilot"), "completion utility models are skipped"); + assert.ok(!ids.includes("disabled-by-policy"), "policy.state=disabled is not routable"); + assert.ok(!ids.includes("hidden-from-picker"), "model_picker_enabled=false is not routable"); }); test("#3121 a model NOT in the live response is not advertised (entitlement filtering)", () => { diff --git a/tests/unit/github-live-catalog-combo-12137.test.ts b/tests/unit/github-live-catalog-combo-12137.test.ts new file mode 100644 index 0000000000..ccfd1054e4 --- /dev/null +++ b/tests/unit/github-live-catalog-combo-12137.test.ts @@ -0,0 +1,86 @@ +/** + * #12137 — explicit GitHub combo members vs live synced catalog. + */ +import test from "node:test"; +import assert from "node:assert/strict"; +import { catalogContainsModel } from "../../src/lib/db/models/activeSyncedCatalog.ts"; +import { + comboCheckProvider, + ghComboGate, +} from "../../src/sse/handlers/chat/githubLiveCatalogFilter.ts"; + +test("fail-open when the GitHub catalog is not authoritative yet", () => { + assert.equal( + catalogContainsModel( + { + authoritative: false, + models: [{ id: "claude-sonnet-5", name: "Claude Sonnet 5", source: "imported" }], + }, + "claude-sonnet-5" + ), + null + ); +}); + +test("rejects explicit members missing from an authoritative GitHub catalog", () => { + assert.equal( + catalogContainsModel( + { + authoritative: true, + models: [{ id: "claude-sonnet-5", name: "Claude Sonnet 5", source: "imported" }], + }, + "github/claude-fable-5" + ), + false + ); +}); + +test("accepts prefixed and bare ids that are in the live catalog", () => { + const catalog = { + authoritative: true, + models: [{ id: "claude-sonnet-5", name: "Claude Sonnet 5", source: "imported" }], + }; + assert.equal(catalogContainsModel(catalog, "claude-sonnet-5"), true); + assert.equal(catalogContainsModel(catalog, "github/claude-sonnet-5"), true); +}); + +test("comboCheckProvider applies the prefix-override guard", () => { + assert.equal(comboCheckProvider("github/claude-sonnet-5", { provider: "github" }), "github"); + assert.equal( + comboCheckProvider("github/claude-sonnet-5", { provider: "github" }, "github"), + "github" + ); + assert.equal( + comboCheckProvider("xiaomi/mimo-v2-flash", { provider: "xiaomi" }, "opengate"), + "opengate" + ); + assert.equal(comboCheckProvider("gh/claude-sonnet-5", { provider: "github" }, "gh"), "github"); +}); + +test("ghComboGate allows undetermined providers and fail-opens unsynced catalogs", async () => { + const scope = {}; + const unsynced = async () => ({ + authoritative: false, + models: [{ id: "claude-sonnet-5", name: "Claude Sonnet 5", source: "imported" as const }], + }); + assert.equal(await ghComboGate(scope, "", "claude-sonnet-5", unsynced), true); + assert.equal(await ghComboGate(scope, "openai", "gpt-4", unsynced), null); + assert.equal(await ghComboGate(scope, "github", "claude-sonnet-5", unsynced), null); +}); + +test("ghComboGate skips GitHub members missing from an authoritative catalog", async () => { + const scope = {}; + let loads = 0; + const load = async () => { + loads += 1; + return { + authoritative: true, + models: [{ id: "claude-sonnet-5", name: "Claude Sonnet 5", source: "imported" as const }], + }; + }; + assert.equal(await ghComboGate(scope, "github", "claude-fable-5", load), false); + assert.equal(await ghComboGate(scope, "github", "claude-sonnet-5", load), null); + assert.equal(await ghComboGate(scope, "github", "github/claude-sonnet-5", load), null); + assert.equal(loads, 1, "catalog fetch is memoized per request scope"); + assert.equal(await ghComboGate({}, "gh", "claude-fable-5", load), false); +}); From ac94dd9bcfd7e847a502507d17858a16b92f2378 Mon Sep 17 00:00:00 2001 From: Ravi Tharuma <25951435+RaviTharuma@users.noreply.github.com> Date: Fri, 4 Sep 2026 04:40:03 +0200 Subject: [PATCH 089/143] docs(arch): one-process recipe for tens of long /v1/responses (#12493) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 10 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check-file-size` OK e **241/242** nos 29 arquivos de teste que os PRs tocam. A única "falha" não é falha: `tests/unit/autoCombo/strict-zero-cost-filter.test.ts` é um teste em estilo Vitest que eu incluí por engano na invocação do runner nativo do Node — ele quebra no import (`@vitest/runner`), não numa asserção. Ao investigar, descobri que esse arquivo não roda em nenhum dos dois runners hoje (o glob do `test:unit` não lista `autoCombo` e o `include` do Vitest só pega `.tsx` nessa pasta); é um problema pré-existente do repositório, sem relação com esta leva, e vou registrá-lo separadamente. O #12636 conflitava apenas na lista de testes do `@omniroute/opencode-plugin/package.json`, de forma aditiva: o tip já tinha `models-fetcher.test.ts` (do #12607, irmão desta mesma leva) e o #12636 acrescenta `telemetry.test.ts`. Fiz a união dos dois lados (25 arquivos contra 24 de cada) em vez de escolher um, o que teria removido um arquivo da suíte do plugin em silêncio. Obrigado, @RaviTharuma. --- .env.example | 21 +++- docs/architecture/admission-lanes.md | 19 ++++ docs/guides/DOCKER_GUIDE.md | 24 ++-- docs/reference/ENVIRONMENT.md | 10 +- src/shared/middleware/chatBodyAdmission.ts | 23 ++-- .../11024-n-instance-scale-out-docs.test.ts | 43 ++++++- ...at-admission-byte-healthy-headroom.test.ts | 107 ++++++++++++++++++ tests/unit/chat-body-admission-queue.test.ts | 35 ++++-- tests/unit/chat-body-admission.test.ts | 11 +- tests/unit/with-chat-admission-10786.test.ts | 5 + 10 files changed, 250 insertions(+), 48 deletions(-) create mode 100644 tests/unit/chat-admission-byte-healthy-headroom.test.ts diff --git a/.env.example b/.env.example index aa1c588b8f..e0fa9d6bed 100644 --- a/.env.example +++ b/.env.example @@ -418,7 +418,9 @@ ALLOW_API_KEY_REVEAL=false # provider dispatch. Heavyweight capacity is reserved before parsing; excess work # receives 503 + Retry-After instead of overlapping until the process OOMs. # Used by: src/shared/middleware/chatBodyAdmission.ts -# Actual bodies at or above this size require a heavyweight lease. Default 262144 (256 KB). +# Actual bodies at or above this size take the heavyweight lease (BYTE path, +# including POST /v1/responses) and use the same #10437 healthy-headroom escape +# as structure-heavy. Default 262144 (256 KB). # OMNIROUTE_CHAT_LARGE_BODY_BYTES=262144 # Actual-byte hard cap enforced during bounded ingestion. Default 52428800 (50 MB). # OMNIROUTE_CHAT_HARD_MAX_BODY_BYTES=52428800 @@ -426,6 +428,11 @@ ALLOW_API_KEY_REVEAL=false # left unset, heavyweight admission is gated by OMNIROUTE_CHAT_MAX_INFLIGHT_BYTES below # instead (an auto-derived byte budget), fixing coding-agent fan-out (multiple # subagents/CLIs) collapsing to an effective concurrency of ~1 and 503ing. +# Two overlapping ~750k-token /v1/responses abort ~12 Gi heaps (#7849) — a +# memory-budget warning, not a hard product max of 2. A healthy heap may admit +# more via HEALTHY_HEADROOM. Tens of long SSE clients (40-50) is heap + +# OMNIROUTE_CHAT_MAX_INFLIGHT_BYTES / #10110. Multiply heaps with N independent +# DATA_DIRs (#11024); never replicas>1 on one SQLite. # OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT=1 # Override for the auto-derived ingest byte budget (#503-fanout). Default: 25% of the # process's effective memory ceiling (V8 heap limit, or the tighter cgroup/container @@ -433,13 +440,15 @@ ALLOW_API_KEY_REVEAL=false # 2 GiB; explicit overrides are clamped to the same safe range. Read # chatAdmission.maxInflightBytes/budgetSource at /api/monitoring/health before overriding. # OMNIROUTE_CHAT_MAX_INFLIGHT_BYTES=134217728 -# Heap-pressure shed ratio (heapUsed/heap_size_limit) for the structural admission gate -# (#10183, #10268): a second concurrent heavyweight request past OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT -# is only shed with a retryable 503 when the heap is ALSO under this much pressure — on a -# healthy heap it is admitted instead. Range (0, 1]. Default 0.75. +# Heap-pressure shed ratio (heapUsed/heap_size_limit) for BYTE and STRUCTURE +# heavyweight admission (#10183, #10268, #10437): a concurrent heavyweight request +# past OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT is only shed with a retryable 503 when the +# heap is ALSO under this much pressure — on a healthy heap it is admitted via +# healthy-headroom instead. Range (0, 1]. Default 0.75. # OMNIROUTE_CHAT_ADMISSION_HEAP_SHED_RATIO=0.75 # Bounded extra capacity for the healthy-heap fast path above OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT -# (#10437): once this many concurrent leases are active through the healthy-heap bypass, +# (#10437, BYTE + STRUCTURE, including bodies >= OMNIROUTE_CHAT_LARGE_BODY_BYTES): +# once this many concurrent leases are active through the healthy-heap bypass, # further busy requests fall through to the same bounded-wait/shed path used under real heap # pressure. 0 disables the bypass entirely. Default 1. # OMNIROUTE_CHAT_ADMISSION_HEALTHY_HEADROOM=1 diff --git a/docs/architecture/admission-lanes.md b/docs/architecture/admission-lanes.md index 2b2a4cf96b..ddb8ea5fac 100644 --- a/docs/architecture/admission-lanes.md +++ b/docs/architecture/admission-lanes.md @@ -120,3 +120,22 @@ against the **parent's** tenant lane. The byte-level lanes bound the memory-heavy parse/compress path; the adaptive lanes bound dispatch cost per tenant. #9654's criterion 1 ("one session's burst does not 503 another") is enforced by system 1 unconditionally and by system 2 once opt-in is enabled. + +## 4. One-process long `/v1/responses` (healthy-headroom) + +[#10437](https://github.com/diegosouzapw/OmniRoute/pull/10437) added +`tryAcquireHealthyHeadroom` so a second structurally-heavy request is admitted +when the heap is below `OMNIROUTE_CHAT_ADMISSION_HEAP_SHED_RATIO`. The BYTE +path used by `admitChatRequest` (bodies ≥ `OMNIROUTE_CHAT_LARGE_BODY_BYTES`, +default 256 KiB, including `POST /v1/responses`) uses the **same** escape. + +This is the supported **one-process** recipe for more than two concurrent long +SSE `/v1/responses`: raise primary + healthy-headroom only as far as the heap +and the process-wide inflight-byte budget (`OMNIROUTE_CHAT_MAX_INFLIGHT_BYTES` +/ #10110) allow. Tens of long SSE clients (40–50) is that memory-budget +question, not a hard “max 2” product limit. A pressured heap still sheds with +retryable `503` so #7849 does not return. + +To **multiply heaps**, run N independent `DATA_DIR`s (#11024). Never +`replicas > 1` on one SQLite file (#10350). This section is not a reopen of +the DATA_DIR scale-out recipe. diff --git a/docs/guides/DOCKER_GUIDE.md b/docs/guides/DOCKER_GUIDE.md index 7d5137f6f9..1bf026c0b8 100644 --- a/docs/guides/DOCKER_GUIDE.md +++ b/docs/guides/DOCKER_GUIDE.md @@ -567,19 +567,23 @@ External Postgres / multi-writer HA is **not** a documented stock path. If you n ## Scale-out: N independent processes -One Node process is **one V8 heap**. Two overlapping ~3 MiB / ~750k-token coding-agent `POST /v1/responses` (RTK + Caveman) abort that heap at ~12 Gi (`FATAL ERROR: Reached heap limit`) and can OOM a 16 Gi cgroup. See [#7849](https://github.com/diegosouzapw/OmniRoute/issues/7849). Heavyweight chat admission is gated by an auto-derived ingest byte budget (`OMNIROUTE_CHAT_MAX_INFLIGHT_BYTES`, `src/shared/middleware/admissionBudget.ts`) sized from that same V8/cgroup ceiling -- it already scales itself to the process's real memory, so overriding it upward (or setting the legacy `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` request-count cap) on an already-sized process reintroduces the abort. Small chats, `/healthz`, `/v1/models`, and MCP are **not** in that cap. +One Node process is **one V8 heap**. Two overlapping ~3 MiB / ~750k-token coding-agent `POST /v1/responses` (RTK + Caveman) abort that heap at ~12 Gi (`FATAL ERROR: Reached heap limit`) and can OOM a 16 Gi cgroup. See [#7849](https://github.com/diegosouzapw/OmniRoute/issues/7849). That measurement is a **memory-budget** warning, not a product hard-max of two concurrent long `/v1/responses`. Heavyweight chat admission is gated by an auto-derived ingest byte budget (`OMNIROUTE_CHAT_MAX_INFLIGHT_BYTES`, `src/shared/middleware/admissionBudget.ts`) sized from that same V8/cgroup ceiling — overriding it upward (or setting the legacy `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` request-count cap) on an already-sized process reintroduces the abort. Small chats, `/healthz`, `/v1/models`, and MCP are **not** in that cap. -To go beyond two concurrent **large** jobs **today**: +### One-process: more than two long `/v1/responses` -| Do | Do not | -| -------------------------------------------------------------------------------------------- | ---------------------------------------------------- | -| Run **N containers/pods**, each with its **own** `DATA_DIR` / volume | Set `replicas > 1` against one SQLite file | -| Keep each instance at 1–2 heavy in-flight and 12–16 Gi cgroup | Give one process 8× RAM and `max=8` | -| Optional: `QUOTA_STORE_DRIVER=redis` + `QUOTA_STORE_REDIS_URL` for **shared quota counters** | Treat Redis as shared SQLite — it is not | -| Duplicate provider secrets into each instance (or accept partitioned dashboards) | Expect one dashboard / one call-log across instances | -| Front with any load balancer; sticky by API key or session is enough | Require a vendor-specific size-aware middleware | +A **healthy** process (heap below `OMNIROUTE_CHAT_ADMISSION_HEAP_SHED_RATIO`, default `0.75`) **may** run more than two concurrent long `POST /v1/responses` when the process-wide inflight-byte budget (`OMNIROUTE_CHAT_MAX_INFLIGHT_BYTES` / #10110) still has room. Bodies at or above `OMNIROUTE_CHAT_LARGE_BODY_BYTES` (default 256 KiB) take the same heavyweight lease as structure-heavy requests and use the same [#10437](https://github.com/diegosouzapw/OmniRoute/pull/10437) `tryAcquireHealthyHeadroom` escape (`OMNIROUTE_CHAT_ADMISSION_HEALTHY_HEADROOM`). Tens of concurrent long SSE clients (operators often need 40–50) is a **memory-budget** question — size heap + primary/headroom slots + `OMNIROUTE_CHAT_MAX_INFLIGHT_BYTES` — not a hard “max 2” product limit. A pressured heap still sheds with retryable `503` so #7849 does not return. -Hardware: `concurrent_large ≈ N × 2` at ~8–12 Gi heap / ~12–16 Gi cgroup **per instance**. Host RAM must cover `N × cgroup`, not “one 16 Gi pod with N=8.” +To **multiply heaps** (independent V8 old-spaces) **today**: + +| Do | Do not | +| --------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------- | +| Run **N containers/pods**, each with its **own** `DATA_DIR` / volume | Set `replicas > 1` against one SQLite file | +| Size heavy in-flight + healthy-headroom from heap / inflight-byte budget; 1–2 is the conservative #7849 default, not a hard product max | Give one process 8× RAM and an unbounded count cap | +| Optional: `QUOTA_STORE_DRIVER=redis` + `QUOTA_STORE_REDIS_URL` for **shared quota counters** | Treat Redis as shared SQLite — it is not | +| Duplicate provider secrets into each instance (or accept partitioned dashboards) | Expect one dashboard / one call-log across instances | +| Front with any load balancer; sticky by API key or session is enough | Require a vendor-specific size-aware middleware | + +Hardware: per-instance concurrent long `/v1/responses` is a **memory-budget** question (heap + inflight-byte / #10110). `N` independent `DATA_DIR`s still multiply heaps: host RAM must cover `N × cgroup`, not “one 16 Gi pod with N=8.” Never `replicas > 1` on one SQLite file. Compose sketch (two heaps, two volumes — not `deploy.replicas: 2`): diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index ae5abf8781..8b743b2ecc 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -217,12 +217,12 @@ OmniRoute uses **SQLite** (via `better-sqlite3`) for all persistence. These vari | `NO_LOG_API_KEY_IDS` | _(empty)_ | `src/lib/compliance/index.ts` | Comma-separated API key IDs that bypass request logging (GDPR compliance). | | `DEFAULT_RATE_LIMIT_PER_DAY` | _(unset = unlimited)_ | `src/shared/utils/apiKeyPolicy.ts` | Fallback per-day request budget applied to API keys whose `rate_limits` column is null. Unset or empty: no implicit cap (#2289, #11017). `0` is the same (unlimited). Positive integer N enables N/day, 5N/week, 20N/month. Malformed non-empty values fall back to the legacy 1000/day, 5000/week, 20000/month windows. | | `MAX_BODY_SIZE_BYTES` | `10485760` (10 MB) | `src/shared/middleware/bodySizeGuard.ts` | Maximum allowed request body size. Rejects payloads exceeding this limit. | -| `OMNIROUTE_CHAT_LARGE_BODY_BYTES` | `262144` (256 KB) | `src/shared/middleware/chatBodyAdmission.ts` | Actual request bodies at or above this threshold require an atomic process-local heavyweight admission lease before JSON parsing. | +| `OMNIROUTE_CHAT_LARGE_BODY_BYTES` | `262144` (256 KB) | `src/shared/middleware/chatBodyAdmission.ts` | Actual request bodies at or above this threshold take the atomic process-local heavyweight admission lease before JSON parsing (BYTE path, including `POST /v1/responses`). Same [#10437](https://github.com/diegosouzapw/OmniRoute/pull/10437) healthy-headroom escape as structure-heavy; still bounded by `OMNIROUTE_CHAT_MAX_INFLIGHT_BYTES` / [#10110](https://github.com/diegosouzapw/OmniRoute/issues/10110) so [#7849](https://github.com/diegosouzapw/OmniRoute/issues/7849) does not return. | | `OMNIROUTE_CHAT_HARD_MAX_BODY_BYTES` | `52428800` (50 MB) | `src/shared/middleware/chatBodyAdmission.ts` | Chat-route hard cap enforced against bytes read during bounded ingestion, including requests with missing, invalid, or dishonest `Content-Length`; excess receives `413`. | -| `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` | _(unset — no request-count cap)_ | `src/shared/middleware/chatBodyAdmission.ts` | **#503-fanout:** this legacy request-COUNT cap now binds only when explicitly set. Left unset (the default), heavyweight chat admission is instead gated by `OMNIROUTE_CHAT_MAX_INFLIGHT_BYTES` — an auto-derived BYTE budget sized from the process's real memory ceiling in **one process** (one V8 heap), fixing a bug where coding-agent fan-out (multiple subagents/CLIs, bodies routinely > 256 KB) collapsed to an effective concurrency of ~1 and 503'd under normal load. Setting this var restores the old fixed-count behavior on top of the byte budget for a deployment that already tuned it. Overload is retryable `503` with `Retry-After`. Two overlapping ~750k-token `/v1/responses` already abort ~12 Gi heaps (#7849) — the byte budget accounts for that ceiling automatically, so raising this manually is no longer the recommended lever. Multiply capacity with **N independent `DATA_DIR`s** (#11024), not `replicas>1` on one SQLite file. | -| `OMNIROUTE_CHAT_MAX_INFLIGHT_BYTES` | _(auto-derived)_ | `src/shared/middleware/admissionBudget.ts` | **#503-fanout:** override for the auto-derived ingest byte budget (25% of the tighter V8/cgroup memory ceiling divided by 8x transient amplification). Derived and explicit values clamp to 8 MiB–2 GiB. A body larger than the effective budget fails immediately with `413 body_exceeds_budget`; contention between individually serviceable bodies remains retryable `503`. Read `chatAdmission.maxInflightBytes` / `budgetSource` / `pressureSeverity` at `/api/monitoring/health` before tuning. | -| `OMNIROUTE_CHAT_ADMISSION_HEAP_SHED_RATIO` | `0.75` | `src/shared/middleware/chatBodyAdmission.ts` | Heap-pressure shed ratio (`heapUsed / heap_size_limit`) for the structural admission gate (#10183, #10268). A second concurrent heavyweight request past `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` is only shed with the retryable `503` when the heap is ALSO at or above this ratio; on a healthy heap it is admitted instead. | -| `OMNIROUTE_CHAT_ADMISSION_HEALTHY_HEADROOM` | `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` (default `1`) | `src/shared/middleware/chatBodyAdmission.ts` | Bounded extra capacity for the healthy-heap fast path above (#10437). Without this bound, every busy-but-healthy-heap request bypassed admission with no ceiling at all — a slow leak or a burst that never quite trips the heap-shed ratio could still pile up unlimited concurrent heavyweight work. Once this many concurrent leases are active through the healthy-heap path, further busy requests fall through to the SAME bounded-wait/shed path used under real heap pressure. `0` disables the bypass entirely. | +| `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` | _(unset — no request-count cap)_ | `src/shared/middleware/chatBodyAdmission.ts` | **#503-fanout:** this legacy request-COUNT cap now binds only when explicitly set. Left unset (the default), heavyweight chat admission is instead gated by `OMNIROUTE_CHAT_MAX_INFLIGHT_BYTES` — an auto-derived BYTE budget sized from the process's real memory ceiling in **one process** (one V8 heap). Two overlapping ~750k-token `/v1/responses` abort ~12 Gi heaps (#7849) — a **memory-budget** warning, not a hard product max of 2. A healthy process (heap below the shed ratio) MAY admit more concurrent long `/v1/responses` via `OMNIROUTE_CHAT_ADMISSION_HEALTHY_HEADROOM`. Tens of long SSE clients (40–50) is heap + `OMNIROUTE_CHAT_MAX_INFLIGHT_BYTES` / #10110, not “max 2”. Blindly raising this to “use the host” reintroduces #7849. Multiply **heaps** with **N independent `DATA_DIR`s** (#11024); never `replicas>1` on one SQLite file. | +| `OMNIROUTE_CHAT_MAX_INFLIGHT_BYTES` | _(auto-derived)_ | `src/shared/middleware/admissionBudget.ts` | **#503-fanout:** override for the auto-derived ingest byte budget (25% of the tighter V8/cgroup memory ceiling divided by 8x transient amplification). Derived and explicit values clamp to 8 MiB–2 GiB. A body larger than the effective budget fails immediately with `413 body_exceeds_budget`; contention between individually serviceable bodies remains retryable `503`. 40–50 concurrent long SSE clients is this budget + heap, not a hard “max 2”. Read `chatAdmission.maxInflightBytes` / `budgetSource` / `pressureSeverity` at `/api/monitoring/health` before tuning. | +| `OMNIROUTE_CHAT_ADMISSION_HEAP_SHED_RATIO` | `0.75` | `src/shared/middleware/chatBodyAdmission.ts` | Heap-pressure shed ratio (`heapUsed / heap_size_limit`) for BYTE and STRUCTURE heavyweight admission (#10183, #10268, #10437). A concurrent heavyweight request past `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` is only shed with the retryable `503` when the heap is ALSO at or above this ratio; on a healthy heap it is admitted via healthy-headroom instead. | +| `OMNIROUTE_CHAT_ADMISSION_HEALTHY_HEADROOM` | `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` (default `1`) | `src/shared/middleware/chatBodyAdmission.ts` | Bounded extra capacity for the healthy-heap fast path (#10437) on **both** STRUCTURE and BYTE (`admitChatRequest`, including bodies ≥ `OMNIROUTE_CHAT_LARGE_BODY_BYTES`). Without this bound, every busy-but-healthy-heap request bypassed admission with no ceiling. Once this many concurrent leases are active through the healthy-heap path, further busy requests fall through to the SAME bounded-wait/shed path used under real heap pressure. `0` disables the bypass entirely. | | `OMNIROUTE_CHAT_HEAVY_MESSAGE_COUNT` | `200` | `src/shared/middleware/chatBodyAdmission.ts` | Message count that classifies a chat request as heavyweight even when its body is below the byte threshold. | | `OMNIROUTE_CHAT_HEAVY_TOOL_COUNT` | `64` | `src/shared/middleware/chatBodyAdmission.ts` | Tool count that classifies a chat request as heavyweight even when its body is below the byte threshold. | | `OMNIROUTE_CHAT_HEAVY_ESTIMATED_TOKENS` | `32000` | `src/shared/middleware/chatBodyAdmission.ts` | Conservative string-size token estimate that classifies a request as heavyweight; this is an admission-cost proxy, not provider billing tokenization. | diff --git a/src/shared/middleware/chatBodyAdmission.ts b/src/shared/middleware/chatBodyAdmission.ts index b30ad4cad7..f838e92224 100644 --- a/src/shared/middleware/chatBodyAdmission.ts +++ b/src/shared/middleware/chatBodyAdmission.ts @@ -945,14 +945,7 @@ function rebuildRequest(request: Request, body: Uint8Array): Request { } as RequestInit & { duplex: "half" }); } -/** - * Reserve heavyweight capacity and ingest the body with a hard byte bound before JSON - * parsing. Missing/invalid Content-Length is sniffed only up to the heavyweight threshold; - * a lease is acquired atomically before retaining bytes at or beyond that threshold. - * - * Internal self-loop sub-requests (vision-bridge describe calls) bypass the lease - * reservation — they run inside a parent request that already holds the lease. - */ +/** Reserve heavyweight capacity and ingest the body with a hard byte bound. */ export async function admitChatRequest( request: Request, options: { @@ -961,6 +954,7 @@ export async function admitChatRequest( largeBodyBytes?: number; hardMaxBytes?: number; queueMs?: number; + heapPressureCheck?: () => boolean; } = {} ): Promise { const sessionId = options.sessionId ?? resolveSessionId(request); @@ -1028,15 +1022,16 @@ export async function admitChatRequest( return { admit: false, response: bodyExceedsBudgetResponse(controller.maxInflightBytes) }; } + const heapPressureCheck = options.heapPressureCheck ?? defaultHeapPressureCheck; let lease: ChatAdmissionLease | null = null; + // #10437: busy primary + healthy heap uses tryAcquireHealthyHeadroom; else queue/shed. + // Bodies at/above OMNIROUTE_CHAT_LARGE_BODY_BYTES take this same heavyweight lease. const reserve = async (bytes = 0): Promise => { if (lease) return true; - const countLease = await controller.acquireHeavyWithin( - queueMs, - request.signal, - bytes, - sessionId - ); + const countLease = + controller.tryAcquireHeavy() ?? + (!heapPressureCheck() ? controller.tryAcquireHealthyHeadroom() : null) ?? + (await controller.acquireHeavyWithin(queueMs, request.signal, bytes, sessionId)); if (!countLease) return false; // Additive ingest byte-budget gate (#503-fanout), layered on top of the diff --git a/tests/unit/11024-n-instance-scale-out-docs.test.ts b/tests/unit/11024-n-instance-scale-out-docs.test.ts index 0189324e94..df209680c3 100644 --- a/tests/unit/11024-n-instance-scale-out-docs.test.ts +++ b/tests/unit/11024-n-instance-scale-out-docs.test.ts @@ -2,8 +2,14 @@ import test from "node:test"; import assert from "node:assert/strict"; import { readFileSync } from "node:fs"; -const dockerGuide = readFileSync(new URL("../../docs/guides/DOCKER_GUIDE.md", import.meta.url), "utf8"); -const envDoc = readFileSync(new URL("../../docs/reference/ENVIRONMENT.md", import.meta.url), "utf8"); +const dockerGuide = readFileSync( + new URL("../../docs/guides/DOCKER_GUIDE.md", import.meta.url), + "utf8" +); +const envDoc = readFileSync( + new URL("../../docs/reference/ENVIRONMENT.md", import.meta.url), + "utf8" +); test("DOCKER_GUIDE documents N independent DATA_DIRs as the large-job scale-out (#11024)", () => { assert.match(dockerGuide, /## Scale-out: N independent processes/); @@ -19,8 +25,39 @@ test("DOCKER_GUIDE documents N independent DATA_DIRs as the large-job scale-out }); test("ENVIRONMENT.md points CHAT_MAX_HEAVY_IN_FLIGHT at per-process V8, not host RAM (#11024)", () => { - const row = envDoc.split("\n").find((line) => line.includes("`OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT`")); + const row = envDoc + .split("\n") + .find((line) => line.includes("`OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT`")); assert.ok(row); assert.match(row, /one process|per process|V8/i); assert.match(row, /DATA_DIR|#11024/); }); + +test("DOCKER_GUIDE documents the one-process long /v1/responses recipe (healthy-headroom, not max 2)", () => { + assert.match(dockerGuide, /One-process: more than two long/); + assert.match(dockerGuide, /tryAcquireHealthyHeadroom/); + assert.match(dockerGuide, /OMNIROUTE_CHAT_LARGE_BODY_BYTES/); + assert.match(dockerGuide, /40–50|40-50/); + assert.match(dockerGuide, /memory-budget/); + assert.match(dockerGuide, /#10110|#10437/); + assert.match(dockerGuide, /replicas > 1/); +}); + +test("ENVIRONMENT.md documents LARGE_BODY_BYTES healthy-headroom and no hard max-2", () => { + const large = envDoc + .split("\n") + .find((line) => line.startsWith("| `OMNIROUTE_CHAT_LARGE_BODY_BYTES`")); + const heavy = envDoc + .split("\n") + .find((line) => line.startsWith("| `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT`")); + const headroom = envDoc + .split("\n") + .find((line) => line.startsWith("| `OMNIROUTE_CHAT_ADMISSION_HEALTHY_HEADROOM`")); + assert.ok(large); + assert.ok(heavy); + assert.ok(headroom); + assert.match(large, /healthy-headroom|#10437/); + assert.match(large, /#10110|#7849/); + assert.match(heavy, /memory-budget|not a hard product max|not a hard “max 2”/i); + assert.match(headroom, /BYTE|admitChatRequest/); +}); diff --git a/tests/unit/chat-admission-byte-healthy-headroom.test.ts b/tests/unit/chat-admission-byte-healthy-headroom.test.ts new file mode 100644 index 0000000000..9a4cdbb840 --- /dev/null +++ b/tests/unit/chat-admission-byte-healthy-headroom.test.ts @@ -0,0 +1,107 @@ +// Byte-path call-site slice of #10437: admitChatRequest (POST /v1/responses large +// bodies) must use the existing tryAcquireHealthyHeadroom budget when the primary +// lease is busy and the heap is not pressured. The STRUCTURE path already does this; +// the BYTE path still called acquireHeavyWithin()/tryAcquireHeavy() only. +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + ChatAdmissionController, + CHAT_LARGE_BODY_BYTES, + admitChatRequest, +} from "../../src/shared/middleware/chatBodyAdmission.ts"; + +function byteHeavyBody(minBytes = 40): string { + return JSON.stringify({ input: [{ role: "user", content: "x".repeat(minBytes) }] }); +} + +function responsesRequest(body: string): Request { + return new Request("http://x/v1/responses", { + method: "POST", + headers: { "content-type": "application/json", "content-length": String(body.length) }, + body, + }); +} + +test("byte-heavy admitChatRequest: two concurrent bodies, 1+1 budget, second admits on a healthy heap", async () => { + const controller = new ChatAdmissionController(1, undefined, 1); + const body = byteHeavyBody(); + const options = { + controller, + largeBodyBytes: 32, + hardMaxBytes: 1024, + queueMs: 0, + heapPressureCheck: () => false, + }; + + const [first, second] = await Promise.all([ + admitChatRequest(responsesRequest(body), options), + admitChatRequest(responsesRequest(body), options), + ]); + + assert.equal(first.admit, true, "first byte-heavy request must take the primary lease"); + assert.equal(second.admit, true, "second byte-heavy request must use healthy-headroom"); + assert.equal(controller.activeHeavy, 1); + assert.equal(controller.activeHealthyHeadroom, 1); + if (first.admit) first.lease?.release(); + if (second.admit) second.lease?.release(); + assert.equal(controller.activeHeavy, 0); + assert.equal(controller.activeHealthyHeadroom, 0); +}); + +test("byte-heavy admitChatRequest: a pressured heap still 503s the second concurrent body", async () => { + const controller = new ChatAdmissionController(1, undefined, 1); + const body = byteHeavyBody(); + const options = { + controller, + largeBodyBytes: 32, + hardMaxBytes: 1024, + queueMs: 0, + heapPressureCheck: () => true, + }; + + const [first, second] = await Promise.all([ + admitChatRequest(responsesRequest(body), options), + admitChatRequest(responsesRequest(body), options), + ]); + + const results = [first, second]; + const admitted = results.filter((result) => result.admit); + const rejected = results.filter((result) => !result.admit); + assert.equal(admitted.length, 1, "primary budget still admits exactly one pressured request"); + assert.equal(rejected.length, 1, "pressured heap must not spend healthy-headroom"); + assert.equal(controller.activeHeavy, 1); + assert.equal(controller.activeHealthyHeadroom, 0); + const shed = rejected[0]; + if (!shed.admit) { + assert.equal(shed.response.status, 503); + assert.equal((await shed.response.json()).error.code, "chat_admission_busy"); + } + for (const result of admitted) if (result.admit) result.lease?.release(); +}); + +test("OMNIROUTE_CHAT_LARGE_BODY_BYTES default threshold takes the heavyweight lease and healthy-headroom", async () => { + const controller = new ChatAdmissionController(1, undefined, 1); + const body = byteHeavyBody(CHAT_LARGE_BODY_BYTES); + assert.ok( + body.length >= CHAT_LARGE_BODY_BYTES, + "fixture must sit at or above the default LARGE_BODY_BYTES threshold" + ); + const options = { + controller, + hardMaxBytes: CHAT_LARGE_BODY_BYTES * 2, + queueMs: 0, + heapPressureCheck: () => false, + }; + + const [first, second] = await Promise.all([ + admitChatRequest(responsesRequest(body), options), + admitChatRequest(responsesRequest(body), options), + ]); + + assert.equal(first.admit, true, "body at LARGE_BODY_BYTES must take the primary lease"); + assert.equal(second.admit, true, "second LARGE_BODY_BYTES body must use healthy-headroom"); + assert.equal(controller.activeHeavy, 1); + assert.equal(controller.activeHealthyHeadroom, 1); + if (first.admit) first.lease?.release(); + if (second.admit) second.lease?.release(); +}); diff --git a/tests/unit/chat-body-admission-queue.test.ts b/tests/unit/chat-body-admission-queue.test.ts index 345954c142..b80a11436d 100644 --- a/tests/unit/chat-body-admission-queue.test.ts +++ b/tests/unit/chat-body-admission-queue.test.ts @@ -109,7 +109,13 @@ test("waiting for admission times out into a retryable 503", async () => { test("byte-heavy admission waits for capacity when queueMs is set", async () => { const controller = new ChatAdmissionController(1); const body = JSON.stringify({ messages: [{ role: "user", content: "x".repeat(40) }] }); - const options = { controller, largeBodyBytes: 32, hardMaxBytes: 1024, queueMs: 500 }; + const options = { + controller, + largeBodyBytes: 32, + hardMaxBytes: 1024, + queueMs: 500, + heapPressureCheck: () => true, + }; const first = await admitChatRequest(chatRequest(body), options); assert.equal(first.admit, true); @@ -230,20 +236,24 @@ test("aborting the admission wait settles early, grants no lease, and removes th settledAfterAbort = true; }); await new Promise((resolve) => setTimeout(resolve, 50)); - assert.equal(settledAfterAbort, true, "abort must settle the wait promptly, not park for queueMs"); + assert.equal( + settledAfterAbort, + true, + "abort must settle the wait promptly, not park for queueMs" + ); const lease = await pending; assert.equal(lease, null, "abort must not grant a lease"); - assert.equal(controller.activeHeavy, 1, "the holder keeps its lease; the aborted wait consumed nothing"); + assert.equal( + controller.activeHeavy, + 1, + "the holder keeps its lease; the aborted wait consumed nothing" + ); // Releasing must NOT wake the removed waiter: capacity stays free. held.release(); await new Promise((resolve) => setTimeout(resolve, 0)); - assert.equal( - controller.activeHeavy, - 0, - "releasing after abort must not wake the removed waiter" - ); + assert.equal(controller.activeHeavy, 0, "releasing after abort must not wake the removed waiter"); }); test("aborting the head waiter preserves FIFO order for remaining waiters", async () => { @@ -341,7 +351,13 @@ test("byte-heavy admission enforces the queued-bytes cap end-to-end", async () = assert.ok(held); const body = JSON.stringify({ messages: [{ role: "user", content: "x".repeat(40) }] }); - const options = { controller, largeBodyBytes: 32, hardMaxBytes: 1024, queueMs: 2_000 }; + const options = { + controller, + largeBodyBytes: 32, + hardMaxBytes: 1024, + queueMs: 2_000, + heapPressureCheck: () => true, + }; // First request parks: declared length (~70B) fits the budget. const first = admitChatRequest(chatRequest(body), options); @@ -452,6 +468,7 @@ test("aborting the request signal cancels a queued byte-heavy wait", async () => largeBodyBytes: 32, hardMaxBytes: 1024, queueMs: 2_000, + heapPressureCheck: () => true, }); let settled = false; diff --git a/tests/unit/chat-body-admission.test.ts b/tests/unit/chat-body-admission.test.ts index 05afae869c..b7867bc069 100644 --- a/tests/unit/chat-body-admission.test.ts +++ b/tests/unit/chat-body-admission.test.ts @@ -344,7 +344,13 @@ test("an existing byte-heavy lease is reused for structure-heavy admission", asy test("heavyweight admission is atomic and returns retryable 503 at capacity", async () => { const controller = new ChatAdmissionController(1); const body = JSON.stringify({ messages: [{ role: "user", content: "x".repeat(40) }] }); - const options = { controller, largeBodyBytes: 32, hardMaxBytes: 1024 }; + const options = { + controller, + largeBodyBytes: 32, + hardMaxBytes: 1024, + // #10437 byte-path: shedding still requires real heap pressure. + heapPressureCheck: () => true, + }; const first = await admitChatRequest(chatRequest(body), options); assert.equal(first.admit, true); @@ -407,6 +413,7 @@ test("unknown or lying-small lengths cannot bypass occupied heavyweight capacity controller, largeBodyBytes: 32, hardMaxBytes: 1024, + heapPressureCheck: () => true, }); assert.equal(result.admit, false); if (!result.admit) assert.equal(result.response.status, 503); @@ -771,6 +778,7 @@ test("external clients cannot use the bypass header without a trusted self-loop controller, largeBodyBytes: 32, hardMaxBytes: 10 * 1024 * 1024, + heapPressureCheck: () => true, }); // Unknown key + bypass header must NOT bypass — capacity is exhausted → 503. @@ -877,6 +885,7 @@ test("sk_omniroute sentinel is rejected once an env key is configured (REQUIRE_A controller, largeBodyBytes: 32, hardMaxBytes: 10 * 1024 * 1024, + heapPressureCheck: () => true, }); assert.equal(result.admit, false, "sentinel must not bypass when an env key is configured"); diff --git a/tests/unit/with-chat-admission-10786.test.ts b/tests/unit/with-chat-admission-10786.test.ts index dc5f28c127..4c1c7c0a20 100644 --- a/tests/unit/with-chat-admission-10786.test.ts +++ b/tests/unit/with-chat-admission-10786.test.ts @@ -25,6 +25,10 @@ test("withChatAdmission does not invoke the handler when a second large body is const first = await admitChatRequest(chatRequest("http://x/v1/responses", body), options); assert.equal(first.admit, true); + // Occupy the #10437 healthy-headroom slot so the wrapper still hits today's 503 + // path (withChatAdmission does not inject heapPressureCheck). + const headroom = controller.tryAcquireHealthyHeadroom(); + assert.ok(headroom); let called = false; const wrapped = withChatAdmission(async () => { @@ -39,6 +43,7 @@ test("withChatAdmission does not invoke the handler when a second large body is const json = await res.json(); assert.equal(json.error.code, "chat_admission_busy"); first.lease?.release(); + headroom.release(); }); test("withChatAdmission invokes the handler and forwards the admitted request", async () => { From 5f9c358e9b81114abd7a9fe7a9ff7ac8ea1ad8de Mon Sep 17 00:00:00 2001 From: Ravi Tharuma <25951435+RaviTharuma@users.noreply.github.com> Date: Fri, 4 Sep 2026 04:40:22 +0200 Subject: [PATCH 090/143] fix(monitoring): serve cached credentialHealth off the request path (#12533) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 10 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check-file-size` OK e **241/242** nos 29 arquivos de teste que os PRs tocam. A única "falha" não é falha: `tests/unit/autoCombo/strict-zero-cost-filter.test.ts` é um teste em estilo Vitest que eu incluí por engano na invocação do runner nativo do Node — ele quebra no import (`@vitest/runner`), não numa asserção. Ao investigar, descobri que esse arquivo não roda em nenhum dos dois runners hoje (o glob do `test:unit` não lista `autoCombo` e o `include` do Vitest só pega `.tsx` nessa pasta); é um problema pré-existente do repositório, sem relação com esta leva, e vou registrá-lo separadamente. O #12636 conflitava apenas na lista de testes do `@omniroute/opencode-plugin/package.json`, de forma aditiva: o tip já tinha `models-fetcher.test.ts` (do #12607, irmão desta mesma leva) e o #12636 acrescenta `telemetry.test.ts`. Fiz a união dos dois lados (25 arquivos contra 24 de cada) em vez de escolher um, o que teria removido um arquivo da suíte do plugin em silêncio. Obrigado, @RaviTharuma. --- src/app/api/monitoring/health/route.ts | 389 ++++++++++-------- src/lib/credentialHealth/cache.ts | 79 +++- src/lib/credentialHealth/scheduler.ts | 10 +- .../monitoring-health-cache.test.ts | 22 +- ...onitoring-health-cached-credential.test.ts | 111 +++++ 5 files changed, 407 insertions(+), 204 deletions(-) create mode 100644 tests/unit/monitoring-health-cached-credential.test.ts diff --git a/src/app/api/monitoring/health/route.ts b/src/app/api/monitoring/health/route.ts index f913dbb2c5..c859ad7bdd 100644 --- a/src/app/api/monitoring/health/route.ts +++ b/src/app/api/monitoring/health/route.ts @@ -14,14 +14,23 @@ import { requireManagementAuth } from "@/lib/api/requireManagementAuth"; * Returns system info, provider health (circuit breakers), * rate limit status, and database stats. */ -// §8.2 optimization: short-TTL cache for the health payload. Health is a -// frequently-polled endpoint and rebuilding it every request (DB reads + -// status aggregation across 8 subsystems) is wasteful under rapid polling. 1s -// stays near-real-time for monitoring; the cache is invalidated on DELETE -// (circuit-breaker reset) so a manual reset is reflected immediately. +// §8.2 / #12532: short-TTL cache with stale-while-revalidate. Health is a +// frequently-polled endpoint; rebuilding it on the request path (DB reads + +// status aggregation) shares the event loop with GET /healthz. After the first +// fill, scrapes always receive the last payload immediately. An expired entry +// is refreshed in the background — never by awaiting live credential probes. let healthPayloadCache: { payload: unknown; expiresAt: number } | null = null; +let healthPayloadRefreshInFlight = false; +let healthPayloadCacheGeneration = 0; const HEALTH_PAYLOAD_TTL_MS = 1000; +/** Test-only: drop the in-process health payload cache. */ +export function __test_resetMonitoringHealthPayloadCache(): void { + healthPayloadCache = null; + healthPayloadRefreshInFlight = false; + healthPayloadCacheGeneration += 1; +} + // GHSA-mvf8-qc78-5mxm: the full health payload fingerprints the host (version, // node version, pid, memory, provider config). An anonymous caller — the common // case on a keyless install, and what a liveness/load-balancer probe needs — gets @@ -34,15 +43,70 @@ function publicHealthView(payload: unknown): Record { }; } +function serveHealthPayload(fullView: boolean, payload: unknown) { + return NextResponse.json(fullView ? payload : publicHealthView(payload)); +} + +function scheduleHealthPayloadRefresh(): void { + if (healthPayloadRefreshInFlight) return; + healthPayloadRefreshInFlight = true; + setImmediate(() => { + rebuildHealthPayload() + .catch((error) => { + console.warn( + "[API] GET /api/monitoring/health background refresh failed:", + error instanceof Error ? error.message : error + ); + }) + .finally(() => { + healthPayloadRefreshInFlight = false; + }); + }); +} + export async function GET(request: Request) { const fullView = (await requireManagementAuth(request, { alwaysRequireAuth: true })) === null; const cachedNow = Date.now(); - if (healthPayloadCache && cachedNow <= healthPayloadCache.expiresAt) { - return NextResponse.json( - fullView ? healthPayloadCache.payload : publicHealthView(healthPayloadCache.payload) - ); + if (healthPayloadCache) { + if (cachedNow > healthPayloadCache.expiresAt) { + scheduleHealthPayloadRefresh(); + } + return serveHealthPayload(fullView, healthPayloadCache.payload); } + try { + const payload = await rebuildHealthPayload(); + return serveHealthPayload(fullView, payload); + } catch (error) { + console.error("[API] GET /api/monitoring/health error:", error); + return NextResponse.json({ + status: "degraded", + error: "Health check partially unavailable", + timestamp: new Date().toISOString(), + providerBreakers: [], + providerHealth: {}, + rateLimitStatus: {}, + learnedLimits: {}, + lockouts: [], + quotaMonitor: { + active: 0, + alerting: 0, + exhausted: 0, + errors: 0, + statusCounts: { starting: 0, idle: 0, healthy: 0, warning: 0, exhausted: 0, error: 0 }, + byProvider: {}, + monitors: [], + }, + sessions: { activeCount: 0, stickyBoundCount: 0, byApiKey: {}, top: [] }, + adaptiveAdmission: null, + chatAdmission: null, + dedup: { inflightRequests: 0 }, + }); + } +} + +async function rebuildHealthPayload(): Promise { + const generation = healthPayloadCacheGeneration; const readHealthValue = (label: string, reader: () => T, fallback: T): T => { try { return reader(); @@ -64,179 +128,150 @@ export async function GET(request: Request) { byProvider: {}, }; - try { - const [ - circuitBreakerModule, - rateLimitModule, - accountFallbackModule, - requestDedupModule, - quotaMonitorModule, - sessionManagerModule, - credentialHealthModule, - localHealthModule, - adaptiveAdmissionModule, - chatAdmissionModule, - settingsResult, - connectionsResult, - ] = await Promise.allSettled([ - import("@/shared/utils/circuitBreaker"), - import("@omniroute/open-sse/services/rateLimitManager"), - import("@omniroute/open-sse/services/accountFallback"), - import("@omniroute/open-sse/services/requestDedup.ts"), - import("@omniroute/open-sse/services/quotaMonitor.ts"), - import("@omniroute/open-sse/services/sessionManager.ts"), - import("@/lib/credentialHealth/cache"), - import("@/lib/localHealthCheck"), - import("@omniroute/open-sse/services/admission/runtime.ts"), - import("@/shared/middleware/chatBodyAdmission"), - getCachedSettings(), - getProviderConnections(), - ]); + const [ + circuitBreakerModule, + rateLimitModule, + accountFallbackModule, + requestDedupModule, + quotaMonitorModule, + sessionManagerModule, + credentialHealthModule, + localHealthModule, + adaptiveAdmissionModule, + chatAdmissionModule, + settingsResult, + connectionsResult, + ] = await Promise.allSettled([ + import("@/shared/utils/circuitBreaker"), + import("@omniroute/open-sse/services/rateLimitManager"), + import("@omniroute/open-sse/services/accountFallback"), + import("@omniroute/open-sse/services/requestDedup.ts"), + import("@omniroute/open-sse/services/quotaMonitor.ts"), + import("@omniroute/open-sse/services/sessionManager.ts"), + import("@/lib/credentialHealth/cache"), + import("@/lib/localHealthCheck"), + import("@omniroute/open-sse/services/admission/runtime.ts"), + import("@/shared/middleware/chatBodyAdmission"), + getCachedSettings(), + getProviderConnections(), + ]); - const circuitBreakers = - circuitBreakerModule.status === "fulfilled" - ? readHealthValue( - "circuit breakers", - () => circuitBreakerModule.value.getAllCircuitBreakerStatuses(), - [] - ) - : []; - const rateLimitStatus = - rateLimitModule.status === "fulfilled" - ? readHealthValue("rate limits", () => rateLimitModule.value.getAllRateLimitStatus(), {}) - : {}; - const learnedLimits = - rateLimitModule.status === "fulfilled" - ? readHealthValue("learned limits", () => rateLimitModule.value.getLearnedLimits(), {}) - : {}; - const lockouts = - accountFallbackModule.status === "fulfilled" - ? readHealthValue( - "model lockouts", - () => accountFallbackModule.value.getAllModelLockouts(), - [] - ) - : []; - const quotaMonitorSummary = - quotaMonitorModule.status === "fulfilled" - ? readHealthValue( - "quota monitor summary", - () => quotaMonitorModule.value.getQuotaMonitorSummary(), - fallbackQuotaMonitorSummary - ) - : fallbackQuotaMonitorSummary; - const quotaMonitorMonitors = - quotaMonitorModule.status === "fulfilled" - ? readHealthValue( - "quota monitor snapshots", - () => quotaMonitorModule.value.getQuotaMonitorSnapshots(), - [] - ) - : []; - const activeSessions = - sessionManagerModule.status === "fulfilled" - ? readHealthValue( - "active sessions", - () => sessionManagerModule.value.getActiveSessions(), - [] - ) - : []; - const activeSessionsByKey = - sessionManagerModule.status === "fulfilled" - ? readHealthValue( - "active sessions by key", - () => sessionManagerModule.value.getAllActiveSessionCountsByKey(), - {} - ) - : {}; - const credentialHealth = - credentialHealthModule.status === "fulfilled" - ? readHealthValue( - "credential health", - () => credentialHealthModule.value.getCredentialHealthSummary(), - undefined - ) - : undefined; - const localProviders = - localHealthModule.status === "fulfilled" - ? readHealthValue( - "local providers", - () => localHealthModule.value.getAllHealthStatuses(), - {} - ) - : {}; - const settings = settingsResult.status === "fulfilled" ? settingsResult.value : {}; - const connections = connectionsResult.status === "fulfilled" ? connectionsResult.value : []; - const adaptiveAdmission = - adaptiveAdmissionModule.status === "fulfilled" - ? readHealthValue( - "adaptive admission", - () => adaptiveAdmissionModule.value.getAdaptiveAdmissionRuntime().snapshot(), - null - ) - : null; - // #11244: the STRUCTURAL admission gate (chatBodyAdmission.ts — bounded - // heavyweight lease + shed counters), exposed next to but distinct from the - // adaptive shadow-mode snapshot above. Additive key — nothing existing moves. - const chatAdmission = - chatAdmissionModule.status === "fulfilled" - ? readHealthValue( - "chat admission", - () => chatAdmissionModule.value.perConnectionAdmissionController.snapshot(), - null - ) - : null; + const circuitBreakers = + circuitBreakerModule.status === "fulfilled" + ? readHealthValue( + "circuit breakers", + () => circuitBreakerModule.value.getAllCircuitBreakerStatuses(), + [] + ) + : []; + const rateLimitStatus = + rateLimitModule.status === "fulfilled" + ? readHealthValue("rate limits", () => rateLimitModule.value.getAllRateLimitStatus(), {}) + : {}; + const learnedLimits = + rateLimitModule.status === "fulfilled" + ? readHealthValue("learned limits", () => rateLimitModule.value.getLearnedLimits(), {}) + : {}; + const lockouts = + accountFallbackModule.status === "fulfilled" + ? readHealthValue( + "model lockouts", + () => accountFallbackModule.value.getAllModelLockouts(), + [] + ) + : []; + const quotaMonitorSummary = + quotaMonitorModule.status === "fulfilled" + ? readHealthValue( + "quota monitor summary", + () => quotaMonitorModule.value.getQuotaMonitorSummary(), + fallbackQuotaMonitorSummary + ) + : fallbackQuotaMonitorSummary; + const quotaMonitorMonitors = + quotaMonitorModule.status === "fulfilled" + ? readHealthValue( + "quota monitor snapshots", + () => quotaMonitorModule.value.getQuotaMonitorSnapshots(), + [] + ) + : []; + const activeSessions = + sessionManagerModule.status === "fulfilled" + ? readHealthValue("active sessions", () => sessionManagerModule.value.getActiveSessions(), []) + : []; + const activeSessionsByKey = + sessionManagerModule.status === "fulfilled" + ? readHealthValue( + "active sessions by key", + () => sessionManagerModule.value.getAllActiveSessionCountsByKey(), + {} + ) + : {}; + const credentialHealth = + credentialHealthModule.status === "fulfilled" + ? readHealthValue( + "credential health", + () => credentialHealthModule.value.getCachedCredentialHealthSummary(), + undefined + ) + : undefined; + const localProviders = + localHealthModule.status === "fulfilled" + ? readHealthValue("local providers", () => localHealthModule.value.getAllHealthStatuses(), {}) + : {}; + const settings = settingsResult.status === "fulfilled" ? settingsResult.value : {}; + const connections = connectionsResult.status === "fulfilled" ? connectionsResult.value : []; + const adaptiveAdmission = + adaptiveAdmissionModule.status === "fulfilled" + ? readHealthValue( + "adaptive admission", + () => adaptiveAdmissionModule.value.getAdaptiveAdmissionRuntime().snapshot(), + null + ) + : null; + // #11244: the STRUCTURAL admission gate (chatBodyAdmission.ts — bounded + // heavyweight lease + shed counters), exposed next to but distinct from the + // adaptive shadow-mode snapshot above. Additive key — nothing existing moves. + const chatAdmission = + chatAdmissionModule.status === "fulfilled" + ? readHealthValue( + "chat admission", + () => chatAdmissionModule.value.perConnectionAdmissionController.snapshot(), + null + ) + : null; - const payload = buildHealthPayload({ - appVersion: APP_CONFIG.version, - // #10427: surface the artifact's git SHA so a deployment can be audited over HTTP - // instead of SSH + grepping compiled chunks (the 2026-08-14 gateway outage). - buildSha: readRunningBuildSha(), - catalogCount: Object.keys(AI_PROVIDERS).length, - settings, - connections, - circuitBreakers, - rateLimitStatus, - learnedLimits, - lockouts, - localProviders, - inflightRequests: - requestDedupModule.status === "fulfilled" - ? readHealthValue( - "inflight requests", - () => requestDedupModule.value.getInflightCount(), - 0 - ) - : 0, - quotaMonitorSummary, - quotaMonitorMonitors, - activeSessions, - activeSessionsByKey, - credentialHealth, - adaptiveAdmission, - chatAdmission, - }); + const payload = buildHealthPayload({ + appVersion: APP_CONFIG.version, + // #10427: surface the artifact's git SHA so a deployment can be audited over HTTP + // instead of SSH + grepping compiled chunks (the 2026-08-14 gateway outage). + buildSha: readRunningBuildSha(), + catalogCount: Object.keys(AI_PROVIDERS).length, + settings, + connections, + circuitBreakers, + rateLimitStatus, + learnedLimits, + lockouts, + localProviders, + inflightRequests: + requestDedupModule.status === "fulfilled" + ? readHealthValue("inflight requests", () => requestDedupModule.value.getInflightCount(), 0) + : 0, + quotaMonitorSummary, + quotaMonitorMonitors, + activeSessions, + activeSessionsByKey, + credentialHealth, + adaptiveAdmission, + chatAdmission, + }); + if (generation === healthPayloadCacheGeneration) { healthPayloadCache = { payload, expiresAt: Date.now() + HEALTH_PAYLOAD_TTL_MS }; - return NextResponse.json(fullView ? payload : publicHealthView(payload)); - } catch (error) { - console.error("[API] GET /api/monitoring/health error:", error); - return NextResponse.json({ - status: "degraded", - error: "Health check partially unavailable", - timestamp: new Date().toISOString(), - providerBreakers: [], - providerHealth: {}, - rateLimitStatus: {}, - learnedLimits: {}, - lockouts: [], - quotaMonitor: { ...fallbackQuotaMonitorSummary, monitors: [] }, - sessions: { activeCount: 0, stickyBoundCount: 0, byApiKey: {}, top: [] }, - adaptiveAdmission: null, - chatAdmission: null, - dedup: { inflightRequests: 0 }, - }); } + return payload; } /** diff --git a/src/lib/credentialHealth/cache.ts b/src/lib/credentialHealth/cache.ts index e3283cb616..85bd9431c9 100644 --- a/src/lib/credentialHealth/cache.ts +++ b/src/lib/credentialHealth/cache.ts @@ -175,29 +175,80 @@ export function getAllCredentialHealth(): Record return result; } -/** - * Get cache summary stats for health API. - */ -export function getCredentialHealthSummary(): { +export interface CredentialHealthSummary { total: number; healthy: number; failed: number; unknown: number; stale: number; -} { - const all = getAllCredentialHealth(); - const entries = Object.values(all); - const now = Date.now(); +} - return { - total: entries.length, - healthy: entries.filter((e) => e.status === "active").length, - failed: entries.filter((e) => e.status === "error").length, - unknown: entries.filter((e) => e.status === "unknown").length, - stale: entries.filter((e) => now - e.lastTested.getTime() > STALE_THRESHOLD_MS).length, +/** + * Snapshot credential health for GET /api/monitoring/health. + * + * Never probes upstream and never expires entries on read. Expired / old + * rows stay in the counts so a scrape can return immediately while the + * background scheduler refreshes them (#12532). + */ +export function getCachedCredentialHealthSummary(): CredentialHealthSummary { + const state = getCacheState(); + const now = Date.now(); + let total = 0; + let healthy = 0; + let failed = 0; + let unknown = 0; + let stale = 0; + + for (const entry of state.cache.values()) { + total += 1; + if (entry.status.status === "active") healthy += 1; + else if (entry.status.status === "error") failed += 1; + else unknown += 1; + if (now - entry.status.lastTested.getTime() > STALE_THRESHOLD_MS || now > entry.expiresAt) { + stale += 1; + } + } + + return { total, healthy, failed, unknown, stale }; +} + +/** + * Get cache summary stats for health API. + * Monitoring scrapes must use the stale-safe snapshot (no live probes). + */ +export function getCredentialHealthSummary(): CredentialHealthSummary { + return getCachedCredentialHealthSummary(); +} + +/** Test-only: drop every cached credential-health row. */ +export function __test_resetCredentialHealthCache(): void { + globalThis.__omnirouteCredentialCache = { + initialized: false, + cache: new Map(), }; } +/** Test-only: insert a cache row, including expired / stale timestamps. */ +export function __test_putCredentialHealth(entry: { + connectionId: string; + provider: string; + status: "active" | "error" | "unknown"; + lastTested: Date; + expiresAt?: number; +}): void { + const state = getCacheState(); + state.cache.set(entry.connectionId, { + status: { + connectionId: entry.connectionId, + provider: entry.provider, + status: entry.status, + lastTested: entry.lastTested, + consecutiveFailures: 0, + }, + expiresAt: entry.expiresAt ?? Date.now() + DEFAULT_TTL_MS, + }); +} + /** * Mark cache as initialized (called by scheduler on startup). */ diff --git a/src/lib/credentialHealth/scheduler.ts b/src/lib/credentialHealth/scheduler.ts index d5922e87d2..dbbeb1d842 100644 --- a/src/lib/credentialHealth/scheduler.ts +++ b/src/lib/credentialHealth/scheduler.ts @@ -17,13 +17,12 @@ * - Resets to default on success */ +import { setImmediate as yieldToEventLoop } from "node:timers/promises"; + import { testSingleConnection } from "@/app/api/providers/[id]/test/route"; import { getProviderConnections } from "@/lib/db/providers"; import { getCachedSettings } from "@/lib/db/readCache"; -import { - setCredentialHealth, - initCredentialCache, -} from "@/lib/credentialHealth/cache"; +import { setCredentialHealth, initCredentialCache } from "@/lib/credentialHealth/cache"; import { isCredentialProbeInconclusive, resolveInconclusiveProbeRecheckDelayMs, @@ -386,6 +385,9 @@ export async function sweep(): Promise { } for (const batch of batches) { + // Yield so GET /healthz and cached /api/monitoring/health can drain + // while this background sweep talks to providers (#12532). + await yieldToEventLoop(); await Promise.allSettled( batch.map((conn) => testConnection(conn.id, conn.provider, getConnIntervalMs(conn, globalIntervalMs)) diff --git a/tests/integration/monitoring-health-cache.test.ts b/tests/integration/monitoring-health-cache.test.ts index 3df1d63b10..2b38924ff1 100644 --- a/tests/integration/monitoring-health-cache.test.ts +++ b/tests/integration/monitoring-health-cache.test.ts @@ -1,12 +1,12 @@ /** * Integration test for the short-TTL cache on GET /api/monitoring/health. * - * Health is a frequently-polled endpoint; rebuilding it every request (DB reads - * + status aggregation across subsystems) is wasteful under rapid polling. The - * route caches the payload for HEALTH_PAYLOAD_TTL_MS (1s) and invalidates it on - * DELETE (circuit-breaker reset). We assert the behavior via the payload's - * `timestamp` field, which is stamped at build time: identical timestamp ⇒ the - * cached payload was served; a fresh timestamp ⇒ it was rebuilt. + * Health is a frequently-polled endpoint; rebuilding it on the request path + * (DB reads + status aggregation) starves GET /healthz (#12532). The route + * caches the payload for HEALTH_PAYLOAD_TTL_MS (1s). After the first fill, + * expired entries are served immediately (stale-while-revalidate) and + * refreshed off the request path. DELETE (circuit-breaker reset) invalidates + * the cache so the next GET rebuilds. We assert via `timestamp`. */ import test from "node:test"; import assert from "node:assert/strict"; @@ -20,7 +20,8 @@ process.env.REQUIRE_API_KEY = "false"; process.env.JWT_SECRET = "test-health-cache-secret"; await import("../../src/lib/db/core.ts"); -const { GET, DELETE } = await import("../../src/app/api/monitoring/health/route.ts"); +const { GET, DELETE, __test_resetMonitoringHealthPayloadCache } = + await import("../../src/app/api/monitoring/health/route.ts"); // GHSA-mvf8-qc78-5mxm: the detailed health payload (the one carrying `timestamp`) // is reserved for a management principal — GET now takes the Request and an @@ -52,19 +53,22 @@ async function healthTimestamp(): Promise { } test("GET within the TTL serves the cached payload (identical timestamp)", async () => { + __test_resetMonitoringHealthPayloadCache(); const t1 = await healthTimestamp(); const t2 = await healthTimestamp(); assert.equal(t2, t1, "a second GET within the TTL must return the cached payload"); }); -test("cache expires after the TTL — a fresh payload is built", async () => { +test("expired cache is served immediately (stale-while-revalidate)", async () => { + __test_resetMonitoringHealthPayloadCache(); const t1 = await healthTimestamp(); await new Promise((r) => setTimeout(r, 1100)); // TTL is 1000ms const t2 = await healthTimestamp(); - assert.notEqual(t2, t1, "after the 1s TTL the payload must be rebuilt"); + assert.equal(t2, t1, "after the 1s TTL the stale cached payload must be returned immediately"); }); test("DELETE (circuit-breaker reset) invalidates the cache immediately", async () => { + __test_resetMonitoringHealthPayloadCache(); const t1 = await healthTimestamp(); // populate cache const delRes = await DELETE(authedRequest("DELETE")); assert.ok(delRes.status < 400, `DELETE should succeed, got ${delRes.status}`); diff --git a/tests/unit/monitoring-health-cached-credential.test.ts b/tests/unit/monitoring-health-cached-credential.test.ts new file mode 100644 index 0000000000..91a8b0e127 --- /dev/null +++ b/tests/unit/monitoring-health-cached-credential.test.ts @@ -0,0 +1,111 @@ +/** + * #12532 — GET /api/monitoring/health must serve cached credentialHealth + * immediately and must not run live credential probes on the request path. + */ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-health-cred-cache-")); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.REQUIRE_API_KEY = "false"; +process.env.JWT_SECRET = "test-health-cred-cache-secret"; + +await import("../../src/lib/db/core.ts"); + +const { + getCachedCredentialHealthSummary, + getCredentialHealthSummary, + __test_resetCredentialHealthCache, + __test_putCredentialHealth, +} = await import("../../src/lib/credentialHealth/cache.ts"); + +const { GET, __test_resetMonitoringHealthPayloadCache } = + await import("../../src/app/api/monitoring/health/route.ts"); + +const { SignJWT } = await import("jose"); +const AUTH_TOKEN = await new SignJWT({ authenticated: true }) + .setProtectedHeader({ alg: "HS256" }) + .setExpirationTime("30d") + .sign(new TextEncoder().encode(process.env.JWT_SECRET as string)); + +function authedRequest(): Request { + return new Request("http://localhost/api/monitoring/health", { + method: "GET", + headers: { cookie: `auth_token=${AUTH_TOKEN}` }, + }); +} + +const STALE_MS = 11 * 60 * 1000; + +test("getCachedCredentialHealthSummary includes expired and stale rows without deleting them", () => { + __test_resetCredentialHealthCache(); + const lastTested = new Date(Date.now() - STALE_MS); + __test_putCredentialHealth({ + connectionId: "conn-stale", + provider: "openai", + status: "active", + lastTested, + expiresAt: Date.now() - 1000, + }); + + const summary = getCachedCredentialHealthSummary(); + assert.deepEqual(summary, { + total: 1, + healthy: 1, + failed: 0, + unknown: 0, + stale: 1, + }); + assert.deepEqual(getCredentialHealthSummary(), summary); + assert.deepEqual(getCachedCredentialHealthSummary(), summary); +}); + +test("GET /api/monitoring/health returns the stale cached summary immediately", async () => { + __test_resetCredentialHealthCache(); + __test_resetMonitoringHealthPayloadCache(); + const lastTested = new Date(Date.now() - STALE_MS); + __test_putCredentialHealth({ + connectionId: "conn-stale-get", + provider: "anthropic", + status: "error", + lastTested, + expiresAt: Date.now() - 5000, + }); + + const started = Date.now(); + const res = await GET(authedRequest()); + const elapsedMs = Date.now() - started; + const body = (await res.json()) as { + credentialHealth?: { + total: number; + healthy: number; + failed: number; + unknown: number; + stale: number; + }; + }; + + assert.equal(res.status, 200); + assert.deepEqual(body.credentialHealth, { + total: 1, + healthy: 0, + failed: 1, + unknown: 0, + stale: 1, + }); + assert.ok(elapsedMs < 2000, `stale summary must return immediately, took ${elapsedMs}ms`); +}); + +test("monitoring health route never imports live credential probes", () => { + const source = fs.readFileSync( + path.join(process.cwd(), "src/app/api/monitoring/health/route.ts"), + "utf8" + ); + assert.doesNotMatch(source, /testSingleConnection/); + assert.doesNotMatch(source, /credentialHealth\/scheduler/); + assert.doesNotMatch(source, /forceSweep/); + assert.match(source, /getCachedCredentialHealthSummary/); +}); From 3b7c541f72fd28f3c86e2ac0559154bead3c55a2 Mon Sep 17 00:00:00 2001 From: Ravi Tharuma <25951435+RaviTharuma@users.noreply.github.com> Date: Fri, 4 Sep 2026 04:42:09 +0200 Subject: [PATCH 091/143] feat(opencode-plugin): map gateway cost/usage/tok/s onto OpenCode payloads (#12636) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 10 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo, `check-file-size` OK e **241/242** nos 29 arquivos de teste que os PRs tocam. A única "falha" não é falha: `tests/unit/autoCombo/strict-zero-cost-filter.test.ts` é um teste em estilo Vitest que eu incluí por engano na invocação do runner nativo do Node — ele quebra no import (`@vitest/runner`), não numa asserção. Ao investigar, descobri que esse arquivo não roda em nenhum dos dois runners hoje (o glob do `test:unit` não lista `autoCombo` e o `include` do Vitest só pega `.tsx` nessa pasta); é um problema pré-existente do repositório, sem relação com esta leva, e vou registrá-lo separadamente. O #12636 conflitava apenas na lista de testes do `@omniroute/opencode-plugin/package.json`, de forma aditiva: o tip já tinha `models-fetcher.test.ts` (do #12607, irmão desta mesma leva) e o #12636 acrescenta `telemetry.test.ts`. Fiz a união dos dois lados (25 arquivos contra 24 de cada) em vez de escolher um, o que teria removido um arquivo da suíte do plugin em silêncio. Obrigado, @RaviTharuma. --- @omniroute/opencode-plugin/package.json | 2 +- @omniroute/opencode-plugin/src/index.ts | 5 +- @omniroute/opencode-plugin/src/telemetry.ts | 249 ++++++++++++++++++ .../opencode-plugin/tests/telemetry.test.ts | 103 ++++++++ ...12636-opencode-plugin-gateway-telemetry.md | 1 + 5 files changed, 358 insertions(+), 2 deletions(-) create mode 100644 @omniroute/opencode-plugin/src/telemetry.ts create mode 100644 @omniroute/opencode-plugin/tests/telemetry.test.ts create mode 100644 changelog.d/features/12636-opencode-plugin-gateway-telemetry.md diff --git a/@omniroute/opencode-plugin/package.json b/@omniroute/opencode-plugin/package.json index 705429fc7b..96ae7b0729 100644 --- a/@omniroute/opencode-plugin/package.json +++ b/@omniroute/opencode-plugin/package.json @@ -23,7 +23,7 @@ "scripts": { "build": "tsup", "clean": "rm -rf dist", - "test": "node --import tsx/esm --test tests/scaffold.test.ts tests/auth.test.ts tests/options-schema.test.ts tests/multi-instance.test.ts tests/fetch-interceptor.test.ts tests/provider.test.ts tests/gemini-sanitize.test.ts tests/combos.test.ts tests/config-shim.test.ts tests/features.test.ts tests/feature-defaults.test.ts tests/usable-combo.test.ts tests/disk-snapshot-perms.test.ts tests/fork-features.test.ts tests/auto-combo-context.test.ts tests/provider-id-routing.test.ts tests/management-read-token.test.ts tests/auto-sync.test.ts tests/model-allowlist.test.ts tests/log-level.test.ts tests/effort-tier-variants.test.ts tests/naming.test.ts tests/models-fetcher.test.ts tests/free-budget-magnitude.test.ts", + "test": "node --import tsx/esm --test tests/scaffold.test.ts tests/auth.test.ts tests/options-schema.test.ts tests/multi-instance.test.ts tests/fetch-interceptor.test.ts tests/telemetry.test.ts tests/provider.test.ts tests/gemini-sanitize.test.ts tests/combos.test.ts tests/config-shim.test.ts tests/features.test.ts tests/feature-defaults.test.ts tests/usable-combo.test.ts tests/disk-snapshot-perms.test.ts tests/fork-features.test.ts tests/auto-combo-context.test.ts tests/provider-id-routing.test.ts tests/management-read-token.test.ts tests/auto-sync.test.ts tests/model-allowlist.test.ts tests/log-level.test.ts tests/effort-tier-variants.test.ts tests/naming.test.ts tests/free-budget-magnitude.test.ts tests/models-fetcher.test.ts", "prepublishOnly": "npm run clean && npm run build && npm test" }, "keywords": [ diff --git a/@omniroute/opencode-plugin/src/index.ts b/@omniroute/opencode-plugin/src/index.ts index 3ed215f341..4738c0e422 100644 --- a/@omniroute/opencode-plugin/src/index.ts +++ b/@omniroute/opencode-plugin/src/index.ts @@ -75,6 +75,7 @@ import { AUTO_VARIANT_DESCRIPTIONS, type FreeModelFreeType, } from "./naming.js"; +import { applyOmniRouteInferenceTelemetry } from "./telemetry.js"; /** * Minimal leveled logger sink accepted by the default fetchers and the static @@ -3769,6 +3770,8 @@ export function createOmniRouteFetchInterceptor(config: { baseOrigin = baseUrl.origin; const basePath = ensureV1Suffix(baseUrl.pathname); inferencePaths.add(`${basePath}/chat/completions`); + inferencePaths.add(`${basePath}/responses`); + inferencePaths.add(`${basePath}/messages`); inferencePaths.add(`${basePath}/models`); } catch { // Credential-attached base URLs are not schema-validated. A malformed @@ -3812,7 +3815,7 @@ export function createOmniRouteFetchInterceptor(config: { headers.set("Content-Type", "application/json"); } - return fetch(input, { ...init, headers }); + return applyOmniRouteInferenceTelemetry(await fetch(input, { ...init, headers })); }; } diff --git a/@omniroute/opencode-plugin/src/telemetry.ts b/@omniroute/opencode-plugin/src/telemetry.ts new file mode 100644 index 0000000000..36ba07ea25 --- /dev/null +++ b/@omniroute/opencode-plugin/src/telemetry.ts @@ -0,0 +1,249 @@ +/** + * Map gateway-reported OmniRoute inference telemetry onto the JSON/SSE + * payload OpenCode already consumes. Prefer headers / usage fields from the + * gateway. Never invent tok/s from tokens / latency (that includes TTFT). + */ +export type OmniRouteInferenceTelemetry = { + costUsd?: number; + tokensIn?: number; + tokensOut?: number; + tokensPerSecond?: number; + ttftMs?: number; + latencyMs?: number; + model?: string; + provider?: string; +}; + +const HEADER = { + cost: "x-omniroute-response-cost", + tokensIn: "x-omniroute-tokens-in", + tokensOut: "x-omniroute-tokens-out", + tokensPerSecond: "x-omniroute-tokens-per-second", + ttftMs: "x-omniroute-ttft-ms", + latencyMs: "x-omniroute-latency-ms", + model: "x-omniroute-model", + provider: "x-omniroute-provider", +} as const; + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +function readFiniteNumber(raw: string | null): number | undefined { + if (raw == null) return undefined; + const trimmed = raw.trim(); + if (trimmed === "") return undefined; + const parsed = Number(trimmed); + return Number.isFinite(parsed) ? parsed : undefined; +} + +function readPositiveNumber(raw: string | null): number | undefined { + const parsed = readFiniteNumber(raw); + if (parsed === undefined || parsed <= 0) return undefined; + return parsed; +} + +function readNonNegativeInt(raw: string | null): number | undefined { + const parsed = readFiniteNumber(raw); + if (parsed === undefined || parsed < 0) return undefined; + return Math.round(parsed); +} + +function readToken(raw: string | null): string | undefined { + if (raw == null) return undefined; + const trimmed = raw.trim(); + return trimmed === "" ? undefined : trimmed; +} + +export function parseOmniRouteInferenceTelemetry(headers: Headers): OmniRouteInferenceTelemetry { + const out: OmniRouteInferenceTelemetry = {}; + const cost = readFiniteNumber(headers.get(HEADER.cost)); + if (cost !== undefined && cost >= 0) out.costUsd = cost; + const tokensIn = readNonNegativeInt(headers.get(HEADER.tokensIn)); + if (tokensIn !== undefined) out.tokensIn = tokensIn; + const tokensOut = readNonNegativeInt(headers.get(HEADER.tokensOut)); + if (tokensOut !== undefined) out.tokensOut = tokensOut; + const tps = readPositiveNumber(headers.get(HEADER.tokensPerSecond)); + if (tps !== undefined) out.tokensPerSecond = tps; + const ttft = readPositiveNumber(headers.get(HEADER.ttftMs)); + if (ttft !== undefined) out.ttftMs = ttft; + const latency = readPositiveNumber(headers.get(HEADER.latencyMs)); + if (latency !== undefined) out.latencyMs = latency; + const model = readToken(headers.get(HEADER.model)); + if (model) out.model = model; + const provider = readToken(headers.get(HEADER.provider)); + if (provider) out.provider = provider; + return out; +} + +function telemetryFromUsage(usage: Record): OmniRouteInferenceTelemetry { + const out: OmniRouteInferenceTelemetry = {}; + const tps = usage.tokens_per_second; + if (typeof tps === "number" && Number.isFinite(tps) && tps > 0) { + out.tokensPerSecond = tps; + } + const ttft = usage.ttft_ms; + if (typeof ttft === "number" && Number.isFinite(ttft) && ttft > 0) { + out.ttftMs = ttft; + } + return out; +} + +function mergeTelemetry( + base: OmniRouteInferenceTelemetry, + extra: OmniRouteInferenceTelemetry, +): OmniRouteInferenceTelemetry { + return { + ...base, + ...Object.fromEntries(Object.entries(extra).filter(([, value]) => value !== undefined)), + }; +} + +function isInferencePayload(payload: Record): boolean { + return ( + isRecord(payload.usage) || + Array.isArray(payload.choices) || + payload.object === "chat.completion" || + payload.object === "response" || + payload.type === "message" || + Array.isArray(payload.output) + ); +} + +function attachToUsage( + usage: Record, + telemetry: OmniRouteInferenceTelemetry, +): Record { + const next = { ...usage }; + if ( + telemetry.tokensPerSecond !== undefined && + (typeof next.tokens_per_second !== "number" || next.tokens_per_second <= 0) + ) { + next.tokens_per_second = telemetry.tokensPerSecond; + } + if (telemetry.ttftMs !== undefined && (typeof next.ttft_ms !== "number" || next.ttft_ms <= 0)) { + next.ttft_ms = telemetry.ttftMs; + } + if (telemetry.costUsd !== undefined && typeof next.cost !== "number") { + next.cost = telemetry.costUsd; + } + return next; +} + +export function attachOmniRouteTelemetryToPayload( + payload: unknown, + telemetry: OmniRouteInferenceTelemetry, +): unknown { + if (!isRecord(payload) || !isInferencePayload(payload)) { + return payload; + } + const next: Record = { ...payload }; + if (telemetry.model) { + next.model = telemetry.model; + } + if (isRecord(next.usage)) { + next.usage = attachToUsage(next.usage, mergeTelemetry(telemetry, telemetryFromUsage(next.usage))); + } + if (isRecord(next.response) && isRecord(next.response.usage)) { + next.response = { + ...next.response, + usage: attachToUsage( + next.response.usage, + mergeTelemetry(telemetry, telemetryFromUsage(next.response.usage)), + ), + }; + } + return next; +} + +export function attachOmniRouteTelemetryToSseLine( + line: string, + telemetry: OmniRouteInferenceTelemetry, +): string { + const trimmed = line.trim(); + if (!trimmed.startsWith("data:")) { + return line; + } + const jsonText = trimmed.slice("data:".length).trim(); + if (!jsonText.startsWith("{")) { + return line; + } + try { + const parsed = JSON.parse(jsonText) as unknown; + const updated = attachOmniRouteTelemetryToPayload(parsed, telemetry); + if (updated === parsed) { + return line; + } + const prefix = line.slice(0, line.indexOf(jsonText)); + const suffix = line.endsWith("\r") ? "\r" : ""; + return `${prefix}${JSON.stringify(updated)}${suffix}`; + } catch { + return line; + } +} + +export async function applyOmniRouteInferenceTelemetry(response: Response): Promise { + const telemetry = parseOmniRouteInferenceTelemetry(response.headers); + const contentType = response.headers.get("content-type") ?? ""; + if (contentType.includes("text/event-stream") && response.body) { + return new Response(mapSseBody(response.body, telemetry), { + status: response.status, + statusText: response.statusText, + headers: response.headers, + }); + } + if (!contentType.includes("json")) { + return response; + } + const text = await response.text(); + let parsed: unknown; + try { + parsed = JSON.parse(text); + } catch { + return new Response(text, { + status: response.status, + statusText: response.statusText, + headers: response.headers, + }); + } + const next = attachOmniRouteTelemetryToPayload(parsed, telemetry); + if (next === parsed) { + return new Response(text, { + status: response.status, + statusText: response.statusText, + headers: response.headers, + }); + } + return new Response(JSON.stringify(next), { + status: response.status, + statusText: response.statusText, + headers: response.headers, + }); +} + +function mapSseBody( + body: ReadableStream, + telemetry: OmniRouteInferenceTelemetry, +): ReadableStream { + const decoder = new TextDecoder(); + const encoder = new TextEncoder(); + let pending = ""; + let live = { ...telemetry }; + return body.pipeThrough( + new TransformStream({ + transform(chunk, controller) { + pending += decoder.decode(chunk, { stream: true }); + const lines = pending.split("\n"); + pending = lines.pop() ?? ""; + for (const line of lines) { + controller.enqueue(encoder.encode(`${attachOmniRouteTelemetryToSseLine(line, live)}\n`)); + } + }, + flush(controller) { + if (pending.length > 0) { + controller.enqueue(encoder.encode(attachOmniRouteTelemetryToSseLine(pending, live))); + } + }, + }), + ); +} diff --git a/@omniroute/opencode-plugin/tests/telemetry.test.ts b/@omniroute/opencode-plugin/tests/telemetry.test.ts new file mode 100644 index 0000000000..fc36db266d --- /dev/null +++ b/@omniroute/opencode-plugin/tests/telemetry.test.ts @@ -0,0 +1,103 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { + applyOmniRouteInferenceTelemetry, + attachOmniRouteTelemetryToPayload, + attachOmniRouteTelemetryToSseLine, + parseOmniRouteInferenceTelemetry, +} from "../src/telemetry.js"; + +test("parseOmniRouteInferenceTelemetry: copies cost, tokens, tok/s, winning model", () => { + const headers = new Headers({ + "X-OmniRoute-Response-Cost": "0.0123", + "X-OmniRoute-Tokens-In": "10", + "X-OmniRoute-Tokens-Out": "200", + "X-OmniRoute-Tokens-Per-Second": "100.5", + "X-OmniRoute-Ttft-Ms": "300", + "X-OmniRoute-Latency-Ms": "2300", + "X-OmniRoute-Model": "winner-model", + "X-OmniRoute-Provider": "openai", + }); + const got = parseOmniRouteInferenceTelemetry(headers); + assert.equal(got.costUsd, 0.0123); + assert.equal(got.tokensIn, 10); + assert.equal(got.tokensOut, 200); + assert.equal(got.tokensPerSecond, 100.5); + assert.equal(got.ttftMs, 300); + assert.equal(got.model, "winner-model"); + assert.equal(got.provider, "openai"); +}); + +test("parseOmniRouteInferenceTelemetry: omits tok/s when header missing (do not invent from latency)", () => { + const headers = new Headers({ + "X-OmniRoute-Tokens-Out": "200", + "X-OmniRoute-Latency-Ms": "2000", + }); + const got = parseOmniRouteInferenceTelemetry(headers); + assert.equal(got.tokensPerSecond, undefined); + assert.equal(got.tokensOut, 200); + const payload = attachOmniRouteTelemetryToPayload( + { object: "chat.completion", usage: { prompt_tokens: 10, completion_tokens: 200 } }, + got, + ) as { usage: { tokens_per_second?: number } }; + assert.equal(payload.usage.tokens_per_second, undefined); +}); + +test("attachOmniRouteTelemetryToPayload: writes usage.tokens_per_second and winning model", () => { + const got = attachOmniRouteTelemetryToPayload( + { + object: "chat.completion", + model: "combo/auto", + usage: { prompt_tokens: 10, completion_tokens: 200 }, + }, + { tokensPerSecond: 80, ttftMs: 250, costUsd: 0, model: "gpt-winner" }, + ) as { + model: string; + usage: { tokens_per_second: number; ttft_ms: number; cost: number }; + }; + assert.equal(got.model, "gpt-winner"); + assert.equal(got.usage.tokens_per_second, 80); + assert.equal(got.usage.ttft_ms, 250); + assert.equal(got.usage.cost, 0); +}); + +test("attachOmniRouteTelemetryToPayload: does not mutate /v1/models catalog JSON", () => { + const catalog = { object: "list", data: [{ id: "m1" }] }; + const got = attachOmniRouteTelemetryToPayload(catalog, { + tokensPerSecond: 99, + model: "should-not-apply", + }); + assert.deepEqual(got, catalog); +}); + +test("attachOmniRouteTelemetryToSseLine: patches terminal usage data line", () => { + const line = + 'data: {"object":"chat.completion.chunk","usage":{"completion_tokens":200}}'; + const got = attachOmniRouteTelemetryToSseLine(line, { tokensPerSecond: 50 }); + assert.match(got, /"tokens_per_second":50/); + assert.match(got, /^data: /); +}); + +test("applyOmniRouteInferenceTelemetry: JSON response gets header tok/s", async () => { + const response = new Response( + JSON.stringify({ + object: "chat.completion", + model: "combo/auto", + usage: { prompt_tokens: 1, completion_tokens: 20 }, + }), + { + headers: { + "Content-Type": "application/json", + "X-OmniRoute-Tokens-Per-Second": "40", + "X-OmniRoute-Model": "winner", + }, + }, + ); + const next = await applyOmniRouteInferenceTelemetry(response); + const body = JSON.parse(await next.text()) as { + model: string; + usage: { tokens_per_second: number }; + }; + assert.equal(body.model, "winner"); + assert.equal(body.usage.tokens_per_second, 40); +}); diff --git a/changelog.d/features/12636-opencode-plugin-gateway-telemetry.md b/changelog.d/features/12636-opencode-plugin-gateway-telemetry.md new file mode 100644 index 0000000000..390f3df419 --- /dev/null +++ b/changelog.d/features/12636-opencode-plugin-gateway-telemetry.md @@ -0,0 +1 @@ +- **feat(opencode-plugin): map gateway cost/usage/tok/s onto OpenCode inference payloads** — the official plugin copies `X-OmniRoute-Response-Cost`, token counts, `X-OmniRoute-Tokens-Per-Second` / `usage.tokens_per_second`, TTFT, and the winning `X-OmniRoute-Model` onto the JSON/SSE body OpenCode already consumes. Missing tok/s is left unset (never `tokens / latency`). (#12636) From 57d7c8bc880976d8450c1ac2de7da7652bf0a04b Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Thu, 3 Sep 2026 22:42:40 -0400 Subject: [PATCH 092/143] fix(providers): sanitize boolean required and nested bare maps for Gemini (#12269) (#12624) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 4 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **119/119** nos 9 arquivos de teste que trazem. Três dos quatro conflitavam apenas no `config/quality/file-size-baseline.json`, todos de forma aditiva (chaves `_rebaseline_` distintas que devem coexistir); resolvidos com validação de JSON a cada passo. Registro que o **#12637 não é duplicata do #12566**, apesar do título quase idêntico: o autor documenta que aquele escopou o cooldown de preflight por família e este cobre o `genericQuotaFetcher`, que é o que o roteamento reset-aware efetivamente chama. Traz também validação ao vivo em VPS (imagem X500, `onmi-gemini3.6` → HTTP 200), satisfazendo a Hard Rule #18. Obrigado, @HouMinXi. --- open-sse/translator/helpers/geminiHelper.ts | 120 ++++++++ ...formed-required-and-bare-map-12269.test.ts | 267 ++++++++++++++++++ 2 files changed, 387 insertions(+) create mode 100644 tests/unit/gemini-malformed-required-and-bare-map-12269.test.ts diff --git a/open-sse/translator/helpers/geminiHelper.ts b/open-sse/translator/helpers/geminiHelper.ts index 8a67b07bc5..95fea6dcea 100644 --- a/open-sse/translator/helpers/geminiHelper.ts +++ b/open-sse/translator/helpers/geminiHelper.ts @@ -327,6 +327,123 @@ function toRecord(value: unknown): JsonRecord { return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : {}; } +// Maps of schemas — the container itself is not a bare property map (#12269). +const SCHEMA_MAP_KEYS = new Set([ + "properties", + "$defs", + "definitions", + "patternProperties", + "dependentSchemas", +]); + +const SCHEMA_NODE_KEYS = new Set([ + "additionalItems", + "additionalProperties", + "contentSchema", + "contains", + "default", + "dependencies", + "dependentRequired", + "dependentSchemas", + "discriminator", + "else", + "example", + "examples", + "externalDocs", + "if", + "patternProperties", + "propertyNames", + "then", + "unevaluatedItems", + "unevaluatedProperties", + "xml", +]); + +function isSchemaNode(record: JsonRecord): boolean { + if (Object.keys(record).some((key) => key.startsWith("x-") || SCHEMA_NODE_KEYS.has(key))) { + return true; + } + if (typeof record.type === "string" || Array.isArray(record.type)) return true; + if (record.properties !== undefined || Array.isArray(record.required)) return true; + if (record.items !== undefined || record.prefixItems !== undefined) return true; + if (record.anyOf !== undefined || record.oneOf !== undefined || record.allOf !== undefined) { + return true; + } + if (record.not !== undefined || record.$ref !== undefined || record.enum !== undefined) { + return true; + } + return record.const !== undefined; +} + +function isBarePropertyMap(record: JsonRecord): boolean { + const keys = Object.keys(record); + if (keys.length === 0 || isSchemaNode(record)) return false; + return keys.every((key) => { + const value = record[key]; + return Boolean(value) && typeof value === "object" && !Array.isArray(value); + }); +} + +function promoteBooleanRequired(record: JsonRecord): void { + const properties = toRecord(record.properties); + if (Object.keys(properties).length === 0) return; + + const required = Array.isArray(record.required) + ? record.required.filter((field): field is string => typeof field === "string") + : []; + + for (const [name, schema] of Object.entries(properties)) { + if (!schema || typeof schema !== "object" || Array.isArray(schema)) continue; + const child = schema as JsonRecord; + if (child.required === true) { + if (!required.includes(name)) required.push(name); + } + if ("required" in child && !Array.isArray(child.required)) { + delete child.required; + } + } + + if (required.length > 0) { + record.required = required; + } else if (!Array.isArray(record.required)) { + delete record.required; + } +} + +// Pre-pass for Cloud Code (#12269): boolean `required` on a property and nested +// bare property maps both survive the later phases and 400 Gemini's proto. +// Mirrors CLIProxyAPI normalizeMalformedSchemaObjects. +function normalizeMalformedSchemaObjects(obj: unknown, parentKey?: string): void { + if (!obj || typeof obj !== "object") return; + + if (Array.isArray(obj)) { + for (const item of obj) { + normalizeMalformedSchemaObjects(item, parentKey); + } + return; + } + + const record = obj as JsonRecord; + if (parentKey === undefined || !SCHEMA_MAP_KEYS.has(parentKey)) { + if (isBarePropertyMap(record)) { + const props = { ...record }; + for (const key of Object.keys(record)) { + delete record[key]; + } + record.type = "object"; + record.properties = props; + } + } + + promoteBooleanRequired(record); + + for (const [key, value] of Object.entries(record)) { + if (value && typeof value === "object") { + normalizeMalformedSchemaObjects(value, key); + } + } +} + function decodeJsonPointerSegment(segment: unknown): string { return String(segment).replace(/~1/g, "/").replace(/~0/g, "~"); } @@ -627,6 +744,9 @@ export function cleanJSONSchemaForAntigravity(schema: unknown): unknown { const root = cloneSchemaValue(schema); let cleaned = inlineLocalSchemaRefs(root, root); + // Phase 0: #12269 malformed skill/tool schemas (boolean required, bare maps). + normalizeMalformedSchemaObjects(cleaned); + // Phase 1: Convert and prepare convertConstToEnum(cleaned); convertEnumValuesToStrings(cleaned); diff --git a/tests/unit/gemini-malformed-required-and-bare-map-12269.test.ts b/tests/unit/gemini-malformed-required-and-bare-map-12269.test.ts new file mode 100644 index 0000000000..fd1c51a98c --- /dev/null +++ b/tests/unit/gemini-malformed-required-and-bare-map-12269.test.ts @@ -0,0 +1,267 @@ +/** + * Regression for #12269 — skills injection 400s Antigravity Gemini because two + * schema shapes survive `cleanJSONSchemaForAntigravity` and Gemini's proto + * rejects them: + * + * 1. A property carrying boolean `required: true`. `cleanupRequired()` only + * acts when `required` is an array, so the scalar passes through; Gemini + * declares `required` as `repeated string`. + * 2. A nested bare property map (`{ opts: { limit: { type: "number" } } }`). + * `normalizeInputSchema()` only expands string shorthands at the skill root. + * + * Diego named the missing pre-pass after CLIProxyAPI + * `normalizeMalformedSchemaObjects`: promote boolean `required` onto the parent + * array, lift a bare property map into `{ type: "object", properties }`. + */ +import test from "node:test"; +import assert from "node:assert/strict"; + +const { cleanJSONSchemaForAntigravity } = await import( + "../../open-sse/translator/helpers/geminiHelper.ts" +); +const { buildGeminiTools } = await import( + "../../open-sse/translator/helpers/geminiToolsSanitizer.ts" +); + +function paramsOf(tools: ReturnType): Record { + const first = tools[0] as { functionDeclarations?: Array<{ parameters?: unknown }> }; + return first.functionDeclarations?.[0]?.parameters as Record; +} + +function hasBooleanRequired(value: unknown): boolean { + if (!value || typeof value !== "object") return false; + if (Array.isArray(value)) return value.some(hasBooleanRequired); + const record = value as Record; + if (typeof record.required === "boolean") return true; + return Object.values(record).some(hasBooleanRequired); +} + +test("#12269 lifts boolean required:true off a string property onto the parent array", () => { + const cleaned = cleanJSONSchemaForAntigravity({ + type: "object", + properties: { + query: { type: "string", required: true }, + }, + }) as Record; + + const properties = cleaned.properties as Record>; + assert.equal(properties.query.type, "string"); + assert.equal("required" in properties.query, false, "scalar required must leave the property"); + assert.deepEqual(cleaned.required, ["query"]); + assert.equal(hasBooleanRequired(cleaned), false); +}); + +test("#12269 drops required:false instead of promoting it", () => { + const cleaned = cleanJSONSchemaForAntigravity({ + type: "object", + properties: { + query: { type: "string", required: false }, + hint: { type: "string" }, + }, + }) as Record; + + const properties = cleaned.properties as Record>; + assert.equal("required" in properties.query, false); + assert.equal(cleaned.required, undefined); +}); + +test("#12269 lifts a nested bare property map into type/object + properties", () => { + const cleaned = cleanJSONSchemaForAntigravity({ + type: "object", + properties: { + opts: { limit: { type: "number" } }, + }, + }) as Record; + + const properties = cleaned.properties as Record>; + const opts = properties.opts; + assert.equal(opts.type, "object"); + assert.equal("limit" in opts, false, "bare key must move under properties"); + const optsProps = opts.properties as Record>; + assert.equal(optsProps.limit.type, "number"); +}); + +test("#12269 boolean required and bare map survive buildGeminiTools together", () => { + const tools = buildGeminiTools([ + { + type: "function", + function: { + name: "skill_search", + parameters: { + type: "object", + properties: { + query: { type: "string", required: true }, + opts: { limit: { type: "number" } }, + }, + }, + }, + }, + ]); + + const params = paramsOf(tools); + const properties = params.properties as Record>; + assert.equal(properties.query.type, "string"); + assert.equal("required" in properties.query, false); + assert.deepEqual(params.required, ["query"]); + const opts = properties.opts; + assert.equal(opts.type, "object"); + assert.equal((opts.properties as Record>).limit.type, "number"); + assert.equal(hasBooleanRequired(params), false); +}); + +test("#12269 does not wrap schema-keyword objects as property maps", () => { + const cleaned = cleanJSONSchemaForAntigravity({ + type: "object", + properties: { + options: { + additionalProperties: { type: "string" }, + }, + }, + }) as Record; + + const properties = cleaned.properties as Record>; + assert.deepEqual(properties.options, {}); + assert.equal("properties" in properties.options, false); +}); + +test("#12269 strips unsupported validation keywords without wrapping them as property maps", () => { + // These keys are in GEMINI_UNSUPPORTED_SCHEMA_KEYS (Gemini 400s on them); + // they must be REMOVED, and removal must not go through the bare-map lift + // (which would turn `{minLength: 1}` into `{type:"object", properties:{...}}`). + for (const [keyword, value] of Object.entries({ + minLength: 1, + maxLength: 8, + multipleOf: 2, + minItems: 1, + maxItems: 4, + uniqueItems: true, + })) { + const cleaned = cleanJSONSchemaForAntigravity({ + type: "object", + properties: { + value: { [keyword]: value }, + }, + }) as Record; + + const properties = cleaned.properties as Record>; + assert.deepEqual(properties.value, {}, `unsupported ${keyword} must be stripped`); + assert.equal("properties" in properties.value, false); + } +}); + +test("#12269 preserves supported validation keywords without wrapping them", () => { + // `minimum`/`maximum`/`pattern` are accepted by Antigravity and must survive + // untouched; `minProperties`/`maxProperties` are not in the strip set either. + for (const [keyword, value] of Object.entries({ + minimum: 0, + maximum: 10, + pattern: "^[a-z]+$", + minProperties: 1, + })) { + const cleaned = cleanJSONSchemaForAntigravity({ + type: "object", + properties: { + value: { [keyword]: value }, + }, + }) as Record; + + const properties = cleaned.properties as Record>; + assert.deepEqual(properties.value, { [keyword]: value }, `supported ${keyword} must be preserved`); + assert.equal("properties" in properties.value, false); + } +}); + +test("#12269 recursively lifts more than one nested bare property map", () => { + const cleaned = cleanJSONSchemaForAntigravity({ + type: "object", + properties: { + opts: { settings: { limit: { type: "number" } } }, + }, + }) as Record; + + const opts = (cleaned.properties as Record>).opts; + const settings = (opts.properties as Record>).settings; + assert.equal(opts.type, "object"); + assert.equal(settings.type, "object"); + assert.equal( + (settings.properties as Record>).limit.type, + "number" + ); +}); + +test("#12269 preserves a pre-existing parent required entry without duplication", () => { + const cleaned = cleanJSONSchemaForAntigravity({ + type: "object", + properties: { + query: { type: "string", required: true }, + }, + required: ["query"], + }) as Record; + + assert.deepEqual(cleaned.required, ["query"]); +}); + +test("#12269 removes every non-array property-level required value", () => { + const cleaned = cleanJSONSchemaForAntigravity({ + type: "object", + properties: { + numeric: { type: "string", required: 1 }, + textual: { type: "string", required: "yes" }, + nil: { type: "string", required: null }, + }, + }) as Record; + + const properties = cleaned.properties as Record>; + assert.equal("required" in properties.numeric, false); + assert.equal("required" in properties.textual, false); + assert.equal("required" in properties.nil, false); + assert.equal(cleaned.required, undefined); +}); + +test("#12269 does not promote an object whose only child is an array", () => { + const cleaned = cleanJSONSchemaForAntigravity({ + type: "object", + properties: { + malformed: { values: [1, 2, 3] }, + }, + }) as Record; + + const malformed = (cleaned.properties as Record>).malformed; + assert.equal(malformed.type, undefined); + assert.equal(malformed.properties, undefined); +}); + +test("#12269 promotes required:true from a typed object child before cleaning it", () => { + const cleaned = cleanJSONSchemaForAntigravity({ + type: "object", + properties: { + config: { + type: "object", + required: true, + properties: { + timeout: { type: "number" }, + }, + }, + }, + }) as Record; + + assert.deepEqual(cleaned.required, ["config"]); + const properties = cleaned.properties as Record>; + assert.equal("required" in properties.config, false); +}); + +test("#12269 preserves a well-formed object schema byte-stable on required/properties", () => { + const input = { + type: "object", + properties: { + query: { type: "string" }, + limit: { type: "number" }, + }, + required: ["query"], + }; + const cleaned = cleanJSONSchemaForAntigravity(input) as Record; + const properties = cleaned.properties as Record>; + assert.equal(properties.query.type, "string"); + assert.equal(properties.limit.type, "number"); + assert.deepEqual(cleaned.required, ["query"]); +}); From 36be267a1778fa7d004b4037d79e783baa2f4844 Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Thu, 3 Sep 2026 22:43:38 -0400 Subject: [PATCH 093/143] fix(providers): add CLAUDE_CODE_CLIENT_VERSION and GITHUB_COPILOT_CLI_VERSION env overrides (#12632) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 4 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **119/119** nos 9 arquivos de teste que trazem. Três dos quatro conflitavam apenas no `config/quality/file-size-baseline.json`, todos de forma aditiva (chaves `_rebaseline_` distintas que devem coexistir); resolvidos com validação de JSON a cada passo. Registro que o **#12637 não é duplicata do #12566**, apesar do título quase idêntico: o autor documenta que aquele escopou o cooldown de preflight por família e este cobre o `genericQuotaFetcher`, que é o que o roteamento reset-aware efetivamente chama. Traz também validação ao vivo em VPS (imagem X500, `onmi-gemini3.6` → HTTP 200), satisfazendo a Hard Rule #18. Obrigado, @HouMinXi. --- .env.example | 10 ++ changelog.d/fixes/12417-cli-version-env.md | 1 + docs/reference/ENVIRONMENT.md | 2 + open-sse/config/anthropicHeaders.ts | 11 ++ .../config/claudeCodeCompatibleIdentity.ts | 6 + open-sse/config/cliFingerprints.ts | 4 +- open-sse/config/glmProvider.ts | 3 +- open-sse/config/providerHeaderProfiles.ts | 43 +++++- .../config/providers/registry/claude/index.ts | 1 - open-sse/config/providers/shared.ts | 3 +- open-sse/executors/base.ts | 8 +- open-sse/executors/claudeIdentity.ts | 6 +- open-sse/services/ccBridgeTransforms.ts | 6 +- open-sse/services/claudeCodeCompatible.ts | 4 +- open-sse/services/usage/claude.ts | 4 +- src/lib/oauth/constants/oauth.ts | 3 + src/lib/oauth/providers/claude.ts | 4 +- src/lib/oauth/providers/ghe-copilot.ts | 5 +- src/lib/oauth/providers/github.ts | 5 +- src/shared/constants/claudeCodeClient.ts | 27 +++- .../unit/cli-client-version-env-12417.test.ts | 130 ++++++++++++++++++ 21 files changed, 257 insertions(+), 29 deletions(-) create mode 100644 changelog.d/fixes/12417-cli-version-env.md create mode 100644 tests/unit/cli-client-version-env-12417.test.ts diff --git a/.env.example b/.env.example index e0fa9d6bed..67097b7104 100644 --- a/.env.example +++ b/.env.example @@ -1332,6 +1332,16 @@ CURSOR_USER_AGENT="Cursor/3.4" # Override Codex client version sent in headers independently of the # CODEX_USER_AGENT string. Used by: open-sse/config/codexClient.ts. # CODEX_CLIENT_VERSION=0.144.1 +# +# Override the advertised Claude Code client version independently of +# CLAUDE_USER_AGENT. Anthropic gates some models (Fable 5.1) on this +# value; a UA-only override is not enough (#12417). Used by: +# src/shared/constants/claudeCodeClient.ts. +# CLAUDE_CODE_CLIENT_VERSION=2.1.259 +# +# Override the advertised GitHub Copilot CLI version independently of +# GITHUB_USER_AGENT. Used by: open-sse/config/providerHeaderProfiles.ts. +# GITHUB_COPILOT_CLI_VERSION=1.0.82 # Kill-switch to strip non-standard `codex.*` SSE events (e.g. codex.rate_limits) # from the Codex Responses stream. These frames break the OpenAI SDK's diff --git a/changelog.d/fixes/12417-cli-version-env.md b/changelog.d/fixes/12417-cli-version-env.md new file mode 100644 index 0000000000..5fbe8327a1 --- /dev/null +++ b/changelog.d/fixes/12417-cli-version-env.md @@ -0,0 +1 @@ +- **fix(providers):** add `CLAUDE_CODE_CLIENT_VERSION` and `GITHUB_COPILOT_CLI_VERSION` env overrides so Anthropic/Copilot client-version gates can be unblocked without a rebuild ([#12417](https://github.com/diegosouzapw/OmniRoute/issues/12417)) diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index 8b743b2ecc..da053f1db3 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -636,6 +636,8 @@ process.env[`${PROVIDER_ID}_USER_AGENT`] | `CLAUDE_DISABLE_TOOL_NAME_CLOAK` | `false` | `executors/base.ts` + `executors/cliproxyapi.ts` | Set to `1`/`true` to forward third-party harness tool names verbatim to Anthropic on both Anthropic-bound paths (native OAuth and CLIProxyAPI). By default the executor deterministically aliases non-Claude-Code tool names (Claude Code canonical mapping where one exists, otherwise PascalCase) and reverses them on the response via `_toolNameMap`, so harnesses with snake_case tools are not refused as fingerprinted third-party clients. Debugging only. | | `CODEX_USER_AGENT` | `codex-cli/0.142.0 (Windows 10.0.26200; x64)` | When OpenAI updates the Codex CLI | | `CODEX_CLIENT_VERSION` | `0.131.0` | Override Codex client version independently of full UA string | +| `CLAUDE_CODE_CLIENT_VERSION` | `2.1.258` | Override advertised Claude Code version independently of `CLAUDE_USER_AGENT`. Anthropic gates some models on this value (#12417). | +| `GITHUB_COPILOT_CLI_VERSION` | `1.0.81-6` | Override advertised Copilot CLI version independently of `GITHUB_USER_AGENT` | | `GITHUB_USER_AGENT` | `GitHubCopilotChat/0.54.0` | When GitHub Copilot Chat updates | | `ANTIGRAVITY_USER_AGENT` | `antigravity/2.0.1 darwin/arm64` | When Antigravity IDE updates | | `KIRO_USER_AGENT` | `AWS-SDK-JS/3.0.0 kiro-ide/1.0.0` | When Kiro IDE updates | diff --git a/open-sse/config/anthropicHeaders.ts b/open-sse/config/anthropicHeaders.ts index a030cd1c4a..6625c4dd07 100644 --- a/open-sse/config/anthropicHeaders.ts +++ b/open-sse/config/anthropicHeaders.ts @@ -4,6 +4,8 @@ import { CLAUDE_CODE_CLIENT_VERSION, CLAUDE_CODE_RUNTIME_VERSION, CLAUDE_CODE_SDK_PACKAGE_VERSION, + getClaudeCodeClientBillingVersion, + getClaudeCodeClientVersion, getClaudeCodeUserAgent, } from "@/shared/constants/claudeCodeClient"; import { modelSupportsContext1mBeta } from "../config/context1m.ts"; @@ -166,8 +168,17 @@ export function normalizeAnthropicHeaderVariants(headers: Record } export const CLAUDE_CLI_VERSION = CLAUDE_CODE_CLIENT_VERSION; +export function getClaudeCliVersion(): string { + return getClaudeCodeClientVersion(); +} export const CLAUDE_CLI_BUILD_REVISION = CLAUDE_CODE_CLIENT_BUILD_REVISION; +/** Captured-pin snapshot. Wire billing uses getClaudeCliBillingVersion(). */ export const CLAUDE_CLI_BILLING_VERSION = CLAUDE_CODE_CLIENT_BILLING_VERSION; +export function getClaudeCliBillingVersion(): string { + return getClaudeCodeClientBillingVersion(); +} +/** Module-load snapshot of the pin (or env if set before import). Wire UA uses getClaudeCodeUserAgent(). */ export const CLAUDE_CLI_USER_AGENT = getClaudeCodeUserAgent("cli"); +export { getClaudeCodeUserAgent }; export const CLAUDE_CLI_STAINLESS_PACKAGE_VERSION = CLAUDE_CODE_SDK_PACKAGE_VERSION; export const CLAUDE_CLI_STAINLESS_RUNTIME_VERSION = CLAUDE_CODE_RUNTIME_VERSION; diff --git a/open-sse/config/claudeCodeCompatibleIdentity.ts b/open-sse/config/claudeCodeCompatibleIdentity.ts index b614eea324..29f996278c 100644 --- a/open-sse/config/claudeCodeCompatibleIdentity.ts +++ b/open-sse/config/claudeCodeCompatibleIdentity.ts @@ -2,11 +2,17 @@ import { CLAUDE_CODE_CLIENT_VERSION, CLAUDE_CODE_RUNTIME_VERSION, CLAUDE_CODE_SDK_PACKAGE_VERSION, + getClaudeCodeClientVersion, getClaudeCodeUserAgent, } from "@/shared/constants/claudeCodeClient"; export const CLAUDE_CODE_COMPATIBLE_VERSION = CLAUDE_CODE_CLIENT_VERSION; +export function getClaudeCodeCompatibleVersion(): string { + return getClaudeCodeClientVersion(); +} +/** Module-load snapshot. Wire UA uses getClaudeCodeUserAgent("sdk-cli"). */ export const CLAUDE_CODE_COMPATIBLE_USER_AGENT = getClaudeCodeUserAgent("sdk-cli"); +export { getClaudeCodeUserAgent }; export const CLAUDE_CODE_COMPATIBLE_STAINLESS_PACKAGE_VERSION = CLAUDE_CODE_SDK_PACKAGE_VERSION; export const CLAUDE_CODE_COMPATIBLE_STAINLESS_RUNTIME_VERSION = CLAUDE_CODE_RUNTIME_VERSION; const CONTEXT_1M_NATIVE_MODELS = ["claude-fable-5-1", "claude-opus-5"]; diff --git a/open-sse/config/cliFingerprints.ts b/open-sse/config/cliFingerprints.ts index efa902e137..8891f73d9d 100644 --- a/open-sse/config/cliFingerprints.ts +++ b/open-sse/config/cliFingerprints.ts @@ -12,7 +12,7 @@ import { isClaudeCodeCompatible } from "../services/provider.ts"; import { getAntigravityUserAgent, - GITHUB_COPILOT_CHAT_USER_AGENT, + getGitHubCopilotChatUserAgent, } from "./providerHeaderProfiles.ts"; import { normalizeCliCompatProviderId } from "@/shared/utils/cliCompat"; @@ -169,7 +169,7 @@ export const CLI_FINGERPRINTS: Record = { "intent_threshold", "intent_content", ], - userAgent: GITHUB_COPILOT_CHAT_USER_AGENT, + userAgent: getGitHubCopilotChatUserAgent, }, antigravity: { headerOrder: [ diff --git a/open-sse/config/glmProvider.ts b/open-sse/config/glmProvider.ts index 3bb6c85210..22d05864e7 100644 --- a/open-sse/config/glmProvider.ts +++ b/open-sse/config/glmProvider.ts @@ -255,6 +255,7 @@ export const GLMT_REQUEST_DEFAULTS = Object.freeze({ }); export const GLM_COUNT_TOKENS_TIMEOUT_MS = 3_000; +/** Module-load snapshot. Wire UA uses getClaudeCodeUserAgent("sdk-cli"). */ export const GLM_CLAUDE_CODE_USER_AGENT = getClaudeCodeUserAgent("sdk-cli"); export const GLM_ANTHROPIC_BETA = [ "claude-code-20250219", @@ -582,7 +583,7 @@ export function buildGlmBaseHeaders(apiKey: string, stream = true): Record = { "copilot-integration-id": GITHUB_COPILOT_INTEGRATION_ID, - "editor-version": GITHUB_COPILOT_EDITOR_VERSION, - "user-agent": GITHUB_COPILOT_CLI_USER_AGENT, + "editor-version": `copilot/${version}`, + "user-agent": `copilot/${version}`, "openai-intent": options.intent || GITHUB_COPILOT_OPENAI_INTENT, "x-interaction-type": GITHUB_COPILOT_INTERACTION_TYPE, "copilot-harness-id": GITHUB_COPILOT_HARNESS_ID, @@ -128,23 +155,25 @@ export function getQwenCliUserAgent(version = QWEN_CLI_VERSION): string { } export function getGitHubCopilotInternalUserHeaders(authorization: string): Record { + const version = getGitHubCopilotCliVersion(); return { Authorization: authorization, Accept: "application/json", "X-GitHub-Api-Version": GITHUB_COPILOT_API_VERSION, - "User-Agent": GITHUB_COPILOT_CHAT_USER_AGENT, - "Editor-Version": GITHUB_COPILOT_EDITOR_VERSION, - "Editor-Plugin-Version": GITHUB_COPILOT_CHAT_PLUGIN_VERSION, + "User-Agent": `GitHubCopilotChat/${version}`, + "Editor-Version": `copilot/${version}`, + "Editor-Plugin-Version": `copilot-chat/${version}`, }; } export function getGitHubCopilotRefreshHeaders(authorization: string): Record { + const version = getGitHubCopilotCliVersion(); return { Authorization: authorization, Accept: "application/json", "User-Agent": GITHUB_COPILOT_REFRESH_USER_AGENT, - "Editor-Version": GITHUB_COPILOT_EDITOR_VERSION, - "Editor-Plugin-Version": GITHUB_COPILOT_REFRESH_PLUGIN_VERSION, + "Editor-Version": `copilot/${version}`, + "Editor-Plugin-Version": `copilot/${version}`, }; } diff --git a/open-sse/config/providers/registry/claude/index.ts b/open-sse/config/providers/registry/claude/index.ts index ba3129678b..8388812300 100644 --- a/open-sse/config/providers/registry/claude/index.ts +++ b/open-sse/config/providers/registry/claude/index.ts @@ -7,7 +7,6 @@ import { ANTHROPIC_VERSION_HEADER, CLAUDE_CLI_STAINLESS_PACKAGE_VERSION, CLAUDE_CLI_STAINLESS_RUNTIME_VERSION, - CLAUDE_CLI_USER_AGENT, resolvePublicCred, } from "../../shared.ts"; diff --git a/open-sse/config/providers/shared.ts b/open-sse/config/providers/shared.ts index 6c67c4eb5f..8b9b45bfec 100644 --- a/open-sse/config/providers/shared.ts +++ b/open-sse/config/providers/shared.ts @@ -16,6 +16,7 @@ import { CLAUDE_CLI_STAINLESS_PACKAGE_VERSION, CLAUDE_CLI_STAINLESS_RUNTIME_VERSION, CLAUDE_CLI_USER_AGENT, + getClaudeCodeUserAgent, } from "../anthropicHeaders.ts"; import { getCodexDefaultHeaders } from "../codexClient.ts"; import { @@ -761,7 +762,7 @@ export function getClaudeCliHeaders(): Record { "Anthropic-Version": ANTHROPIC_VERSION_HEADER, "Anthropic-Beta": ANTHROPIC_BETA_CLAUDE_OAUTH, "Anthropic-Dangerous-Direct-Browser-Access": "true", - "User-Agent": CLAUDE_CLI_USER_AGENT, + "User-Agent": getClaudeCodeUserAgent("cli"), "X-App": "cli", "X-Stainless-Helper-Method": "stream", "X-Stainless-Retry-Count": "0", diff --git a/open-sse/executors/base.ts b/open-sse/executors/base.ts index 11d5f9087c..b18c27e20c 100644 --- a/open-sse/executors/base.ts +++ b/open-sse/executors/base.ts @@ -6,8 +6,8 @@ import { type AlternateFormat, } from "../config/providers/alternateFormats.ts"; import { - CLAUDE_CLI_BILLING_VERSION, CLAUDE_CLI_STAINLESS_RUNTIME_VERSION, + getClaudeCliBillingVersion, mergeClientAnthropicBeta, normalizeAnthropicHeaderVariants, } from "../config/anthropicHeaders.ts"; @@ -86,7 +86,7 @@ import { } from "../services/contextManager.ts"; import { randomUUID } from "node:crypto"; import { - CLAUDE_CODE_VERSION, + getClaudeCodeVersion, CLAUDE_CODE_STAINLESS_VERSION, buildUserIdJson, getSessionId, @@ -1163,7 +1163,7 @@ export class BaseExecutor { // system[0] (billing) and system[1] (sentinel) must not carry // cache_control — that belongs on upstream prompt blocks at [2..]. - const billingLine = `x-anthropic-billing-header: cc_version=${CLAUDE_CLI_BILLING_VERSION}; cc_entrypoint=cli; cch=00000;`; + const billingLine = `x-anthropic-billing-header: cc_version=${getClaudeCliBillingVersion()}; cc_entrypoint=cli; cch=00000;`; const SENTINEL = "You are Claude Code, Anthropic's official CLI for Claude."; const sysBlocks: Array> = Array.isArray(tb.system) @@ -1259,7 +1259,7 @@ export class BaseExecutor { ), "anthropic-dangerous-direct-browser-access": "true", "x-app": "cli", - "User-Agent": `claude-cli/${CLAUDE_CODE_VERSION} (external, cli)`, + "User-Agent": `claude-cli/${getClaudeCodeVersion()} (external, cli)`, "X-Stainless-Package-Version": CLAUDE_CODE_STAINLESS_VERSION, "X-Stainless-Timeout": "600", "accept-encoding": "gzip, deflate, br, zstd", diff --git a/open-sse/executors/claudeIdentity.ts b/open-sse/executors/claudeIdentity.ts index 78ee1c8b0f..92126d4a45 100644 --- a/open-sse/executors/claudeIdentity.ts +++ b/open-sse/executors/claudeIdentity.ts @@ -13,11 +13,15 @@ import { createHash, randomBytes, randomUUID } from "node:crypto"; import { CLAUDE_CODE_CLIENT_VERSION, CLAUDE_CODE_SDK_PACKAGE_VERSION, + getClaudeCodeClientVersion, } from "@/shared/constants/claudeCodeClient"; // ---------- Versions ------------------------------------------------------ export const CLAUDE_CODE_VERSION = CLAUDE_CODE_CLIENT_VERSION; +export function getClaudeCodeVersion(): string { + return getClaudeCodeClientVersion(); +} /** Bundled @anthropic-ai/sdk version for the pinned CLI release. */ export const CLAUDE_CODE_STAINLESS_VERSION = CLAUDE_CODE_SDK_PACKAGE_VERSION; @@ -156,7 +160,7 @@ export async function fetchClaudeBootstrap(accessToken: string): Promise( + entries: Record, + fn: () => T | Promise +): Promise { + const previous = new Map(); + for (const [key, value] of Object.entries(entries)) { + previous.set(key, process.env[key]); + if (value === undefined) { + delete process.env[key]; + } else { + process.env[key] = value; + } + } + try { + return await fn(); + } finally { + for (const [key, value] of previous.entries()) { + if (value === undefined) { + delete process.env[key]; + } else { + process.env[key] = value; + } + } + } +} + +test("#12417 Claude pin stays the captured 2.1.258 binary", () => { + assert.equal(canonical.CLAUDE_CODE_CLIENT_VERSION, "2.1.258"); +}); + +test("#12417 getClaudeCodeClientVersion falls back to the captured pin", async () => { + await withEnv({ CLAUDE_CODE_CLIENT_VERSION: undefined }, () => { + assert.equal(canonical.getClaudeCodeClientVersion(), canonical.CLAUDE_CODE_CLIENT_VERSION); + }); +}); + +test("#12417 getClaudeCodeClientVersion honors a safe env override", async () => { + await withEnv({ CLAUDE_CODE_CLIENT_VERSION: "2.1.259" }, () => { + assert.equal(canonical.getClaudeCodeClientVersion(), "2.1.259"); + assert.equal(canonical.getClaudeCodeUserAgent("cli"), "claude-cli/2.1.259 (external, cli)"); + assert.equal( + canonical.getClaudeCodeUserAgent("sdk-cli"), + "claude-cli/2.1.259 (external, sdk-cli)" + ); + assert.equal( + canonical.getClaudeCodeClientBillingVersion(), + `2.1.259.${canonical.CLAUDE_CODE_CLIENT_BUILD_REVISION}` + ); + }); +}); + +test("#12417 getClaudeCodeClientVersion ignores an unsafe env override", async () => { + await withEnv({ CLAUDE_CODE_CLIENT_VERSION: "bad version value" }, () => { + assert.equal(canonical.getClaudeCodeClientVersion(), canonical.CLAUDE_CODE_CLIENT_VERSION); + }); +}); + +test("#12417 Copilot pin stays the captured 1.0.81-6 CLI", () => { + assert.equal(copilot.GITHUB_COPILOT_CLI_VERSION, "1.0.81-6"); +}); + +test("#12417 getGitHubCopilotCliVersion falls back to the captured pin", async () => { + await withEnv({ GITHUB_COPILOT_CLI_VERSION: undefined }, () => { + assert.equal(copilot.getGitHubCopilotCliVersion(), copilot.GITHUB_COPILOT_CLI_VERSION); + }); +}); + +test("#12417 getGitHubCopilotChatHeaders honors a safe env override", async () => { + await withEnv({ GITHUB_COPILOT_CLI_VERSION: "1.0.82" }, () => { + assert.equal(copilot.getGitHubCopilotCliVersion(), "1.0.82"); + const headers = copilot.getGitHubCopilotChatHeaders(); + assert.equal(headers["user-agent"], "copilot/1.0.82"); + assert.equal(headers["editor-version"], "copilot/1.0.82"); + }); +}); + +test("#12417 getGitHubCopilotCliVersion ignores an unsafe env override", async () => { + await withEnv({ GITHUB_COPILOT_CLI_VERSION: "not a version" }, () => { + assert.equal(copilot.getGitHubCopilotCliVersion(), copilot.GITHUB_COPILOT_CLI_VERSION); + }); +}); + +test("#12417 getClaudeCliHeaders reads the env at call time", async () => { + await withEnv({ CLAUDE_CODE_CLIENT_VERSION: "2.1.259" }, () => { + assert.equal( + claudeHeaders.getClaudeCliHeaders()["User-Agent"], + "claude-cli/2.1.259 (external, cli)" + ); + }); +}); + +test("#12417 applyFingerprint Copilot UA follows the env, pin const does not", async () => { + const fingerprints = await import("../../open-sse/config/cliFingerprints.ts"); + await withEnv({ GITHUB_COPILOT_CLI_VERSION: "1.0.82" }, () => { + const result = fingerprints.applyFingerprint( + "copilot", + { Authorization: "Bearer token", Accept: "application/json" }, + { model: "gpt-4o", messages: [] } + ); + assert.equal(result.headers["User-Agent"], "GitHubCopilotChat/1.0.82"); + assert.equal(copilot.GITHUB_COPILOT_CHAT_USER_AGENT, "GitHubCopilotChat/1.0.81-6"); + }); +}); + +test("#12417 Claude billing pin stays captured while getter follows env", async () => { + const hdr = await import("../../open-sse/config/anthropicHeaders.ts"); + await withEnv({ CLAUDE_CODE_CLIENT_VERSION: "2.1.259" }, () => { + assert.equal(hdr.CLAUDE_CLI_BILLING_VERSION, canonical.CLAUDE_CODE_CLIENT_BILLING_VERSION); + assert.equal( + hdr.getClaudeCliBillingVersion(), + `2.1.259.${canonical.CLAUDE_CODE_CLIENT_BUILD_REVISION}` + ); + }); +}); From d36d077a4d87ce48156278bb4682aa9f29349268 Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Thu, 3 Sep 2026 22:45:47 -0400 Subject: [PATCH 094/143] fix(resilience): keep Overloaded STREAM_EARLY_EOF off the provider breaker (#12626) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 4 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **119/119** nos 9 arquivos de teste que trazem. Três dos quatro conflitavam apenas no `config/quality/file-size-baseline.json`, todos de forma aditiva (chaves `_rebaseline_` distintas que devem coexistir); resolvidos com validação de JSON a cada passo. Registro que o **#12637 não é duplicata do #12566**, apesar do título quase idêntico: o autor documenta que aquele escopou o cooldown de preflight por família e este cobre o `genericQuotaFetcher`, que é o que o roteamento reset-aware efetivamente chama. Traz também validação ao vivo em VPS (imagem X500, `onmi-gemini3.6` → HTTP 200), satisfazendo a Hard Rule #18. Obrigado, @HouMinXi. --- config/quality/file-size-baseline.json | 5 +- open-sse/services/combo.ts | 41 ++++- open-sse/services/combo/comboCooldownRetry.ts | 34 ++++ open-sse/services/combo/comboPredicates.ts | 17 +- src/shared/utils/circuitBreaker.ts | 24 +++ src/sse/handlers/chatPredicates.ts | 3 + stryker.conf.json | 1 + .../unit/breaker-network-error-guard.test.ts | 2 +- tests/unit/combo-cooldown-retry.test.ts | 86 ++++++++- ...dex-turn-pin-model-scoped-fallback.test.ts | 1 + .../overloaded-not-provider-breaker.test.ts | 169 ++++++++++++++++++ tests/unit/repro-9630-combo-false-503.test.ts | 6 +- 12 files changed, 377 insertions(+), 12 deletions(-) create mode 100644 tests/unit/overloaded-not-provider-breaker.test.ts diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index cc292a2f3b..a7941e072e 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -1,4 +1,5 @@ { + "_rebaseline_2026_09_03_overloaded_not_provider_breaker": "fix/overloaded-not-provider-breaker own growth: open-sse/services/combo.ts 4036->4075 (check-file-size split-newline, +39). Circuit-open pre-skip now records the breaker retryAfter and, when every target was skipped that way, waits the short reset via resolveCircuitOpenWaitDecision (new leaf in comboCooldownRetry.ts) instead of crystallizing ALL_TARGETS_SKIPPED in ~43ms. skippedForCircuitOpen / earliestCircuitOpenRetryMs reset each setTry so a later iteration cannot inherit a stale retryAfter. Irreducible at the existing ALL_TARGETS_SKIPPED chokepoint (same pattern as #7301/#8213 cooldown-wait). Predicate itself lives in circuitBreaker.ts / comboPredicates.ts / chatPredicates.ts, all under cap. Covered by tests/unit/overloaded-not-provider-breaker.test.ts + combo-cooldown-retry.test.ts.", "_rebaseline_2026_09_03_12649_free_tier_reaudit_gateways": "PR #12649 (fix/free-tier-quota-reaudit) own growth: src/shared/constants/providers/apikey/gateways.ts 1459->1462 (+3 = the nara authHint rewritten for the re-audited 7M/day plan now wraps to two lines, plus the Prettier reflow of two pre-existing >100-col authHint lines (oneminai, freebuff) that lint-staged enforces on any touch of the file; additive text at the existing registry chokepoint, same god-file no-split rationale as prior gateways.ts rebaselines: #11786 seekai, #10987 logfare, #10531 freebuff). Covered by tests/unit/free-tier-reaudit-2026-09.test.ts and tests/unit/free-providers-batch-2026-07.test.ts.", "_rebaseline_2026_09_03_moonshot_native_quota": "PR feat/moonshot-native-quota own growth on release/v3.8.51: src/lib/db/migrationRunner.ts 1201->1206 (+5, case 172 retroactive guard for daily_quota_reset_* columns); src/sse/handlers/chat.ts 2434->2450 (+16, registerMoonshotQuotaFetcher + startup node scan at the existing quota-fetcher registration chokepoint); src/sse/services/auth.ts 3427->3450 (+23, resolveDailyResetForProvider + dailyReset arg on checkFallbackError); open-sse/services/accountFallback.ts 2422->2461 (+39, compatible-node credits_exhausted carve-out + TPD node-clock lock); tests/unit/account-fallback-service.test.ts 2008->2056 (+48, TPD/empty-wallet cases). Wiring at existing chokepoints; Moonshot host predicates, daily reset clock, and the balance fetcher live in new leaves under cap. Covered by tests/unit/moonshot-*.test.ts + account-fallback-service.test.ts (135/135 focused).", "_rebaseline_2026_09_02_11786_seekai_provider": "PR #11786 (feat/11786-seekai-provider, closes #11786) own growth: src/shared/constants/providers/apikey/gateways.ts 1438->1458 (check-file-size split-newline=1459; the seekai APIKEY_PROVIDERS_GATEWAYS catalog entry plus authHint, additive data at the existing registry chokepoint, same god-file no-split rationale as prior gateways.ts rebaselines: #10987 logfare, #10531 freebuff). Covered by tests/unit/seekai-provider.test.ts.", @@ -423,7 +424,7 @@ "open-sse/mcp-server/server.ts": 1572, "open-sse/services/accountFallback.ts": 2467, "open-sse/services/adobeFireflyBrowserLogin.ts": 1401, - "open-sse/services/combo.ts": 4036, + "open-sse/services/combo.ts": 4075, "open-sse/translator/response/openai-responses.ts": 1466, "open-sse/utils/cursorAgentProtobuf.ts": 1547, "open-sse/utils/proxyFetch.ts": 1271, @@ -556,7 +557,7 @@ "open-sse/services/accountFallback.ts": "1978", "open-sse/services/adobeFireflyClient.ts": "2385", "open-sse/services/claudeCodeCompatible.ts": "1202", - "open-sse/services/combo.ts": "3648", + "open-sse/services/combo.ts": "4075", "open-sse/services/compression/strategySelector.ts": "1060", "open-sse/services/rateLimitManager.ts": "1167", "open-sse/translator/response/openai-responses.ts": "1204", diff --git a/open-sse/services/combo.ts b/open-sse/services/combo.ts index be662c600f..635001b646 100644 --- a/open-sse/services/combo.ts +++ b/open-sse/services/combo.ts @@ -178,6 +178,7 @@ import { } from "./combo/validateQuality.ts"; import { resolveComboCooldownWaitDecision, + resolveCircuitOpenWaitDecision, ResolveComboCooldownDecisionResult, } from "./combo/comboCooldownRetry.ts"; import { @@ -1133,6 +1134,8 @@ async function handleComboChatInner({ let lastError: string | null = null; let earliestRetryAfter: ComboRetryAfter | null = null; let lastStatus: number | null = null; + let skippedForCircuitOpen = false; + let earliestCircuitOpenRetryMs = 0; // #11804: the loop-safety timer is armed per setTry iteration but must be // cleared on EVERY exit path, not just the happy one. Hoisted to function // scope so the `finally` at the end of this function always reaches it — @@ -1151,6 +1154,8 @@ async function handleComboChatInner({ const exhaustedProviders = new Set(); const exhaustedConnections = new Set(); const transientRateLimitedProviders = new Set(); + skippedForCircuitOpen = false; + earliestCircuitOpenRetryMs = 0; if (setTry > 0) { log.info("COMBO", `All targets failed — retrying set (${setTry}/${maxSetRetries})`); await new Promise((resolve) => { @@ -1272,7 +1277,15 @@ async function handleComboChatInner({ }; const cb = getCircuitBreaker(provider); - if (cb.getStatus().state === "OPEN") { + const cbStatus = cb.getStatus(); + if (cbStatus.state === "OPEN") { + skippedForCircuitOpen = true; + if ( + cbStatus.retryAfterMs > 0 && + (earliestCircuitOpenRetryMs === 0 || cbStatus.retryAfterMs < earliestCircuitOpenRetryMs) + ) { + earliestCircuitOpenRetryMs = cbStatus.retryAfterMs; + } log.info("COMBO", `Skipping ${modelStr} — circuit breaker OPEN for ${provider}`); recordComboDecision(traceInvocationId, { step: target.executionKey, @@ -2762,6 +2775,32 @@ async function handleComboChatInner({ // Retry the entire set if more attempts remain if (setTry < maxSetRetries) continue; + if (!lastStatus && recordedAttempts === 0 && comboCooldownWaitEnabled) { + const circuitOpenWait = resolveCircuitOpenWaitDecision({ + skippedForCircuitOpen, + retryAfterMs: earliestCircuitOpenRetryMs, + attempt: comboCooldownAttempt, + budgetLeftMs: comboCooldownBudgetLeftMs, + settings: resilienceSettings.comboCooldownWait, + }); + if (circuitOpenWait.wait) { + log.info( + "COMBO", + `${strategy} circuit-open wait: waiting ${Math.ceil(circuitOpenWait.waitMs / 1000)}s (reason=${circuitOpenWait.reason ?? "circuit_open"}) then retrying (attempt ${comboCooldownAttempt + 1}/${resilienceSettings.comboCooldownWait.maxAttempts})` + ); + const completed = await waitForCooldownAwareRetry(circuitOpenWait.waitMs, signal); + if (!completed) { + return errorResponse(499, "Request aborted"); + } + comboCooldownAttempt += 1; + comboCooldownBudgetLeftMs = Math.max( + 0, + comboCooldownBudgetLeftMs - circuitOpenWait.waitMs + ); + return dispatchWithCooldownRetry(); + } + } + // All set retries exhausted — return the final error // #10681: finalize the decision trace (all targets failed or skipped). finalizeComboTrace(traceInvocationId, orderedTargets); diff --git a/open-sse/services/combo/comboCooldownRetry.ts b/open-sse/services/combo/comboCooldownRetry.ts index 4b2b56609a..4b96552b72 100644 --- a/open-sse/services/combo/comboCooldownRetry.ts +++ b/open-sse/services/combo/comboCooldownRetry.ts @@ -56,6 +56,7 @@ export const COMBO_COOLDOWN_RETRYABLE_REASONS: ReadonlySet = new Set([ "transient", "overloaded", "server_error", + "circuit_open", ]); export interface ComboCooldownWaitSettings { @@ -256,3 +257,36 @@ export function resolveComboCooldownWaitDecision( reason: typeof best.reason === "string" ? best.reason : null, }; } + +export interface ResolveCircuitOpenWaitInput { + skippedForCircuitOpen: unknown; + retryAfterMs: unknown; + attempt: number; + budgetLeftMs: number; + settings: ComboCooldownWaitSettings; +} + +/** + * When every combo target was pre-skipped because the whole-provider breaker is + * OPEN, wait out a SHORT reset instead of crystallizing ALL_TARGETS_SKIPPED. + * Same ceilings as model-lockout waits. Live incident 2026-09-03: offical-fable + * (single claude target) returned 43ms 503 while the breaker reset was 60s. + */ +export function resolveCircuitOpenWaitDecision( + input: ResolveCircuitOpenWaitInput +): ResolveComboCooldownDecisionResult { + if (input.settings.enabled !== true || input.skippedForCircuitOpen !== true) { + return { wait: false, waitMs: 0, reason: null }; + } + const retryAfterMs = toFiniteWaitMs(input.retryAfterMs); + if (retryAfterMs <= 0) return { wait: false, waitMs: 0, reason: null }; + const waitMs = retryAfterMs + COMBO_COOLDOWN_WAIT_MARGIN_MS; + const decision = shouldWaitForComboCooldown({ + reason: "circuit_open", + waitMs, + attempt: input.attempt, + budgetLeftMs: input.budgetLeftMs, + settings: input.settings, + }); + return { ...decision, reason: "circuit_open" }; +} diff --git a/open-sse/services/combo/comboPredicates.ts b/open-sse/services/combo/comboPredicates.ts index 972abc5765..6359607cac 100644 --- a/open-sse/services/combo/comboPredicates.ts +++ b/open-sse/services/combo/comboPredicates.ts @@ -11,7 +11,11 @@ import { remainingPercentFromQuotaWindows } from "../antigravityQuotaFamily.ts"; import { errorResponse } from "../../utils/error.ts"; import { parseModel } from "../model.ts"; import { isSelfInflictedUpstreamTimeout } from "../../handlers/chatCore/cooldownClassification.ts"; -import { isLocalStreamLifecycleError, isLocalExecutionError } from "@/shared/utils/circuitBreaker"; +import { + isLocalStreamLifecycleError, + isLocalExecutionError, + isModelCapacityOverloadError, +} from "@/shared/utils/circuitBreaker"; import { CONTEXT_OVERFLOW_PATTERNS, MODEL_ACCESS_DENIED_PATTERNS } from "../accountFallback.ts"; import { isResourceNotFoundResponse } from "../errorClassifier.ts"; import { getTrustedLocalRateLimitResponse } from "../rateLimitManager/errors.ts"; @@ -213,6 +217,12 @@ export function shouldRecordProviderBreakerFailure(args: { }): boolean { return ( (!args.isStreamReadinessFailure || args.isStreamEarlyEof === true) && + // Overloaded 502 (STREAM_EARLY_EOF wrapping "Overloaded") must not trip + // the whole-provider breaker. The status=529 check is defense in depth: + // 529 is not in PROVIDER_BREAKER_FAILURE_STATUSES today, but a later + // addition of 529 to that set must still stay off the breaker. + !isModelCapacityOverloadError(args.error) && + !isModelCapacityOverloadError(args.status) && PROVIDER_BREAKER_FAILURE_STATUSES.has(args.status) && (!args.sameProviderNext || args.isProxyUnreachable === true) && !args.skipProviderBreaker && @@ -441,10 +451,7 @@ export function quotaRemainingPercentFromQuota( const windows = record.windows; if (windows && typeof windows === "object" && !Array.isArray(windows)) { - const fromWindows = remainingPercentFromQuotaWindows( - windows as Record, - scope - ); + const fromWindows = remainingPercentFromQuotaWindows(windows as Record, scope); if (fromWindows !== null) return fromWindows; } diff --git a/src/shared/utils/circuitBreaker.ts b/src/shared/utils/circuitBreaker.ts index 02e4e67811..c84917f0cd 100644 --- a/src/shared/utils/circuitBreaker.ts +++ b/src/shared/utils/circuitBreaker.ts @@ -102,6 +102,30 @@ export function isLocalExecutionError(error: unknown): boolean { return LOCAL_EXECUTION_PATTERNS.some((p) => p.test(message)); } +/** + * Anthropic/Claude model-capacity overload (HTTP 529, body "Overloaded", or a + * STREAM_EARLY_EOF that wraps that body as 502). This is one model being + * capacity-throttled, not a whole-provider outage — the same account still + * serves sibling models. Must not trip the provider circuit breaker. + * + * Accepts an error object/string OR a numeric HTTP status (529). Callers + * pass both `error` and `status` at the two breaker predicates. + * + * Live incident 2026-09-03: STREAM_EARLY_EOF: Overloaded opened `claude` and + * a single-target combo then pre-skipped with ALL_TARGETS_SKIPPED in ~43ms. + */ +export function isModelCapacityOverloadError(error: unknown): boolean { + if (error === 529) return true; + if (typeof error === "number") return false; + if (!error) return false; + const errObj = typeof error === "object" ? (error as Record) : null; + if (errObj && (errObj.status === 529 || errObj.statusCode === 529)) return true; + const message = + typeof error === "string" ? error : typeof errObj?.message === "string" ? errObj.message : ""; + if (!message) return false; + return /\boverloaded(?:_error)?\b/i.test(message); +} + export const STATE = { CLOSED: "CLOSED", DEGRADED: "DEGRADED", diff --git a/src/sse/handlers/chatPredicates.ts b/src/sse/handlers/chatPredicates.ts index f1ccbbaab2..88911a41b5 100644 --- a/src/sse/handlers/chatPredicates.ts +++ b/src/sse/handlers/chatPredicates.ts @@ -1,6 +1,7 @@ import { isLocalStreamLifecycleError, isLocalExecutionError, + isModelCapacityOverloadError, } from "../../shared/utils/circuitBreaker"; import { isRequestScopedUpstreamFailure } from "./comboFailureLogging"; import { getTrustedLocalRateLimitResponse } from "@omniroute/open-sse/services/rateLimitManager/errors"; @@ -39,6 +40,8 @@ export function shouldTripProviderBreakerForResult( result.errorCode !== "proxy_unreachable" && result.errorCode !== "RATE_LIMIT_QUEUE_TIMEOUT" && result.errorCode !== "RATE_LIMIT_QUEUE_WEDGED" && + !isModelCapacityOverloadError(result.error) && + !isModelCapacityOverloadError(result.status) && PROVIDER_BREAKER_FAILURE_STATUSES.has(Number(result.status)) ); } diff --git a/stryker.conf.json b/stryker.conf.json index c2986bcb28..c8ff964e77 100644 --- a/stryker.conf.json +++ b/stryker.conf.json @@ -152,6 +152,7 @@ "tests/unit/circuit-breaker-registry-cap.test.ts", "tests/unit/circuit-breaker-resolved-5xx-12254.test.ts", "tests/unit/circuit-breaker-stream-controller-4602.test.ts", + "tests/unit/overloaded-not-provider-breaker.test.ts", "tests/unit/claude-code-parity.test.ts", "tests/unit/claude-effort-suffix-strip.test.ts", "tests/unit/claude-oauth-provider.test.ts", diff --git a/tests/unit/breaker-network-error-guard.test.ts b/tests/unit/breaker-network-error-guard.test.ts index 8d53edc080..35e1773b0a 100644 --- a/tests/unit/breaker-network-error-guard.test.ts +++ b/tests/unit/breaker-network-error-guard.test.ts @@ -78,7 +78,7 @@ test("forceLiveComboTest=true prevents breaker trip (combo will try next target) // `{ success: false, status: 5xx }` as a success. test("classifyProviderBreakerResult: a resolved 503 on the single-model path is a failure", () => { const outcome = classifyProviderBreakerResult( - { success: false, status: 503, errorCode: null, errorType: null, error: "overloaded" }, + { success: false, status: 503, errorCode: null, errorType: null, error: "service unavailable" }, false, false ); diff --git a/tests/unit/combo-cooldown-retry.test.ts b/tests/unit/combo-cooldown-retry.test.ts index 91d0339b1b..cc2c76699a 100644 --- a/tests/unit/combo-cooldown-retry.test.ts +++ b/tests/unit/combo-cooldown-retry.test.ts @@ -22,8 +22,12 @@ import test from "node:test"; import assert from "node:assert/strict"; -const { shouldWaitForComboCooldown, resolveComboCooldownWaitDecision, COMBO_COOLDOWN_WAIT_MARGIN_MS } = - await import("../../open-sse/services/combo/comboCooldownRetry.ts"); +const { + shouldWaitForComboCooldown, + resolveComboCooldownWaitDecision, + resolveCircuitOpenWaitDecision, + COMBO_COOLDOWN_WAIT_MARGIN_MS, +} = await import("../../open-sse/services/combo/comboCooldownRetry.ts"); function baseSettings(overrides: Partial> = {}) { return { @@ -77,6 +81,11 @@ test("missing/unknown reason (null) → no wait (only an explicit transient reas assert.equal(r.wait, false); }); +test("reason circuit_open is eligible (whole-provider breaker OPEN is a short reset)", () => { + const r = shouldWaitForComboCooldown(baseInput({ reason: "circuit_open" }) as never); + assert.equal(r.wait, true); +}); + test("waitMs above the configured ceiling → no wait", () => { const r = shouldWaitForComboCooldown( baseInput({ waitMs: 5001, settings: baseSettings({ maxWaitMs: 5000 }) }) as never @@ -146,6 +155,7 @@ test("returned waitMs is clamped to a finite number (0 when input invalid)", () // ── resolveComboCooldownWaitDecision (target resolution + hint/fallback) ────── const M = COMBO_COOLDOWN_WAIT_MARGIN_MS; +assert.equal(M, 50, "wait margin is a pinned production constant, not a free parameter"); function decisionInput(overrides: Record = {}) { return { @@ -269,3 +279,75 @@ test("resolve: lock remaining above the ceiling → no wait (not a SHORT cooldow ); assert.equal(r.wait, false); }); + +// Live incident 2026-09-03: a single-target combo pre-skipped every target because +// the whole-provider breaker was OPEN (claude Overloaded STREAM_EARLY_EOF). That +// path crystallized ALL_TARGETS_SKIPPED in ~43ms and never entered the cooldown +// wait, even though the breaker resetTimeout is 60s and comboCooldownWait was on. + +function circuitOpenInput(overrides: Record = {}) { + return { + skippedForCircuitOpen: true, + retryAfterMs: 30_000, + attempt: 0, + budgetLeftMs: 90_000, + settings: baseSettings({ maxWaitMs: 90_000, budgetMs: 300_000, maxAttempts: 5 }), + ...overrides, + }; +} + +test("circuit-open skip with a short breaker reset → wait", () => { + const r = resolveCircuitOpenWaitDecision(circuitOpenInput() as never); + assert.equal(r.wait, true); + assert.equal(r.waitMs, 30_000 + M); + assert.equal(r.reason, "circuit_open"); +}); + +test("circuit-open skip is ignored when no target was skipped for circuit_open", () => { + const r = resolveCircuitOpenWaitDecision( + circuitOpenInput({ skippedForCircuitOpen: false }) as never + ); + assert.equal(r.wait, false); + assert.equal(r.reason, null); +}); + +test("circuit-open skip with zero retryAfter → no wait", () => { + const r = resolveCircuitOpenWaitDecision(circuitOpenInput({ retryAfterMs: 0 }) as never); + assert.equal(r.wait, false); +}); + +test("circuit-open skip above the wait ceiling → no wait", () => { + const r = resolveCircuitOpenWaitDecision( + circuitOpenInput({ + retryAfterMs: 120_000, + settings: baseSettings({ maxWaitMs: 90_000, budgetMs: 300_000, maxAttempts: 5 }), + }) as never + ); + assert.equal(r.wait, false); +}); + +test("circuit-open skip honors attempt/budget the same as model-lockout waits", () => { + assert.equal( + resolveCircuitOpenWaitDecision(circuitOpenInput({ attempt: 5 }) as never).wait, + false + ); + assert.equal( + resolveCircuitOpenWaitDecision(circuitOpenInput({ budgetLeftMs: 1_000 }) as never).wait, + false + ); +}); + +test("circuit-open skip is off when comboCooldownWait.enabled is false", () => { + const r = resolveCircuitOpenWaitDecision( + circuitOpenInput({ + settings: baseSettings({ + enabled: false, + maxWaitMs: 90_000, + budgetMs: 300_000, + maxAttempts: 5, + }), + }) as never + ); + assert.equal(r.wait, false); + assert.equal(r.reason, null); +}); diff --git a/tests/unit/native-codex-turn-pin-model-scoped-fallback.test.ts b/tests/unit/native-codex-turn-pin-model-scoped-fallback.test.ts index 595161d64e..d37fb55cb8 100644 --- a/tests/unit/native-codex-turn-pin-model-scoped-fallback.test.ts +++ b/tests/unit/native-codex-turn-pin-model-scoped-fallback.test.ts @@ -29,6 +29,7 @@ const providersDb = await import("../../src/lib/db/providers.ts"); const testSettings = { resilienceSettings: { providerCooldown: { enabled: true, minRetryCooldownMs: 5000, maxRetryCooldownMs: 300000 }, + comboCooldownWait: { enabled: false }, }, }; diff --git a/tests/unit/overloaded-not-provider-breaker.test.ts b/tests/unit/overloaded-not-provider-breaker.test.ts new file mode 100644 index 0000000000..0ba022cb94 --- /dev/null +++ b/tests/unit/overloaded-not-provider-breaker.test.ts @@ -0,0 +1,169 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { isModelCapacityOverloadError } from "../../src/shared/utils/circuitBreaker.ts"; +import { + shouldTripProviderBreakerForResult, + classifyProviderBreakerResult, +} from "../../src/sse/handlers/chatPredicates.ts"; +import { shouldRecordProviderBreakerFailure } from "../../open-sse/services/combo/comboPredicates.ts"; + +/** + * Live incident 2026-09-03 (X500 offical-fable): Anthropic returned + * STREAM_EARLY_EOF wrapping "Overloaded" as HTTP 502. That 502 opened the + * whole-provider `claude` breaker. The single-target combo then pre-skipped + * with ALL_TARGETS_SKIPPED in ~43ms even though the account, pin, and quota + * were healthy. Model capacity (529 / Overloaded) is not a provider outage. + */ + +const OTHER_COMBO_ARGS = { + isStreamReadinessFailure: false, + sameProviderNext: false, + skipProviderBreaker: false, + requestScopedFailure: false, +} as const; + +const LIVE_EOF_OVERLOADED = "Stream ended before producing a non-ping SSE event: Overloaded"; +const PLAIN_EOF = "Stream ended before producing a non-ping SSE event"; + +test("isModelCapacityOverloadError: live STREAM_EARLY_EOF Overloaded text", () => { + assert.equal(isModelCapacityOverloadError(LIVE_EOF_OVERLOADED), true); +}); + +test("isModelCapacityOverloadError: bare Overloaded / HTTP 529", () => { + assert.equal(isModelCapacityOverloadError("Overloaded"), true); + assert.equal(isModelCapacityOverloadError("[529]: Overloaded"), true); + assert.equal(isModelCapacityOverloadError(529), true); + assert.equal(isModelCapacityOverloadError({ message: "overloaded_error" }), true); +}); + +test("isModelCapacityOverloadError: a plain early EOF is NOT capacity", () => { + assert.equal(isModelCapacityOverloadError(PLAIN_EOF), false); + assert.equal(isModelCapacityOverloadError("502 Bad Gateway"), false); + assert.equal(isModelCapacityOverloadError(null), false); + assert.equal(isModelCapacityOverloadError(undefined), false); +}); + +test("combo: STREAM_EARLY_EOF Overloaded 502 does NOT record a whole-provider breaker failure", () => { + assert.equal( + shouldRecordProviderBreakerFailure({ + ...OTHER_COMBO_ARGS, + isStreamReadinessFailure: true, + isStreamEarlyEof: true, + status: 502, + error: LIVE_EOF_OVERLOADED, + }), + false + ); +}); + +test("combo: a plain STREAM_EARLY_EOF 502 still records a breaker failure", () => { + assert.equal( + shouldRecordProviderBreakerFailure({ + ...OTHER_COMBO_ARGS, + isStreamReadinessFailure: true, + isStreamEarlyEof: true, + status: 502, + error: PLAIN_EOF, + }), + true + ); +}); + +test("combo: HTTP 529 Overloaded does not record even if someone later adds 529 to the status set", () => { + assert.equal( + shouldRecordProviderBreakerFailure({ + ...OTHER_COMBO_ARGS, + status: 529, + error: "[529]: Overloaded", + }), + false + ); +}); + +test("combo: HTTP 529 status alone does not record a breaker failure", () => { + assert.equal( + shouldRecordProviderBreakerFailure({ + ...OTHER_COMBO_ARGS, + status: 529, + error: "upstream error", + }), + false + ); +}); + +test("combo: a genuine 502 without Overloaded still records", () => { + assert.equal( + shouldRecordProviderBreakerFailure({ + ...OTHER_COMBO_ARGS, + status: 502, + error: "upstream error", + }), + true + ); +}); + +test("single-model: STREAM_EARLY_EOF Overloaded 502 does NOT trip the provider breaker", () => { + assert.equal( + shouldTripProviderBreakerForResult( + { + status: 502, + errorCode: "STREAM_EARLY_EOF", + errorType: "stream_early_eof", + error: LIVE_EOF_OVERLOADED, + }, + false, + false + ), + false + ); +}); + +test("single-model: a genuine 502 without Overloaded still trips", () => { + assert.equal( + shouldTripProviderBreakerForResult( + { status: 502, errorCode: null, errorType: null, error: "upstream error" }, + false, + false + ), + true + ); +}); + +test("single-model: HTTP 529 does not trip", () => { + assert.equal( + shouldTripProviderBreakerForResult( + { status: 529, errorCode: null, errorType: null, error: "Overloaded" }, + false, + false + ), + false + ); +}); + +test("classifyProviderBreakerResult: Overloaded 502 on the single-model path is ignore", () => { + assert.equal( + classifyProviderBreakerResult( + { + success: false, + status: 502, + errorCode: "STREAM_EARLY_EOF", + errorType: "stream_early_eof", + error: LIVE_EOF_OVERLOADED, + }, + false, + false + ), + "ignore" + ); +}); + +test("classifyProviderBreakerResult: Overloaded 529 on the single-model path is ignore", () => { + assert.equal( + classifyProviderBreakerResult( + { success: false, status: 529, errorCode: null, errorType: null, error: "Overloaded" }, + false, + false + ), + "ignore" + ); +}); diff --git a/tests/unit/repro-9630-combo-false-503.test.ts b/tests/unit/repro-9630-combo-false-503.test.ts index 16eae79b05..b3fbbc5bdc 100644 --- a/tests/unit/repro-9630-combo-false-503.test.ts +++ b/tests/unit/repro-9630-combo-false-503.test.ts @@ -70,7 +70,11 @@ test("#9630: combo returns truthful error, not false ALL_ACCOUNTS_INACTIVE, when }, isModelAvailable: async () => true, log: { info: () => {}, warn: () => {}, debug: () => {}, error: () => {} }, - settings: null, + settings: { + resilienceSettings: { + comboCooldownWait: { enabled: false }, + }, + }, relayOptions: null, allCombos: null, }); From 2a6eff0aec305a54205bda16a6a46803153bbd5a Mon Sep 17 00:00:00 2001 From: "Bob.Hou" Date: Thu, 3 Sep 2026 22:47:28 -0400 Subject: [PATCH 095/143] fix(combo): keep Antigravity Gemini usable when Claude weekly is empty (#12637) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 4 PRs desta leva sobre o tip de `release/v3.8.51`: `typecheck:core` limpo e **119/119** nos 9 arquivos de teste que trazem. Três dos quatro conflitavam apenas no `config/quality/file-size-baseline.json`, todos de forma aditiva (chaves `_rebaseline_` distintas que devem coexistir); resolvidos com validação de JSON a cada passo. Registro que o **#12637 não é duplicata do #12566**, apesar do título quase idêntico: o autor documenta que aquele escopou o cooldown de preflight por família e este cobre o `genericQuotaFetcher`, que é o que o roteamento reset-aware efetivamente chama. Traz também validação ao vivo em VPS (imagem X500, `onmi-gemini3.6` → HTTP 200), satisfazendo a Hard Rule #18. Obrigado, @HouMinXi. --- changelog.d/fixes/reset-aware-model-family.md | 1 + config/quality/file-size-baseline.json | 2 + open-sse/services/antigravityQuotaFamily.ts | 8 + open-sse/services/combo.ts | 11 +- .../services/combo/quotaExhaustionCutoff.ts | 2 +- open-sse/services/combo/quotaStrategies.ts | 13 +- open-sse/services/genericQuotaFetcher.ts | 172 ++++++++++--- tests/unit/combo-strategies.test.ts | 10 + .../reset-aware-request-scope-12600.test.ts | 234 ++++++++++++++++++ 9 files changed, 406 insertions(+), 47 deletions(-) create mode 100644 changelog.d/fixes/reset-aware-model-family.md create mode 100644 tests/unit/reset-aware-request-scope-12600.test.ts diff --git a/changelog.d/fixes/reset-aware-model-family.md b/changelog.d/fixes/reset-aware-model-family.md new file mode 100644 index 0000000000..75b65e7613 --- /dev/null +++ b/changelog.d/fixes/reset-aware-model-family.md @@ -0,0 +1 @@ +Keep Antigravity Gemini usable when the same connection's Claude weekly quota is empty; generic quota cache stays per-connection for every other provider. diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index a7941e072e..d27b98aae0 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -1,4 +1,5 @@ { + "_rebaseline_2026_09_03_reset_aware_model_family": "Own growth: open-sse/services/combo.ts 4036->4041 (+5). buildAutoCandidates now keys the reset-aware quota cache by getQuotaFetchScope and spreads requestedModel onto the connection so Gemini windows stay off a Claude-empty Antigravity account. Irreducible wiring at the existing fetchResetAwareQuotaWithCache call site; the family helper itself lives in antigravityQuotaFamily.ts. Covered by tests/unit/reset-aware-request-scope-12600.test.ts.", "_rebaseline_2026_09_03_overloaded_not_provider_breaker": "fix/overloaded-not-provider-breaker own growth: open-sse/services/combo.ts 4036->4075 (check-file-size split-newline, +39). Circuit-open pre-skip now records the breaker retryAfter and, when every target was skipped that way, waits the short reset via resolveCircuitOpenWaitDecision (new leaf in comboCooldownRetry.ts) instead of crystallizing ALL_TARGETS_SKIPPED in ~43ms. skippedForCircuitOpen / earliestCircuitOpenRetryMs reset each setTry so a later iteration cannot inherit a stale retryAfter. Irreducible at the existing ALL_TARGETS_SKIPPED chokepoint (same pattern as #7301/#8213 cooldown-wait). Predicate itself lives in circuitBreaker.ts / comboPredicates.ts / chatPredicates.ts, all under cap. Covered by tests/unit/overloaded-not-provider-breaker.test.ts + combo-cooldown-retry.test.ts.", "_rebaseline_2026_09_03_12649_free_tier_reaudit_gateways": "PR #12649 (fix/free-tier-quota-reaudit) own growth: src/shared/constants/providers/apikey/gateways.ts 1459->1462 (+3 = the nara authHint rewritten for the re-audited 7M/day plan now wraps to two lines, plus the Prettier reflow of two pre-existing >100-col authHint lines (oneminai, freebuff) that lint-staged enforces on any touch of the file; additive text at the existing registry chokepoint, same god-file no-split rationale as prior gateways.ts rebaselines: #11786 seekai, #10987 logfare, #10531 freebuff). Covered by tests/unit/free-tier-reaudit-2026-09.test.ts and tests/unit/free-providers-batch-2026-07.test.ts.", "_rebaseline_2026_09_03_moonshot_native_quota": "PR feat/moonshot-native-quota own growth on release/v3.8.51: src/lib/db/migrationRunner.ts 1201->1206 (+5, case 172 retroactive guard for daily_quota_reset_* columns); src/sse/handlers/chat.ts 2434->2450 (+16, registerMoonshotQuotaFetcher + startup node scan at the existing quota-fetcher registration chokepoint); src/sse/services/auth.ts 3427->3450 (+23, resolveDailyResetForProvider + dailyReset arg on checkFallbackError); open-sse/services/accountFallback.ts 2422->2461 (+39, compatible-node credits_exhausted carve-out + TPD node-clock lock); tests/unit/account-fallback-service.test.ts 2008->2056 (+48, TPD/empty-wallet cases). Wiring at existing chokepoints; Moonshot host predicates, daily reset clock, and the balance fetcher live in new leaves under cap. Covered by tests/unit/moonshot-*.test.ts + account-fallback-service.test.ts (135/135 focused).", @@ -424,6 +425,7 @@ "open-sse/mcp-server/server.ts": 1572, "open-sse/services/accountFallback.ts": 2467, "open-sse/services/adobeFireflyBrowserLogin.ts": 1401, + "open-sse/services/combo.ts": 4041, "open-sse/services/combo.ts": 4075, "open-sse/translator/response/openai-responses.ts": 1466, "open-sse/utils/cursorAgentProtobuf.ts": 1547, diff --git a/open-sse/services/antigravityQuotaFamily.ts b/open-sse/services/antigravityQuotaFamily.ts index 94016c1109..07e0012b2b 100644 --- a/open-sse/services/antigravityQuotaFamily.ts +++ b/open-sse/services/antigravityQuotaFamily.ts @@ -55,6 +55,14 @@ export function getQuotaScopeLabelForProvider( return getAntigravityQuotaFamily(model) === "other" ? "model" : "family"; } +export function getQuotaFetchScope( + provider: string | null | undefined, + model: string | null | undefined +): string { + if (provider !== "antigravity" && provider !== "agy") return "*"; + return getQuotaScopedModelForProvider(provider, model) ?? "*"; +} + export function isAntigravityQuotaProvider(provider: string | null | undefined): boolean { return provider === "antigravity" || provider === "agy"; } diff --git a/open-sse/services/combo.ts b/open-sse/services/combo.ts index 635001b646..870a470749 100644 --- a/open-sse/services/combo.ts +++ b/open-sse/services/combo.ts @@ -69,6 +69,7 @@ import { resolveModelLockoutSettings } from "../../src/lib/resilience/modelLocko import { fetchCodexQuota } from "./codexQuotaFetcher.ts"; import { evaluateQuotaCutoff, getQuotaFetcher, type QuotaInfo } from "./quotaPreflight.ts"; import { resolveProviderId } from "../../src/shared/constants/providers.ts"; +import { getQuotaFetchScope } from "./antigravityQuotaFamily.ts"; import * as semaphore from "./rateLimitSemaphore.ts"; import { getCircuitBreaker } from "../../src/shared/utils/circuitBreaker"; import { parseModel } from "./model.ts"; @@ -596,14 +597,17 @@ export async function buildAutoCandidates( statusPenaltyReason = connectionStatusReason; } if (fetcher && target.connectionId) { - const quotaKey = `${provider}:${target.connectionId}`; + const quotaScope = getQuotaFetchScope(provider, target.modelStr); + const quotaKey = `${provider}:${target.connectionId}:${quotaScope}`; if (!quotaPromises.has(quotaKey)) { quotaPromises.set( quotaKey, fetchResetAwareQuotaWithCache({ provider, connectionId: target.connectionId, - connection, + connection: connection + ? { ...connection, requestedModel: target.modelStr } + : connection, fetcher, config: resetWindowConfig, log: {}, @@ -1393,7 +1397,8 @@ async function handleComboChatInner({ resilienceSettings, quotaCutoffResetWindowConfig, combo.name, - log, modelStr + log, + modelStr ); if (quotaCutoff.blocked) { log.info( diff --git a/open-sse/services/combo/quotaExhaustionCutoff.ts b/open-sse/services/combo/quotaExhaustionCutoff.ts index a73d3628f4..ede1424105 100644 --- a/open-sse/services/combo/quotaExhaustionCutoff.ts +++ b/open-sse/services/combo/quotaExhaustionCutoff.ts @@ -119,7 +119,7 @@ export async function resolveQuotaExhaustionCutoffForTarget( const quota = await fetchResetAwareQuotaWithCache({ provider, connectionId, - connection, + connection: connection ? { ...connection, requestedModel } : connection, fetcher, config: resetWindowConfig, log, diff --git a/open-sse/services/combo/quotaStrategies.ts b/open-sse/services/combo/quotaStrategies.ts index c4cb4c5b52..88401c0f43 100644 --- a/open-sse/services/combo/quotaStrategies.ts +++ b/open-sse/services/combo/quotaStrategies.ts @@ -46,6 +46,7 @@ import { } from "./quotaScoring.ts"; import { rankByHeadroom, type HeadroomSaturation } from "./headroomRanking.ts"; import { preferAntigravityConnectionsWithStoredProject } from "../antigravityProjectPersist.ts"; +import { getQuotaFetchScope } from "../antigravityQuotaFamily.ts"; import { isQuotaExhaustedForRequest } from "../../../src/domain/quotaCache.ts"; const RESET_AWARE_CONNECTION_CACHE_TTL_MS = 30_000; @@ -269,14 +270,17 @@ async function scoreQuotaAwareTargets({ const provider = getResetAwareProvider(target); const fetcher = provider ? getQuotaFetcher(provider) : null; if (fetcher && provider && target.connectionId) { - const quotaKey = `${provider}:${target.connectionId}`; + const quotaKey = `${provider}:${target.connectionId}:${getQuotaFetchScope(provider, target.modelStr)}`; if (!quotaPromises.has(quotaKey)) { + const connection = connectionById.get(target.connectionId); quotaPromises.set( quotaKey, fetchResetAwareQuotaWithCache({ provider, connectionId: target.connectionId, - connection: connectionById.get(target.connectionId), + connection: connection + ? { ...connection, requestedModel: target.modelStr } + : connection, fetcher, config, log, @@ -354,7 +358,10 @@ export async function fetchResetAwareQuotaWithCache({ log: { debug?: (...args: unknown[]) => void; warn?: (...args: unknown[]) => void }; comboName: string; }): Promise { - const cacheKey = `${provider}:${connectionId}`; + const requestedModel = + typeof connection?.requestedModel === "string" ? connection.requestedModel : null; + const cacheScope = getQuotaFetchScope(provider, requestedModel); + const cacheKey = `${provider}:${connectionId}:${cacheScope}`; const ttlMs = config.quotaCacheTtlMs; const maxStaleMs = config.quotaCacheMaxStaleMs; const now = Date.now(); diff --git a/open-sse/services/genericQuotaFetcher.ts b/open-sse/services/genericQuotaFetcher.ts index 81419d5a6b..ac89aae30a 100644 --- a/open-sse/services/genericQuotaFetcher.ts +++ b/open-sse/services/genericQuotaFetcher.ts @@ -24,6 +24,10 @@ import { type QuotaFetcher, type QuotaInfo, } from "./quotaPreflight.ts"; +import { + getAntigravityQuotaFamily, + getQuotaFetchScope, +} from "./antigravityQuotaFamily.ts"; type UsageFetcher = ( connection: Parameters[0], @@ -54,7 +58,7 @@ export function __agePendingForceRefreshForTests( connectionId: string, ageMs: number ): void { - pendingForceRefresh.set(cacheKey(provider, connectionId), Date.now() - ageMs); + pendingForceRefresh.set(connectionKey(provider, connectionId), Date.now() - ageMs); } /** Test-only: backdate a convert-null miss so the 60s hammer-guard is unit-testable. */ @@ -63,7 +67,7 @@ export function __agePendingForceRefreshMissForTests( connectionId: string, ageMs: number ): void { - pendingForceRefreshMiss.set(cacheKey(provider, connectionId), Date.now() - ageMs); + pendingForceRefreshMiss.set(connectionKey(provider, connectionId), Date.now() - ageMs); } /** Test-only: drop all wrapper/flag maps so tests cannot leak across ids. */ @@ -80,10 +84,25 @@ interface CacheEntry { const cache = new Map(); -function cacheKey(provider: string, connectionId: string): string { +function connectionKey(provider: string, connectionId: string): string { return `${provider.trim()}::${connectionId.trim()}`; } +function quotaCacheScope( + provider: string, + requestedModel?: string | null +): string { + return getQuotaFetchScope(provider, requestedModel); +} + +function cacheKey( + provider: string, + connectionId: string, + requestedModel?: string | null +): string { + return `${connectionKey(provider, connectionId)}::${quotaCacheScope(provider, requestedModel)}`; +} + function dropExpiredPendingForceRefresh(key: string, now: number): boolean { const stampedAt = pendingForceRefresh.get(key); if (stampedAt === undefined) return true; @@ -216,7 +235,15 @@ interface ConnectionInputs { * / shape-unknown / missing). Exported for unit testing — the production path * is `fetchGenericQuota`, which adds caching + the upstream call. */ -export function convertUsageToQuotaInfo(usage: unknown): QuotaInfo | null { +type UsageToQuotaContext = { + requestedModel?: string | null; + provider?: string | null; +}; + +export function convertUsageToQuotaInfo( + usage: unknown, + context: UsageToQuotaContext = {} +): QuotaInfo | null { if (!usage || typeof usage !== "object") return null; const usageRecord = usage as Record; if ( @@ -235,31 +262,51 @@ export function convertUsageToQuotaInfo(usage: unknown): QuotaInfo | null { } const windows: Record = {}; - let worstPercent = 0; - let worstResetAt: string | null = null; for (const [name, entry] of Object.entries(quotasObj as Record)) { const percentUsed = percentUsedForQuota(entry); if (percentUsed === null) continue; - const resetAt = resetAtForQuota(entry); - windows[name] = { percentUsed, resetAt }; - if (percentUsed > worstPercent) { - worstPercent = percentUsed; - worstResetAt = resetAt; - } + windows[name] = { percentUsed, resetAt: resetAtForQuota(entry) }; } if (Object.keys(windows).length === 0) return null; - const normalized = normalizeQuotaWindows(windows); + const requestedFamily = + isAntigravityProvider(context.provider) && context.requestedModel + ? getAntigravityQuotaFamily(context.requestedModel) + : null; + const providerScopedWindows = + requestedFamily === "gemini" || requestedFamily === "claude" + ? Object.fromEntries( + Object.entries(windows).filter(([key]) => { + if (key.endsWith("_weekly")) { + return antigravityWeeklyWindowMatchesFamily(key, requestedFamily); + } + return getAntigravityQuotaFamily(key) === requestedFamily; + }) + ) + : windows; + if (Object.keys(providerScopedWindows).length === 0) return null; + + const normalized = normalizeQuotaWindows(providerScopedWindows, context); + const scopedEntries = Object.values(providerScopedWindows); + const percentUsed = scopedEntries.reduce( + (worst, entry) => Math.max(worst, entry.percentUsed), + 0 + ); + const resetAt = + scopedEntries.reduce<{ percentUsed: number; resetAt: string | null } | null>( + (worst, entry) => (!worst || entry.percentUsed > worst.percentUsed ? entry : worst), + null + )?.resetAt ?? null; return { used: 0, total: 0, - percentUsed: worstPercent, - resetAt: worstResetAt, - windows, + percentUsed, + resetAt, + windows: providerScopedWindows, ...normalized, - limitReached: worstPercent >= 1 - 1e-9, + limitReached: percentUsed >= 1 - 1e-9, }; } @@ -269,12 +316,29 @@ export function convertUsageToQuotaInfo(usage: unknown): QuotaInfo | null { * naming convention. * * - Claude: "session (5h)" → window5h, "weekly (7d)" → window7d - * - Antigravity: worst per-model quota → window5h; worst *_weekly quota → window7d + * - Antigravity: requested-family model quota → window5h; matching family weekly quota → window7d */ +function isAntigravityProvider(provider: string | null | undefined): boolean { + return provider === "antigravity" || provider === "agy"; +} + +function antigravityWeeklyWindowMatchesFamily( + key: string, + family: "gemini" | "claude" +): boolean { + if (!key.endsWith("_weekly")) return false; + return family === "gemini" ? key === "gemini_weekly" : key === "claude_gpt_weekly"; +} + function normalizeQuotaWindows( - windows: Record + windows: Record, + context: UsageToQuotaContext ): Record { const normalized: Record = {}; + const requestedFamily = + isAntigravityProvider(context.provider) && context.requestedModel + ? getAntigravityQuotaFamily(context.requestedModel) + : null; // Claude-style explicit time windows. if (windows["session (5h)"] && !normalized.window5h) { @@ -284,22 +348,31 @@ function normalizeQuotaWindows( normalized.window7d = windows["weekly (7d)"]; } - // Antigravity-style per-model 5h windows: pick the worst (most used) model quota. + // Antigravity-style per-model windows: pick worst only inside requested family. const modelWindows = Object.entries(windows).filter( ([key]) => key !== "credits" && !key.endsWith("_weekly") && !key.startsWith("window") && !key.includes("(5h)") && - !key.includes("(7d)") + !key.includes("(7d)") && + (requestedFamily === null || + requestedFamily === "other" || + getAntigravityQuotaFamily(key) === requestedFamily) ); if (modelWindows.length > 0 && !normalized.window5h) { const worst = modelWindows.reduce((a, b) => (a[1].percentUsed > b[1].percentUsed ? a : b)); normalized.window5h = worst[1]; } - // Antigravity-style weekly family buckets: pick the worst *_weekly quota. - const weeklyWindows = Object.entries(windows).filter(([key]) => key.endsWith("_weekly")); + // Antigravity-style weekly buckets: pick worst only inside requested family. + const weeklyWindows = Object.entries(windows).filter(([key]) => { + const hasFamilyScope = requestedFamily === "gemini" || requestedFamily === "claude"; + return ( + key.endsWith("_weekly") && + (!hasFamilyScope || antigravityWeeklyWindowMatchesFamily(key, requestedFamily)) + ); + }); if (weeklyWindows.length > 0 && !normalized.window7d) { const worst = weeklyWindows.reduce((a, b) => (a[1].percentUsed > b[1].percentUsed ? a : b)); normalized.window7d = worst[1]; @@ -320,18 +393,21 @@ export const fetchGenericQuota: QuotaFetcher = async (connectionId, connection) const provider = typeof conn.provider === "string" ? conn.provider.trim() : ""; if (!provider) return null; - const key = cacheKey(provider, connectionId); + const requestedModel = + typeof connection.requestedModel === "string" ? connection.requestedModel : undefined; + const key = cacheKey(provider, connectionId, requestedModel); + const forceKey = connectionKey(provider, connectionId); const now = Date.now(); - const forceRefresh = isPendingForceRefresh(key, now); + const forceRefresh = isPendingForceRefresh(forceKey, now); const hit = cachedQuotaIfFresh(key, forceRefresh, now); if (hit) return hit; // convert-null / throw keep the force-refresh flag (agy inner caches are // still stale) but must not hammer those endpoints on every routing tick. - if (isForceRefreshMissCooling(key, forceRefresh, now)) return null; + if (isForceRefreshMissCooling(forceKey, forceRefresh, now)) return null; // Capture before await: a 429 during fetchUsage re-stamps this; writing // the pre-429 snapshot would wipe that flag and recache stale quota. - const refreshStamp = pendingForceRefresh.get(key); + const refreshStamp = pendingForceRefresh.get(forceKey); let usage: unknown; try { @@ -340,28 +416,29 @@ export const fetchGenericQuota: QuotaFetcher = async (connectionId, connection) ...(forceRefresh ? { forceRefresh: true } : {}), }); } catch { - markPendingForceRefreshMiss(key); + markPendingForceRefreshMiss(forceKey); return null; } - const quota = convertUsageToQuotaInfo(usage); + const quota = convertUsageToQuotaInfo(usage, { provider, requestedModel }); if (!quota) { - markPendingForceRefreshMiss(key); + markPendingForceRefreshMiss(forceKey); return null; } // Concurrent 429 re-stamped a still-live flag — do not recache the // pre-429 snapshot. A vanished or expired stamp is not a 429. - if (isConcurrentForceRefresh(key, refreshStamp)) { + if (isConcurrentForceRefresh(forceKey, refreshStamp)) { return quota; } - pendingForceRefresh.delete(key); - pendingForceRefreshMiss.delete(key); + pendingForceRefresh.delete(forceKey); + pendingForceRefreshMiss.delete(forceKey); - // Refresh the static window catalog so the dashboard can render the right - // modal inputs without waiting for the user to open the page. - registerQuotaWindows(provider, Object.keys(quota.windows || {})); + // Refresh the static window catalog from the unscoped usage payload so a + // family-scoped request cannot hide sibling-family dashboard controls. + const unscopedQuota = convertUsageToQuotaInfo(usage, { provider }); + registerQuotaWindows(provider, Object.keys(unscopedQuota?.windows || quota.windows || {})); cache.set(key, { quota, fetchedAt: Date.now() }); return quota; @@ -373,13 +450,16 @@ export const fetchGenericQuota: QuotaFetcher = async (connectionId, connection) * fresh data instead of a 60s stale window. */ export function invalidateGenericQuotaCache(provider: string, connectionId: string): void { - const key = cacheKey(provider, connectionId); - cache.delete(key); + const forceKey = connectionKey(provider, connectionId); + const prefix = `${forceKey}::`; + for (const key of cache.keys()) { + if (key.startsWith(prefix)) cache.delete(key); + } // Next fetch must bypass provider-inner usage caches (agy retrieveUserQuota / // weekly are 60s–5min). Without this, dropping the 60s wrapper recaches stale. // TTL matches those inner caches: after 5min the flag is a no-op. - pendingForceRefresh.set(key, Date.now()); - pendingForceRefreshMiss.delete(key); + pendingForceRefresh.set(forceKey, Date.now()); + pendingForceRefreshMiss.delete(forceKey); } /** @@ -415,3 +495,15 @@ export function registerGenericQuotaFetchers(): void { registerQuotaFetcher(provider, fetchGenericQuota); } } + +export const __testing = { + setUsageFetcher(fetcher: UsageFetcher): void { + usageFetcherOverride = fetcher; + }, + resetUsageFetcher(): void { + usageFetcherOverride = null; + }, + clearCache(): void { + cache.clear(); + }, +}; diff --git a/tests/unit/combo-strategies.test.ts b/tests/unit/combo-strategies.test.ts index dd0c7be367..990a5b9fe4 100644 --- a/tests/unit/combo-strategies.test.ts +++ b/tests/unit/combo-strategies.test.ts @@ -17,6 +17,8 @@ const { clearAllStickyBindings } = const { invalidateCodexQuotaCache, registerCodexConnection, registerCodexQuotaFetcher } = await import("../../open-sse/services/codexQuotaFetcher.ts"); const { registerQuotaFetcher } = await import("../../open-sse/services/quotaPreflight.ts"); +const { getQuotaScopedModelForProvider } = + await import("../../open-sse/services/antigravityQuotaFamily.ts"); const combosDb = await import("../../src/lib/db/combos.ts"); const providersDb = await import("../../src/lib/db/providers.ts"); const { recordComboRequest } = await import("../../open-sse/services/comboMetrics.ts"); @@ -434,6 +436,14 @@ test("reset-aware strategy avoids accounts near 5h exhaustion", async (t) => { assert.equal(await selectedConnectionFor(combo), healthy5h.id); }); +test("Antigravity aliases share one family-scoped cache key", () => { + assert.equal(getQuotaScopedModelForProvider("agy", "gemini-3.7-flash-high"), "family:gemini"); + assert.equal( + getQuotaScopedModelForProvider("antigravity", "gemini-3.7-flash-high"), + "family:gemini" + ); +}); + test("reset-aware strategy rotates similar scores with round-robin tie breaking", async () => { const provider = `tie-provider-${randomUUID()}`; const first = `first-${randomUUID()}`; diff --git a/tests/unit/reset-aware-request-scope-12600.test.ts b/tests/unit/reset-aware-request-scope-12600.test.ts new file mode 100644 index 0000000000..d035175c2b --- /dev/null +++ b/tests/unit/reset-aware-request-scope-12600.test.ts @@ -0,0 +1,234 @@ +import test, { afterEach } from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; + +const genericModule = await import("../../open-sse/services/genericQuotaFetcher.ts"); +const scoringModule = await import("../../open-sse/services/combo/quotaScoring.ts"); +const familyModule = await import("../../open-sse/services/antigravityQuotaFamily.ts"); +const preflightModule = await import("../../open-sse/services/quotaPreflight.ts"); + +const { convertUsageToQuotaInfo, fetchGenericQuota, invalidateGenericQuotaCache } = genericModule; +const { scoreResetAwareQuota, resolveResetAwareConfig } = scoringModule; +const { getQuotaFetchScope } = familyModule; +const { getQuotaWindows } = preflightModule; + +const resetAt5h = new Date(Date.now() + 4 * 60 * 60 * 1000).toISOString(); +const resetAt7d = new Date(Date.now() + 6 * 24 * 60 * 60 * 1000).toISOString(); + +const usage = { + quotas: { + "gemini-3.7-flash-high": { + used: 30, + total: 1000, + remainingPercentage: 97, + resetAt: resetAt5h, + }, + "claude-opus-4-6-thinking": { + used: 1000, + total: 1000, + remainingPercentage: 0, + resetAt: resetAt7d, + }, + "gpt-oss-120b-medium": { + used: 900, + total: 1000, + remainingPercentage: 10, + resetAt: resetAt5h, + }, + gemini_weekly: { + used: 10, + total: 1000, + remainingPercentage: 99, + resetAt: resetAt7d, + }, + claude_gpt_weekly: { + used: 1000, + total: 1000, + remainingPercentage: 0, + resetAt: resetAt7d, + }, + unrelated_weekly: { + used: 1000, + total: 1000, + remainingPercentage: 0, + resetAt: resetAt7d, + }, + }, +}; + +test("reset-aware Gemini scoring ignores depleted Claude family quota", () => { + const quota = convertUsageToQuotaInfo(usage, { + provider: "agy", + requestedModel: "agy/gemini-3.7-flash-high", + }); + + assert.ok(quota); + assert.equal(quota.window5h?.percentUsed, 0.03); + assert.equal(quota.window7d?.percentUsed, 0.01); + assert.equal(quota.percentUsed, 0.03); + assert.equal(quota.limitReached, false); + assert.equal(quota.windows?.["claude-opus-4-6-thinking"], undefined); + assert.equal(quota.windows?.["gpt-oss-120b-medium"], undefined); + assert.equal(quota.windows?.claude_gpt_weekly, undefined); + assert.equal(quota.windows?.unrelated_weekly, undefined); + assert.ok(scoreResetAwareQuota(quota, resolveResetAwareConfig({})).score > 0.3); +}); + +test("opposite-family-only telemetry fails open as unknown", () => { + const gemini = convertUsageToQuotaInfo( + { quotas: { claude_gpt_weekly: usage.quotas.claude_gpt_weekly } }, + { provider: "agy", requestedModel: "gemini-3.7-flash-high" } + ); + const claude = convertUsageToQuotaInfo( + { quotas: { gemini_weekly: usage.quotas.gemini_weekly } }, + { provider: "antigravity", requestedModel: "claude-opus-4-6-thinking" } + ); + + assert.equal(gemini, null); + assert.equal(claude, null); + assert.equal(scoreResetAwareQuota(gemini, resolveResetAwareConfig({})).score, 0.5); + assert.equal(scoreResetAwareQuota(claude, resolveResetAwareConfig({})).score, 0.5); +}); + +test("Claude family excludes unknown weekly buckets", () => { + const quota = convertUsageToQuotaInfo( + { + quotas: { + "claude-opus-4-6-thinking": { + used: 100, + total: 1000, + remainingPercentage: 90, + resetAt: resetAt5h, + }, + claude_gpt_weekly: { + used: 100, + total: 1000, + remainingPercentage: 90, + resetAt: resetAt7d, + }, + unrelated_weekly: usage.quotas.unrelated_weekly, + }, + }, + { provider: "agy", requestedModel: "claude-opus-4-6-thinking" } + ); + + assert.ok(quota); + assert.equal(quota.limitReached, false); + assert.equal(quota.windows?.unrelated_weekly, undefined); + assert.equal(quota.window7d?.percentUsed, 0.1); +}); + +test("unscoped provider-limits conversion retains conservative global windows", () => { + const quota = convertUsageToQuotaInfo(usage); + + assert.ok(quota); + assert.equal(quota.window5h?.percentUsed, 1); + assert.equal(quota.window7d?.percentUsed, 1); + assert.equal(quota.limitReached, true); +}); + +test("reset-aware fetch scope is family-wide for Antigravity and * otherwise", () => { + assert.equal(getQuotaFetchScope("agy", "gemini-3.7-flash-high"), "family:gemini"); + assert.equal(getQuotaFetchScope("antigravity", "claude-opus-4-6-thinking"), "family:claude"); + assert.equal(getQuotaFetchScope("codex", "gpt-5"), "*"); +}); + +test("buildAutoCandidates uses the shared Antigravity fetch-scope helper", () => { + const combo = fs.readFileSync(new URL("../../open-sse/services/combo.ts", import.meta.url), "utf8"); + const strategies = fs.readFileSync( + new URL("../../open-sse/services/combo/quotaStrategies.ts", import.meta.url), + "utf8" + ); + + assert.match(combo, /getQuotaFetchScope\(/); + assert.doesNotMatch( + combo, + /provider === "antigravity" \|\| provider === "agy"\s*\n\s*\? getQuotaScopedModelForProvider/ + ); + assert.match(strategies, /getQuotaFetchScope\(/); + assert.doesNotMatch(strategies, /function getQuotaFetchScope/); +}); + +afterEach(() => { + genericModule.__testing?.resetUsageFetcher?.(); + genericModule.__testing?.clearCache?.(); +}); + +test("fetchGenericQuota scopes Gemini windows and still catalogs sibling families", async () => { + let fetches = 0; + genericModule.__testing.setUsageFetcher(async () => { + fetches += 1; + return usage; + }); + + const quota = await fetchGenericQuota("conn-gemini", { + provider: "agy", + requestedModel: "agy/gemini-3.7-flash-high", + }); + + assert.ok(quota); + assert.equal(quota.window5h?.percentUsed, 0.03); + assert.equal(quota.limitReached, false); + assert.equal(quota.windows?.claude_gpt_weekly, undefined); + assert.equal(quota.windows?.["claude-opus-4-6-thinking"], undefined); + const windows = getQuotaWindows("agy"); + assert.equal(windows.includes("claude_gpt_weekly"), true); + assert.equal(windows.includes("gemini_weekly"), true); + assert.equal(fetches, 1); +}); + +test("invalidateGenericQuotaCache clears every family-scoped entry for a connection", async () => { + let fetches = 0; + genericModule.__testing.setUsageFetcher(async () => { + fetches += 1; + return usage; + }); + + const connectionId = "conn-both-families"; + await fetchGenericQuota(connectionId, { + provider: "agy", + requestedModel: "gemini-3.7-flash-high", + }); + await fetchGenericQuota(connectionId, { + provider: "agy", + requestedModel: "claude-opus-4-6-thinking", + }); + assert.equal(fetches, 2); + + await fetchGenericQuota(connectionId, { + provider: "agy", + requestedModel: "gemini-3.7-flash-high", + }); + assert.equal(fetches, 2); + + invalidateGenericQuotaCache("agy", connectionId); + + await fetchGenericQuota(connectionId, { + provider: "agy", + requestedModel: "gemini-3.7-flash-high", + }); + await fetchGenericQuota(connectionId, { + provider: "agy", + requestedModel: "claude-opus-4-6-thinking", + }); + assert.equal(fetches, 4); +}); + +test("non-Antigravity generic quota cache stays per connection, not per model", async () => { + let fetches = 0; + genericModule.__testing.setUsageFetcher(async () => { + fetches += 1; + return usage; + }); + + const connectionId = "conn-kimi"; + await fetchGenericQuota(connectionId, { + provider: "kimi", + requestedModel: "kimi-k2.5", + }); + await fetchGenericQuota(connectionId, { + provider: "kimi", + requestedModel: "kimi-k2.7", + }); + assert.equal(fetches, 1); +}); From f8a0f9c1f853e906cfba27fe7f4e7752c7cb6676 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Thu, 3 Sep 2026 23:48:48 -0300 Subject: [PATCH 096/143] chore(quality): rebaseline combo.ts for the stacked reset-aware scoring (#12678) Rebaseline medido no tip com os 4 PRs da leva mergeados. --- changelog.d/maintenance/houminxi-combo-filesize.md | 1 + config/quality/file-size-baseline.json | 6 +++--- 2 files changed, 4 insertions(+), 3 deletions(-) create mode 100644 changelog.d/maintenance/houminxi-combo-filesize.md diff --git a/changelog.d/maintenance/houminxi-combo-filesize.md b/changelog.d/maintenance/houminxi-combo-filesize.md new file mode 100644 index 0000000000..8a290c2461 --- /dev/null +++ b/changelog.d/maintenance/houminxi-combo-filesize.md @@ -0,0 +1 @@ +- **chore(quality):** rebaseline `open-sse/services/combo.ts` for the reset-aware scoring the HouMinXi batch stacked ([#12637](https://github.com/diegosouzapw/OmniRoute/pull/12637)) diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index d27b98aae0..c6637c0ea8 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -425,8 +425,7 @@ "open-sse/mcp-server/server.ts": 1572, "open-sse/services/accountFallback.ts": 2467, "open-sse/services/adobeFireflyBrowserLogin.ts": 1401, - "open-sse/services/combo.ts": 4041, - "open-sse/services/combo.ts": 4075, + "open-sse/services/combo.ts": 4080, "open-sse/translator/response/openai-responses.ts": 1466, "open-sse/utils/cursorAgentProtobuf.ts": 1547, "open-sse/utils/proxyFetch.ts": 1271, @@ -642,5 +641,6 @@ "_rebaseline_2026_09_03_12604_claude_code_2_1_258": "PR #12604 (bump da wire identity do Claude Code 2.1.220->2.1.258, commits do @ggiak vindos do #12402) crescimento proprio: src/app/(dashboard)/dashboard/settings/components/RoutingTab.tsx 1606->1607 (+1, a linha do seletor que acompanha a nova versao de identidade). Uma linha num painel de settings ja existente; nao ha o que extrair. Coberto por client-identity-profiles e claude-codex-identity-version-sync (138/138 focados).", "_rebaseline_2026_09_03_hartmark_batch": "Leva hartmark (#12293 #12355 #12447 #12445 #12446 #12460 #12461 #12338 #12448) crescimento proprio, medido no tip com os nove mergeados: src/app/(dashboard)/dashboard/combos/page.tsx 5012->5018 (+6, #12355 impede que a falha de bundling do tiktoken de um provider sem relacao derrube /api/providers, e o painel passa a lidar com o estado degradado); open-sse/services/combo.ts 4023->4036 (+13, #12338 nos fixes do universal-handoff: nota de bare-fallback, escopo por mesma requisicao e log da falha silenciosa). Fiacao em chokepoints existentes do roteamento de combo. NAO cobre codex.ts nem stream.ts, ja violando no tip antes desta leva (drift da base).", "_rebaseline_2026_09_03_error_boundary_campaign": "Campanha de error-boundary (#12431 #12438 #12444 #12454 #12455 #12456 #12457 #12458 #12459 #12465 #12466 #12467 #12469 #12435), medido no tip com os 14 mergeados. open-sse/executors/codex.ts 1499->1505: os primeiros 4 (1499->1503) sao DRIFT ANTERIOR a esta campanha, ja presente no tip antes dela; os 2 ultimos (1503->1505) sao do #12444, que fecha o boundary de falha da resposta do Codex. Absorver o drift junto foi inevitavel porque o cap e um numero so, mas fica registrado aqui que 4 das 6 linhas nao sao desta leva. open-sse/vendor/codex-chatgpt-web/bridge.ts 1322->1335 (+13): tambem do #12444, no mesmo caminho de falha. NAO cobre open-sse/utils/stream.ts, que segue violando por drift anterior e independente.", - "_rebaseline_2026_09_03_12352_apikey_acl": "PR #12352 (fix/api-key-create-acl-12275) crescimento proprio: src/lib/db/apiKeys.ts 1610->1625 (+15). A criacao de API key descartava a ACL enviada no payload; preservar essa ACL exige carregar e persistir o conjunto no mesmo chokepoint de INSERT do modulo de dominio, sem extracao possivel sem partir a funcao de criacao ao meio. Coberto pelos testes do proprio PR (54/54 focados na leva)." + "_rebaseline_2026_09_03_12352_apikey_acl": "PR #12352 (fix/api-key-create-acl-12275) crescimento proprio: src/lib/db/apiKeys.ts 1610->1625 (+15). A criacao de API key descartava a ACL enviada no payload; preservar essa ACL exige carregar e persistir o conjunto no mesmo chokepoint de INSERT do modulo de dominio, sem extracao possivel sem partir a funcao de criacao ao meio. Coberto pelos testes do proprio PR (54/54 focados na leva).", + "_rebaseline_2026_09_03_houminxi_combo_stacked": "Leva HouMinXi (#12624 #12626 #12632 #12637): open-sse/services/combo.ts 4075->4080 (+5), medido no tip com os quatro mergeados. Cada PR registrou o proprio crescimento contra o tip de onde forkou (o #12637 ja subira o cap para 4075); as 5 linhas restantes so aparecem quando eles empilham, porque mais de um toca o mesmo chokepoint de scoring reset-aware em combo.ts. Fiacao em ponto existente, sem extracao possivel sem partir a funcao de selecao de alvos. Coberto por combo-strategies e reset-aware-request-scope-12600 (119/119 focados na leva)." } From 74c2d2639333cc118446dd4118b8167e47d85fcf Mon Sep 17 00:00:00 2001 From: Markus Hartung Date: Fri, 4 Sep 2026 05:03:16 +0200 Subject: [PATCH 097/143] fix(responses-continuation): chain off the effective post-reconstruction input, not the pre-reconstruction client bytes (#12641) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 3 PRs desta leva sobre o tip de `release/v3.8.51`: os três boardaram sem conflito, `typecheck:core` limpo e **22/22** nos arquivos de teste que trazem. O crescimento de `src/sse/handlers/chat.ts` (2450 → 2454) é do #12641 e vai num PR de rebaseline próprio. Obrigado, @hartmark. --- open-sse/utils/requestLogger.ts | 23 +++++- src/lib/db/responsesContinuationStore.ts | 20 ++++- .../videoBridgeSnapshotRedaction.ts | 11 ++- src/sse/handlers/chat.ts | 11 +++ .../unit/responses-continuation-store.test.ts | 79 +++++++++++++++++++ 5 files changed, 138 insertions(+), 6 deletions(-) diff --git a/open-sse/utils/requestLogger.ts b/open-sse/utils/requestLogger.ts index 553fc219e4..f6722ac674 100644 --- a/open-sse/utils/requestLogger.ts +++ b/open-sse/utils/requestLogger.ts @@ -28,7 +28,12 @@ export type RequestPipelinePayloads = { type RequestLogger = { sessionPath: null; - logClientRawRequest: (endpoint: unknown, body: unknown, headers?: HeaderInput) => void; + logClientRawRequest: ( + endpoint: unknown, + body: unknown, + headers?: HeaderInput, + effectiveInput?: unknown + ) => void; logRouteDecision: (decision: unknown) => void; logOpenAIRequest: (body: unknown) => void; logTargetRequest: (url: unknown, headers: HeaderInput, body: unknown) => void; @@ -392,12 +397,26 @@ export async function createRequestLogger( return { sessionPath: null, - logClientRawRequest(endpoint, body, headers = {}) { + logClientRawRequest(endpoint, body, headers = {}, effectiveInput) { payloads.clientRawRequest = { timestamp: new Date().toISOString(), endpoint, headers: maskSensitiveHeaders(headers), body: cloneBoundedForLog(body), + // The actual `input` this request dispatched with, captured AFTER + // OmniRoute's own previous_response_id reconstruction (see + // src/sse/handlers/chat.ts) -- `body` above is deliberately the + // pre-reconstruction raw client bytes (captureDeferredClientRawBody's + // whole point) and is NOT what got sent for a continued turn. + // resolvePreviousResponseState must chain off this field, not + // `body.input`: reading the raw pre-reconstruction input for a + // request that was itself a continuation compounds into progressively + // truncated history a few hops deep (live incident 2026-09-03, + // manifested as a malformed request with no leading system/user + // message rejected by the upstream provider). + ...(effectiveInput !== undefined + ? { effectiveInput: cloneBoundedForLog(effectiveInput) } + : {}), }; }, diff --git a/src/lib/db/responsesContinuationStore.ts b/src/lib/db/responsesContinuationStore.ts index 9c55d8edff..dce175dc92 100644 --- a/src/lib/db/responsesContinuationStore.ts +++ b/src/lib/db/responsesContinuationStore.ts @@ -81,7 +81,8 @@ export function resolvePreviousResponseState( const { artifact, state } = readCallArtifact(row.artifact_relpath); if (state !== "ready" || !artifact?.pipeline) return null; - const clientRawRequest = artifact.pipeline.clientRawRequest as { body?: unknown } | undefined; + const clientRawRequest = artifact.pipeline.clientRawRequest as + { body?: unknown; effectiveInput?: unknown } | undefined; const clientResponse = artifact.pipeline.clientResponse as { output?: unknown; summary?: { output?: unknown } } | undefined; @@ -94,7 +95,22 @@ export function resolvePreviousResponseState( // unconditionally unresolvable for every translate-mode/auto-routed // connection (previous_response_not_found on every attempt, regardless of // whether the id was real and the artifact was otherwise 'ready'). - const input = isPlainRecord(clientRawRequest?.body) ? clientRawRequest.body.input : undefined; + // + // effectiveInput first, body.input as a compat fallback for artifacts + // logged before this field existed: `body` is captureDeferredClientRawBody's + // deliberately pre-reconstruction snapshot of the raw client bytes. For a + // turn that was ITSELF a continuation, that's just the client's own trimmed + // delta, not the full input that actually dispatched -- chaining off it + // compounds into a progressively truncated reconstruction a few hops deep + // (live incident 2026-09-03: a malformed request with no leading + // system/user message, rejected by the upstream provider). effectiveInput + // is captured AFTER reconstruction runs (chat.ts) and is what this function + // must chain off so a multi-hop continuation stays accurate. + const input = Array.isArray(clientRawRequest?.effectiveInput) + ? clientRawRequest.effectiveInput + : isPlainRecord(clientRawRequest?.body) + ? clientRawRequest.body.input + : undefined; // A streaming clientResponse is clientPayloadCollector.build()'s output, which // always nests the caller's summary under `.summary` (see // createStructuredSSECollector in streamPayloadCollector.ts) -- a non-streaming diff --git a/src/lib/guardrails/videoBridgeSnapshotRedaction.ts b/src/lib/guardrails/videoBridgeSnapshotRedaction.ts index 8324b4be90..2815a770c8 100644 --- a/src/lib/guardrails/videoBridgeSnapshotRedaction.ts +++ b/src/lib/guardrails/videoBridgeSnapshotRedaction.ts @@ -105,10 +105,16 @@ interface ClientRawRequestLike { endpoint: unknown; body: unknown; headers?: unknown; + effectiveInput?: unknown; } interface RequestLoggerLike { - logClientRawRequest: (endpoint: unknown, body: unknown, headers?: unknown) => void; + logClientRawRequest: ( + endpoint: unknown, + body: unknown, + headers?: unknown, + effectiveInput?: unknown + ) => void; } /** @@ -130,7 +136,8 @@ export function logClientRawRequestRedacted( videoBridgeObserved ? redactVideoTranscriptFieldsForLog(clientRawRequest.body) : clientRawRequest.body, - clientRawRequest.headers + clientRawRequest.headers, + clientRawRequest.effectiveInput ); } diff --git a/src/sse/handlers/chat.ts b/src/sse/handlers/chat.ts index f72b074b31..6c03803ba1 100644 --- a/src/sse/handlers/chat.ts +++ b/src/sse/handlers/chat.ts @@ -753,6 +753,17 @@ async function handleChatImplementation( clientRawRequest = chatAdmission.resolveClientRawAfterAdmission(clientRawRequest, () => deferredClientRawBody.withClientBody((clientBody) => buildClientRawRequest(request, clientBody)) ); + // Sibling of clientRawRequest.body, not a replacement: .body stays the raw + // pre-reconstruction client bytes (see captureDeferredClientRawBody), while + // this is the `input` actually dispatched with -- after the + // previous_response_id reconstruction above ran, when it applies. A future + // continuation lookup against THIS response must resolve from this field, + // not the raw one. See the logClientRawRequest doc comment in requestLogger.ts. + if (clientRawRequest && Array.isArray((body as { input?: unknown }).input)) { + (clientRawRequest as { effectiveInput?: unknown }).effectiveInput = ( + body as { input: unknown[] } + ).input; + } // Guardrail pre-call pipeline — prompt injection, PII masking, and future custom rules. telemetry.startPhase("validate"); diff --git a/tests/unit/responses-continuation-store.test.ts b/tests/unit/responses-continuation-store.test.ts index 4f414c4e16..0b0bce17c3 100644 --- a/tests/unit/responses-continuation-store.test.ts +++ b/tests/unit/responses-continuation-store.test.ts @@ -131,6 +131,85 @@ test("resolvePreviousResponseState reads output from a wrapped (streaming) clien }); }); +test("resolvePreviousResponseState chains off effectiveInput, not the pre-reconstruction clientRawRequest.body", () => { + // Live incident (2026-09-03): clientRawRequest.body is deliberately captured + // BEFORE chat.ts's own previous_response_id reconstruction runs + // (captureDeferredClientRawBody's whole point -- it must reflect the raw + // client bytes for audit/guardrail purposes, not what OmniRoute rewrote the + // request into). For a turn that was ITSELF a continuation, body.input is + // just the client's own trimmed delta -- a handful of tool-call items with + // no leading system/user message. Chaining a LATER continuation off that + // instead of the request's real effective input compounds into a + // progressively truncated reconstruction, which the upstream provider then + // rejects outright ("Please ensure that function call turn comes + // immediately after a user turn..."). effectiveInput is captured AFTER + // reconstruction and must be what this function chains off. + insertCallLog({ + id: "log-continued-turn", + responseId: "resp_continued", + apiKeyId: "key-1", + detailState: "ready", + artifactRelPath: "2026-01-01/log-continued-turn.json", + }); + writeArtifact("2026-01-01/log-continued-turn.json", { + clientRawRequest: { + // What the client actually sent this turn: just the new delta, relying + // on OmniRoute to have reconstructed full history server-side. + body: { + input: [{ type: "function_call_output", call_id: "call_1", output: "42" }], + }, + // What this request ACTUALLY dispatched with, after chat.ts's own + // reconstruction expanded the prior turn's stored input+output back in. + effectiveInput: [ + { type: "message", role: "user", content: "hi" }, + { type: "message", role: "assistant", content: "calling a tool" }, + { type: "function_call", call_id: "call_1", name: "get_answer", arguments: "{}" }, + { type: "function_call_output", call_id: "call_1", output: "42" }, + ], + }, + providerRequest: { body: { input: [] } }, + clientResponse: { + id: "resp_continued", + output: [{ type: "message", role: "assistant", content: "the answer is 42" }], + }, + }); + + const result = store.resolvePreviousResponseState("resp_continued", "key-1"); + assert.deepEqual(result, { + input: [ + { type: "message", role: "user", content: "hi" }, + { type: "message", role: "assistant", content: "calling a tool" }, + { type: "function_call", call_id: "call_1", name: "get_answer", arguments: "{}" }, + { type: "function_call_output", call_id: "call_1", output: "42" }, + ], + output: [{ type: "message", role: "assistant", content: "the answer is 42" }], + }); +}); + +test("resolvePreviousResponseState falls back to clientRawRequest.body.input when effectiveInput is absent (pre-fix artifacts)", () => { + insertCallLog({ + id: "log-legacy-no-effective-input", + responseId: "resp_legacy", + apiKeyId: "key-1", + detailState: "ready", + artifactRelPath: "2026-01-01/log-legacy-no-effective-input.json", + }); + writeArtifact("2026-01-01/log-legacy-no-effective-input.json", { + clientRawRequest: { body: { input: [{ type: "message", role: "user", content: "hi" }] } }, + providerRequest: { body: { input: [{ type: "message", role: "user", content: "hi" }] } }, + clientResponse: { + id: "resp_legacy", + output: [{ type: "message", role: "assistant", content: "hello" }], + }, + }); + + const result = store.resolvePreviousResponseState("resp_legacy", "key-1"); + assert.deepEqual(result, { + input: [{ type: "message", role: "user", content: "hi" }], + output: [{ type: "message", role: "assistant", content: "hello" }], + }); +}); + test("resolvePreviousResponseState returns null for an unknown response id", () => { const result = store.resolvePreviousResponseState("resp_does_not_exist", "key-1"); assert.equal(result, null); From 6ff7b26277d04cd03664421adf254ad493ffa279 Mon Sep 17 00:00:00 2001 From: Markus Hartung Date: Fri, 4 Sep 2026 05:03:34 +0200 Subject: [PATCH 098/143] fix(dashboard): keep a request's pending-tracking id stable across combo target retries (#12650) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado em lote numa worktree combinada com os 3 PRs desta leva sobre o tip de `release/v3.8.51`: os três boardaram sem conflito, `typecheck:core` limpo e **22/22** nos arquivos de teste que trazem. O crescimento de `src/sse/handlers/chat.ts` (2450 → 2454) é do #12641 e vai num PR de rebaseline próprio. Obrigado, @hartmark. --- src/lib/usage/usageHistory.ts | 49 ++++++++++++++- tests/unit/usage-pending-sweep.test.ts | 85 ++++++++++++++++++++++++++ 2 files changed, 133 insertions(+), 1 deletion(-) diff --git a/src/lib/usage/usageHistory.ts b/src/lib/usage/usageHistory.ts index 4a1a9f9216..a6fb9a7d3a 100644 --- a/src/lib/usage/usageHistory.ts +++ b/src/lib/usage/usageHistory.ts @@ -155,6 +155,7 @@ declare global { details: Record>; }; pendingById: Map; + pendingIdByCorrelation: Map; } | undefined; } @@ -174,6 +175,7 @@ const pendingState = (globalThis.__omnirouteUsageHistoryPendingState ??= { details: Object.create(null) as Record>, }, pendingById: new Map(), + pendingIdByCorrelation: new Map(), }); const pendingRequests = pendingState.pendingRequests; @@ -184,6 +186,21 @@ const pendingRequests = pendingState.pendingRequests; */ const pendingById = pendingState.pendingById; +// Live incident: a combo dispatch calls trackPendingRequest once PER TARGET +// ATTEMPT (open-sse/handlers/chatCore.ts's single "started" call site, hit +// again on every fallback), each generating its OWN fresh id. A dashboard tab +// polling /api/logs/ for the FIRST attempt goes stale the moment that +// attempt finalizes and the combo silently retries with a different target +// under a different id -- the tab has no way to discover the new id, and the +// request keeps streaming (successfully) with nobody watching it live. Since +// correlationId is already stable across every attempt of one client request +// (see the trackPendingRequest call site's `correlationId` metadata field), +// reusing the SAME pending id for every attempt sharing a correlationId keeps +// one dashboard tab's poll target valid across combo fallbacks. Bounded by +// PENDING_SWEEP_INTERVAL_MS's existing reaper cycle (see sweepStalePendingRequests) +// so this never grows unboundedly with one-shot correlation ids. +const pendingIdByCorrelation = pendingState.pendingIdByCorrelation; + const DEFAULT_MAX_PENDING_REQUEST_AGE_MS = 60 * 60 * 1000; const MAX_PENDING_DETAILS = 5000; const PENDING_SWEEP_INTERVAL_MS = 5 * 60 * 1000; @@ -250,6 +267,20 @@ export function sweepStalePendingRequests( for (const detail of oldest) remove(detail); } + // pendingIdByCorrelation entries are correlation ids, never reused across + // separate client requests, so nothing else ever removes them — same + // age/cap sweep as pendingById above, or the map grows unboundedly. + for (const [correlationId, entry] of pendingIdByCorrelation) { + if (now - entry.touchedAt > maxAgeMs) pendingIdByCorrelation.delete(correlationId); + } + if (pendingIdByCorrelation.size > MAX_PENDING_DETAILS) { + const overflow = pendingIdByCorrelation.size - MAX_PENDING_DETAILS; + const oldest = [...pendingIdByCorrelation.entries()] + .sort((a, b) => a[1].touchedAt - b[1].touchedAt) + .slice(0, overflow); + for (const [correlationId] of oldest) pendingIdByCorrelation.delete(correlationId); + } + return removed; } @@ -309,11 +340,23 @@ export function trackPendingRequest( pendingRequests.details[connectionId][modelKey] = []; } const now = Date.now(); + // Reuse the same pending id across every target attempt of one client + // request (see pendingIdByCorrelation's module-level comment) so a + // dashboard tab's live poll survives a combo fallback to a different + // target instead of silently going stale. Concurrent speculative + // attempts (combo.ts's zeroLatencyOptimizationsEnabled hedging) can + // race two "started" calls for the same correlationId — the second + // simply overwrites the id-keyed view of the first's still-live entry, + // no worse than today's per-attempt id (which loses tracking entirely + // once any attempt finalizes) and self-corrects on the next attempt. + const reusableId = normalizedMetadata.correlationId + ? pendingIdByCorrelation.get(normalizedMetadata.correlationId)?.id + : undefined; const newDetail = { // crypto RNG (not Math.random) to satisfy CodeQL js/insecure-randomness — // this pending-request id flows into attempt logging; it's a correlation // id, not a security secret. - id: `${now}-${globalThis.crypto.randomUUID().slice(0, 6)}`, + id: reusableId ?? `${now}-${globalThis.crypto.randomUUID().slice(0, 6)}`, model, provider, connectionId, @@ -322,6 +365,9 @@ export function trackPendingRequest( }; pendingRequests.details[connectionId][modelKey].push(newDetail); pendingById.set(newDetail.id, newDetail); + if (normalizedMetadata.correlationId) { + pendingIdByCorrelation.set(normalizedMetadata.correlationId, { id: newDetail.id, touchedAt: now }); + } return newDetail.id; } else if (!started && nextCount >= 0) { if (pendingRequests.details[connectionId]?.[modelKey]?.length) { @@ -520,6 +566,7 @@ export function clearPendingRequests() { Record >; pendingById.clear(); + pendingIdByCorrelation.clear(); clearCompletedDetails(); } diff --git a/tests/unit/usage-pending-sweep.test.ts b/tests/unit/usage-pending-sweep.test.ts index d3de475b16..070f9fe0c9 100644 --- a/tests/unit/usage-pending-sweep.test.ts +++ b/tests/unit/usage-pending-sweep.test.ts @@ -112,3 +112,88 @@ test("invalid pending sweep max age falls back to one hour", () => { else process.env.MAX_PENDING_REQUEST_AGE_MS = previous; } }); + +// Live incident: a combo dispatch calls trackPendingRequest once PER TARGET +// ATTEMPT (open-sse/handlers/chatCore.ts's single "started" call site, hit +// again on every fallback), each previously generating its OWN fresh id. A +// dashboard tab polling /api/logs/ for the FIRST attempt went stale the +// moment that attempt finalized and the combo silently retried under a +// different id -- the tab had no way to discover the new id, even though the +// request kept streaming successfully. Fix: reuse the same pending id for +// every attempt sharing a correlationId (already passed as metadata on every +// "started" call, already stable across a combo's retries). +test("trackPendingRequest reuses the same id across a combo's target-attempt retries sharing a correlationId", () => { + clearPendingRequests(); + + const firstId = trackPendingRequest("model-a", "provider-a", "conn-a", true, { + correlationId: "corr-retry-1", + }); + assert.ok(firstId, "first attempt should produce an id"); + + // First target attempt finalizes (fails) -- the combo retries with a + // different target, but the SAME client-facing request/correlation. + trackPendingRequest("model-a", "provider-a", "conn-a", false); + assert.equal(getPendingById().has(firstId), false, "finalized attempt is removed by id"); + + const secondId = trackPendingRequest("model-b", "provider-b", "conn-b", true, { + correlationId: "corr-retry-1", + }); + + assert.equal(secondId, firstId, "retry attempt must reuse the first attempt's id"); + assert.equal(getPendingById().has(firstId), true, "reused id is live again under the new attempt"); + assert.equal(getPendingById().get(firstId)?.model, "model-b", "entry reflects the NEW attempt's target"); + + clearPendingRequests(); +}); + +test("trackPendingRequest never reuses an id across two different correlationIds", () => { + clearPendingRequests(); + + const idA = trackPendingRequest("model-a", "provider-a", "conn-a", true, { + correlationId: "corr-unrelated-1", + }); + const idB = trackPendingRequest("model-a", "provider-a", "conn-b", true, { + correlationId: "corr-unrelated-2", + }); + + assert.ok(idA && idB); + assert.notEqual(idA, idB, "unrelated client requests must never share a pending id"); + + clearPendingRequests(); +}); + +test("trackPendingRequest without a correlationId keeps generating a fresh id every attempt (unchanged behavior)", () => { + clearPendingRequests(); + + const idA = trackPendingRequest("model-a", "provider-a", "conn-a", true); + trackPendingRequest("model-a", "provider-a", "conn-a", false); + const idB = trackPendingRequest("model-a", "provider-a", "conn-a", true); + + assert.ok(idA && idB); + assert.notEqual(idA, idB, "no correlationId means no cross-attempt identity to reuse"); + + clearPendingRequests(); +}); + +test("sweepStalePendingRequests evicts stale correlation-id-to-pending-id mappings so an old id can never resurface", () => { + clearPendingRequests(); + + const firstId = trackPendingRequest("model-a", "provider-a", "conn-a", true, { + correlationId: "corr-stale-mapping", + }); + assert.ok(firstId); + trackPendingRequest("model-a", "provider-a", "conn-a", false); + + // Sweep with a max age of 0 so the just-recorded correlation mapping (whose + // touchedAt is "now") is immediately treated as stale, mirroring what a + // real 1-hour-later sweep does to a genuinely abandoned mapping. + sweepStalePendingRequests(Date.now() + HOUR_MS + MINUTE_MS, HOUR_MS); + + const secondId = trackPendingRequest("model-b", "provider-b", "conn-b", true, { + correlationId: "corr-stale-mapping", + }); + + assert.notEqual(secondId, firstId, "an evicted mapping must not resurrect the old id"); + + clearPendingRequests(); +}); From c41ec7f862f66f00f562b4b8a66ec9d450a7416e Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Fri, 4 Sep 2026 00:04:36 -0300 Subject: [PATCH 099/143] chore(quality): rebaseline chat.ts for #12641's effective-input persistence (#12680) Rebaseline medido no tip com o #12641 mergeado. --- changelog.d/maintenance/12641-chat-filesize.md | 1 + config/quality/file-size-baseline.json | 5 +++-- 2 files changed, 4 insertions(+), 2 deletions(-) create mode 100644 changelog.d/maintenance/12641-chat-filesize.md diff --git a/changelog.d/maintenance/12641-chat-filesize.md b/changelog.d/maintenance/12641-chat-filesize.md new file mode 100644 index 0000000000..c295e76bde --- /dev/null +++ b/changelog.d/maintenance/12641-chat-filesize.md @@ -0,0 +1 @@ +- **chore(quality):** rebaseline `src/sse/handlers/chat.ts` for the effective-input persistence the continuation fix needs ([#12641](https://github.com/diegosouzapw/OmniRoute/pull/12641)) diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index c6637c0ea8..56752d58b9 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -457,7 +457,7 @@ "src/shared/components/RequestLoggerV2.tsx": 1718, "src/shared/constants/providers/apikey/gateways.ts": 1462, "src/shared/services/cliRuntime.ts": 1296, - "src/sse/handlers/chat.ts": 2450, + "src/sse/handlers/chat.ts": 2454, "src/sse/services/auth.ts": 3450, "tests/unit/account-fallback-service.test.ts": 2453, "tests/unit/provider-validation-specialty.test.ts": 4656 @@ -642,5 +642,6 @@ "_rebaseline_2026_09_03_hartmark_batch": "Leva hartmark (#12293 #12355 #12447 #12445 #12446 #12460 #12461 #12338 #12448) crescimento proprio, medido no tip com os nove mergeados: src/app/(dashboard)/dashboard/combos/page.tsx 5012->5018 (+6, #12355 impede que a falha de bundling do tiktoken de um provider sem relacao derrube /api/providers, e o painel passa a lidar com o estado degradado); open-sse/services/combo.ts 4023->4036 (+13, #12338 nos fixes do universal-handoff: nota de bare-fallback, escopo por mesma requisicao e log da falha silenciosa). Fiacao em chokepoints existentes do roteamento de combo. NAO cobre codex.ts nem stream.ts, ja violando no tip antes desta leva (drift da base).", "_rebaseline_2026_09_03_error_boundary_campaign": "Campanha de error-boundary (#12431 #12438 #12444 #12454 #12455 #12456 #12457 #12458 #12459 #12465 #12466 #12467 #12469 #12435), medido no tip com os 14 mergeados. open-sse/executors/codex.ts 1499->1505: os primeiros 4 (1499->1503) sao DRIFT ANTERIOR a esta campanha, ja presente no tip antes dela; os 2 ultimos (1503->1505) sao do #12444, que fecha o boundary de falha da resposta do Codex. Absorver o drift junto foi inevitavel porque o cap e um numero so, mas fica registrado aqui que 4 das 6 linhas nao sao desta leva. open-sse/vendor/codex-chatgpt-web/bridge.ts 1322->1335 (+13): tambem do #12444, no mesmo caminho de falha. NAO cobre open-sse/utils/stream.ts, que segue violando por drift anterior e independente.", "_rebaseline_2026_09_03_12352_apikey_acl": "PR #12352 (fix/api-key-create-acl-12275) crescimento proprio: src/lib/db/apiKeys.ts 1610->1625 (+15). A criacao de API key descartava a ACL enviada no payload; preservar essa ACL exige carregar e persistir o conjunto no mesmo chokepoint de INSERT do modulo de dominio, sem extracao possivel sem partir a funcao de criacao ao meio. Coberto pelos testes do proprio PR (54/54 focados na leva).", - "_rebaseline_2026_09_03_houminxi_combo_stacked": "Leva HouMinXi (#12624 #12626 #12632 #12637): open-sse/services/combo.ts 4075->4080 (+5), medido no tip com os quatro mergeados. Cada PR registrou o proprio crescimento contra o tip de onde forkou (o #12637 ja subira o cap para 4075); as 5 linhas restantes so aparecem quando eles empilham, porque mais de um toca o mesmo chokepoint de scoring reset-aware em combo.ts. Fiacao em ponto existente, sem extracao possivel sem partir a funcao de selecao de alvos. Coberto por combo-strategies e reset-aware-request-scope-12600 (119/119 focados na leva)." + "_rebaseline_2026_09_03_houminxi_combo_stacked": "Leva HouMinXi (#12624 #12626 #12632 #12637): open-sse/services/combo.ts 4075->4080 (+5), medido no tip com os quatro mergeados. Cada PR registrou o proprio crescimento contra o tip de onde forkou (o #12637 ja subira o cap para 4075); as 5 linhas restantes so aparecem quando eles empilham, porque mais de um toca o mesmo chokepoint de scoring reset-aware em combo.ts. Fiacao em ponto existente, sem extracao possivel sem partir a funcao de selecao de alvos. Coberto por combo-strategies e reset-aware-request-scope-12600 (119/119 focados na leva).", + "_rebaseline_2026_09_04_12641_continuation_effective_input": "PR #12641 crescimento proprio: src/sse/handlers/chat.ts 2450->2454 (+4). A continuacao por previous_response_id encadeava a partir de clientRawRequest.body.input, que e capturado ANTES da reconstrucao do proprio chat.ts; quando o turno anterior ja era uma continuacao, esse campo guarda so o delta do cliente, e o erro se acumulava a cada salto ate a reconstrucao virar itens de tool sem prefixo. Persistir o input EFETIVO exige as linhas no ponto onde a reconstrucao termina, dentro do fluxo de despacho. Coberto por tests/unit/responses-continuation-store.test.ts (22/22 focados na leva)." } From 488f57e9d3fccc8d1741fdf21d35d5730b118a18 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Fri, 4 Sep 2026 00:45:38 -0300 Subject: [PATCH 100/143] feat(catalog): eligibility-gated free-tier bucket (#12669) * test(catalog): pin the 2026-09-02 free-tier re-audit facts for gemini, ollama-cloud, groq, nara and mistral * fix(catalog): re-audit gemini, ollama-cloud, groq, nara and mistral against official pages * fix(catalog): restore the console-verified Mistral 1B pool and harden its regression test * docs(free-tiers): move headline to the re-audited ~1.50B and refresh pool counts * chore(free-tiers): retire stale Groq free-tier text and preset model; fix catalog header * feat(catalog): eligibilityGate field and gatedRecurringTokens total * docs(free-tiers): state the evidence-comment rule honestly and retire the last "14.4K RPD" Groq texts * feat(check): docs-counts gate validates the eligibility-gated free-tier figure * fix(docs): budget card reads computeFreeModelTotals() instead of regex-parsing the catalog * docs(free-tiers): retire the stale Gemini onboarding quota text * feat(radar): carry eligibilityGate through the feed schema, the merge and the summary API * feat(dashboard): show the eligibility-gated free-tier figure apart from the headline * docs(free-tier): refresh catalog-entry counts to 442 after base sync * feat(catalog): ModelScope as the first eligibility-gated pool; document the gated bucket and how we count * docs(free-tiers): restore README spacing lost in the merge and re-sync the guide counts * docs(free-tiers): correct the unsummed-catalog comparison figure to the current catalog * docs(free-tiers): re-sync numbers after merging release/v3.8.51 (Cerebras reclassified upstream) * docs(free-tiers): re-sync numbers after merging PR1 (Cerebras reclassified upstream) * fix(docs): keep the NaraRouter plans endpoint out of the API-path checker; rebaseline gateways.ts (+3) * fix(catalog): keep eligibility-gated rows out of every headline-adjacent figure The eligibility gate was honored by the steady headline and the pool count, but three adjacent figures still counted gated rows: the credit reductions feeding steadyWithRecurringCreditsTokens/firstMonthRealisticTokens, the uncappedProviders list ("permanently free, no cap"), and the docs gate's free-forever provider set, which was built from freeType alone. - computeFreeModelTotals: filter !isGated in the recurring-credit, one-time-credit and uncapped predicates; gatedProviders semantics unchanged (steady rows only). - check-docs-counts-sync: exclude eligibility-gated rows from the FOREVER set, which moves the live free-forever count 53 -> 52 (the base's value). README, promise-pillars.svg and FREE-TIERS-GUIDE re-synced. - gen-budget-card-svg: skip gated one-time credits like the totals do, and fail loudly on `--out` without a path. - Tests: gated one-time credit does not move firstMonthRealisticTokens; a gated uncapped row is not in uncappedProviders; shipped gated rows carry no credit tokens; the committed budget card is byte-identical to a fresh generation. * test(catalog): allow eligibilityGate in the no-per-row-rating key allowlist The allowlist landed on the base with #12318, after this branch's field was designed; eligibilityGate says who may claim a quota, not how much a row can be trusted. --------- Co-authored-by: diegosouzapw Co-authored-by: diegosouzapw --- README.md | 8 +- docs/diagrams/free-tier-budget.svg | 4 +- docs/getting-started/FREE-TIERS-GUIDE.md | 16 +- docs/reference/FREE_TIERS.md | 16 +- docs/screenshots/free-tier-budget-card.svg | 136 +++++++------ open-sse/config/freeModelCatalog.data.ts | 8 + open-sse/config/freeModelCatalog.ts | 66 +++++- scripts/check/check-docs-counts-sync.mjs | 59 ++++-- scripts/research/gen-budget-card-svg.mjs | 189 ++++++++++++------ .../usage/components/FreeBudgetCard.tsx | 31 ++- src/app/api/free-tier/summary/route.ts | 1 + src/i18n/messages/ar.json | 1 + src/i18n/messages/az.json | 1 + src/i18n/messages/bg.json | 1 + src/i18n/messages/bn.json | 1 + src/i18n/messages/cs.json | 1 + src/i18n/messages/da.json | 1 + src/i18n/messages/de.json | 1 + src/i18n/messages/en.json | 1 + src/i18n/messages/es.json | 1 + src/i18n/messages/fa.json | 1 + src/i18n/messages/fi.json | 1 + src/i18n/messages/fr.json | 1 + src/i18n/messages/gu.json | 1 + src/i18n/messages/he.json | 1 + src/i18n/messages/hi.json | 1 + src/i18n/messages/hu.json | 1 + src/i18n/messages/id.json | 1 + src/i18n/messages/it.json | 1 + src/i18n/messages/ja.json | 1 + src/i18n/messages/ko.json | 1 + src/i18n/messages/mr.json | 1 + src/i18n/messages/ms.json | 1 + src/i18n/messages/nl.json | 1 + src/i18n/messages/no.json | 1 + src/i18n/messages/phi.json | 1 + src/i18n/messages/pl.json | 1 + src/i18n/messages/pt-BR.json | 1 + src/i18n/messages/pt.json | 1 + src/i18n/messages/ro.json | 1 + src/i18n/messages/ru.json | 1 + src/i18n/messages/sk.json | 1 + src/i18n/messages/sv.json | 1 + src/i18n/messages/sw.json | 1 + src/i18n/messages/ta.json | 1 + src/i18n/messages/te.json | 1 + src/i18n/messages/th.json | 1 + src/i18n/messages/tr.json | 1 + src/i18n/messages/uk-UA.json | 1 + src/i18n/messages/ur.json | 1 + src/i18n/messages/vi.json | 1 + src/i18n/messages/zh-CN.json | 1 + src/i18n/messages/zh-TW.json | 1 + src/lib/radar/applyFeed.ts | 11 + src/lib/radar/feedSchema.ts | 2 + src/lib/radar/index.ts | 1 + tests/unit/check-docs-counts-sync.test.ts | 45 ++++- .../free-catalog-no-confidence-field.test.ts | 3 + .../free-model-catalog-gated-bucket.test.ts | 165 +++++++++++++++ tests/unit/free-tier-reaudit-2026-09.test.ts | 34 ++++ .../free-tier-summary-radar-overlay.test.ts | 51 +++++ tests/unit/gen-budget-card-svg.test.ts | 50 +++++ .../unit/radar-feed-eligibility-gate.test.ts | 128 ++++++++++++ tests/unit/ui/free-budget-card-gated.test.tsx | 77 +++++++ 64 files changed, 980 insertions(+), 163 deletions(-) create mode 100644 tests/unit/free-model-catalog-gated-bucket.test.ts create mode 100644 tests/unit/gen-budget-card-svg.test.ts create mode 100644 tests/unit/radar-feed-eligibility-gate.test.ts create mode 100644 tests/unit/ui/free-budget-card-gated.test.tsx diff --git a/README.md b/README.md index 74d8b0e0ba..b64f3dc543 100644 --- a/README.md +++ b/README.md @@ -17,9 +17,9 @@
-> Stacking free tiers by hand is painful — dozens of SDKs, dozens of rate limits, and no idea how much you actually have. OmniRoute catalogs **442 free-tier entries across 34 recurring pool keys** and computes the token headline from the **16 pools with a published positive monthly budget plus five per-model Groq caps**, deduplicated by shared pool. The result stays visible on the dashboard (`/dashboard/free-tiers`). +> Stacking free tiers by hand is painful — dozens of SDKs, dozens of rate limits, and no idea how much you actually have. OmniRoute catalogs **444 free-tier entries across 34 recurring pool keys** and computes the token headline from the **16 pools with a published positive monthly budget plus five per-model Groq caps**, deduplicated by shared pool. Quotas that only open after a regional identity check (today: ModelScope) are shown apart, +~6M behind regional identity verification, and never summed into the headline. The result stays visible on the dashboard (`/dashboard/free-tiers`). -OmniRoute free-tier budget card: ~1.47B free tokens per month steady, up to ~2.10B in the first month with signup credits, from 34 documented recurring pool keys covering 442 cataloged free-tier entries behind one endpoint. Honest pool-deduped math — each shared pool counted once, including 16 recurring pools with a published positive monthly token budget plus five per-model Groq caps; 13 providers are marked avoid in the terms-risk catalog so you decide. Budget bar includes Mistral 1B, Nara 210M, LLM7 150M, Groq 30M (five per-model caps) and smaller pools, plus first-month signup credits and permanently-free no-token-cap providers surfaced separately so they never inflate the headline. Live used/remaining on /dashboard/free-tiers. +OmniRoute free-tier budget card: ~1.47B free tokens per month steady, up to ~2.10B in the first month with signup credits, from 34 documented recurring pool keys covering 444 cataloged free-tier entries behind one endpoint. Honest pool-deduped math — each shared pool counted once, including 16 recurring pools with a published positive monthly token budget plus five per-model Groq caps; 13 providers are marked avoid in the terms-risk catalog so you decide. Budget bar includes Mistral 1B, Nara 210M, LLM7 150M, Groq 30M (five per-model caps) and smaller pools, plus first-month signup credits and permanently-free no-token-cap providers surfaced separately so they never inflate the headline. Live used/remaining on /dashboard/free-tiers. > Animated summary of the live `/dashboard/free-tiers` page. Full methodology (pool dedupe, credit tiers, provider terms): **[docs/reference/FREE_TIERS.md](docs/reference/FREE_TIERS.md)**. > @@ -648,7 +648,7 @@ of your shell history. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md) -> **352 registered providers** across the canonical chat, media, search, local, cloud-agent and system collections, including **152 carrying `hasFree: true` discovery metadata**. The chat model registry covers **229 providers / 2,554 distinct provider-model pairs / 1,283 raw model IDs**; the separate free-budget catalog has **442 per-model rows**, **34 recurring pools** and **52 recurring/keyless free-forever providers**. These are different denominators by design; definitions and pool-deduped calculations live in the [Provider Reference](docs/reference/PROVIDER_REFERENCE.md) and [Free Tiers](docs/reference/FREE_TIERS.md). +> **352 registered providers** across the canonical chat, media, search, local, cloud-agent and system collections, including **152 carrying `hasFree: true` discovery metadata**. The chat model registry covers **229 providers / 2,554 distinct provider-model pairs / 1,283 raw model IDs**; the separate free-budget catalog has **444 per-model rows**, **34 recurring pools** and **52 recurring/keyless free-forever providers**. These are different denominators by design; definitions and pool-deduped calculations live in the [Provider Reference](docs/reference/PROVIDER_REFERENCE.md) and [Free Tiers](docs/reference/FREE_TIERS.md).
@@ -1307,7 +1307,7 @@ Métricas canônicas em 2026-08-24: **1.029 vídeos únicos** · **11.132.922 vi Resilience GuideCircuit breakers, cooldowns, queue, anti-thundering herd, TLS spoofing Auto-Combo Engine16-factor scoring, mode packs, self-healing Proxy Guide3-level proxy system, 1proxy marketplace, registry CRUD - Free TiersConsolidated directory: 34 documented recurring pools / 442 cataloged free-tier entries + Free TiersConsolidated directory: 34 documented recurring pools / 444 cataloged free-tier entries Features GalleryVisual dashboard tour with screenshots Codebase DocumentationBeginner-friendly codebase walkthrough diff --git a/docs/diagrams/free-tier-budget.svg b/docs/diagrams/free-tier-budget.svg index 19e0ca5f16..f706e4824f 100644 --- a/docs/diagrams/free-tier-budget.svg +++ b/docs/diagrams/free-tier-budget.svg @@ -1,4 +1,4 @@ - + Pool-deduplicated chart of the 16 recurring free-token pools with positive published budgets (plus Groq's five per-model caps as one segment), plus signup credits and uncapped providers shown separately. @@ -64,7 +64,7 @@ ~1.47B FREE TOKENS / MONTH · STEADY up to ~2.10B in your first month — signup credits - documented free tiers · 34 recurring pools · 442 catalog entries · one endpoint + documented free tiers · 34 recurring pools · 444 catalog entries · one endpoint diff --git a/docs/getting-started/FREE-TIERS-GUIDE.md b/docs/getting-started/FREE-TIERS-GUIDE.md index 804707209d..f93250470a 100644 --- a/docs/getting-started/FREE-TIERS-GUIDE.md +++ b/docs/getting-started/FREE-TIERS-GUIDE.md @@ -1,6 +1,6 @@ # Free Tiers Guide: Understand and Combine Free AI Access -> **TL;DR**: OmniRoute registers 352 provider IDs, with **152 provider-catalog entries marked `hasFree`**. The stricter audited free-model catalog covers **34 recurring pool keys / 442 entries** (435 active + 7 discontinued). Connect several suitable providers for broader fallback capacity; every quota, approval rule, privacy policy, and paid-overage condition still applies. +> **TL;DR**: OmniRoute registers 352 provider IDs, with **152 provider-catalog entries marked `hasFree`**. The stricter audited free-model catalog covers **34 recurring pool keys / 444 entries** (437 active + 7 discontinued). Connect several suitable providers for broader fallback capacity; every quota, approval rule, privacy policy, and paid-overage condition still applies. --- @@ -159,13 +159,13 @@ provider's quota or access policy. The live, pool-deduplicated catalog currently reports: -| Metric | Current audited value | Interpretation | -| ---------------------------------------------------- | -----------------------------------------------: | ----------------------------------------------------------------------------------------- | -| Recurring quantified grant | **~1.47B tokens/month** | Shared pools counted once; excludes uncapped providers from the sum | -| First month with signup grants | **~2.10B tokens** | Recurring total plus one-time and recurring credits | -| Audited free-model inventory | **34 recurring pool keys / 442 catalog entries** | 435 active + 7 discontinued; distinct from the 352-provider catalog | -| Recurring/keyless free-forever providers represented | **52** | Unique providers across recurring daily/monthly/credit/uncapped and keyless catalog types | -| Provider catalog entries marked `hasFree` | **152 / 352** | Broader provider metadata; not all have a quantifiable recurring quota | +| Metric | Current audited value | Interpretation | +| ---------------------------------------------------- | -----------------------------------------------: | -------------------------------------------------------------------------------------------------------------------------- | +| Recurring quantified grant | **~1.47B tokens/month** | Shared pools counted once; excludes uncapped providers from the sum | +| First month with signup grants | **~2.10B tokens** | Recurring total plus one-time and recurring credits | +| Audited free-model inventory | **34 recurring pool keys / 444 catalog entries** | 437 active + 7 discontinued; distinct from the 352-provider catalog | +| Recurring/keyless free-forever providers represented | **52** | Unique providers across recurring daily/monthly/credit/uncapped and keyless catalog types, eligibility-gated rows excluded | +| Provider catalog entries marked `hasFree` | **152 / 352** | Broader provider metadata; not all have a quantifiable recurring quota | These values are computed from `open-sse/config/freeModelCatalog.ts`; see the [Free Tiers Reference](../reference/FREE_TIERS.md) for pool deduplication, ToS flags, diff --git a/docs/reference/FREE_TIERS.md b/docs/reference/FREE_TIERS.md index 394142ccd2..963bd3e421 100644 --- a/docs/reference/FREE_TIERS.md +++ b/docs/reference/FREE_TIERS.md @@ -1,7 +1,7 @@ --- title: "Free Tiers & Free-Token Budget" version: 3.8.50 -lastUpdated: 2026-09-02 +lastUpdated: 2026-09-03 --- # Free Tiers & Free-Token Budget @@ -19,6 +19,7 @@ lastUpdated: 2026-09-02 | **+ first month with signup credits** | **~2.10B** | Steady + one-time signup credits (Together $25, Z.AI 20M, DeepSeek 5M, …), deduped per account. **First month only** — does not recur. | | **+ permanently free, no published cap** | _un-quantifiable_ | `siliconflow`, `glm-cn` (GLM-4-Flash), `tencent`, `baidu`, `kilo-gateway`, `opencode-zen`, `gemini`, `ollama-cloud` — real recurring access, rate/concurrency-limited, **no token cap to count**. Listed, never summed (counting them at `RPM×24/7` is the inflation we reject). | | **+ deposit-unlock boost** | **+~24M** | A one-time **$10** OpenRouter top-up raises its free pool from 50 → 1000 req/day. Reported separately so it never inflates the steady number. | +| **+ behind a regional identity check** | **+~6M** | `modelscope` (Alibaba Cloud binding + mainland-China real-name verification). Real recurring quota, exposed as `gatedRecurringTokens` / `gatedProviders` and on the dashboard. Never summed into the headline: +~6M behind regional identity verification. | | Theoretical ceiling (all rate limits, 24/7) | ~10B | Sum of every provider rate limit extrapolated to non-stop use. **Not a guarantee** — do not headline this. | **Honest headline:** _OmniRoute aggregates **~1.47B documented free tokens per month** (up to ~2.10B in your first month with signup credits) across 34 free-tier pools — plus a long tail of permanently-free, no-cap providers — and RTK + Caveman compression (15–95% token savings) stretches that further._ @@ -78,10 +79,23 @@ purpose. - Daily token cap → `monthly = daily × 30`. Only RPD documented → `RPD × ~800 output tokens × 30`. Only RPM/TPM (no daily cap) → **uncapped** (see below). - **Permanently free, but no published token cap** (`siliconflow`, `glm-cn`, `tencent`, `baidu`, `kilo-gateway`, `opencode-zen`, `gemini`, `ollama-cloud`): these are real recurring free access, rate/concurrency-limited. We classify them `recurring-uncapped` and **never sum them** — multiplying `RPM × 24/7 × 30d` would produce a fantasy ceiling (the inflation we reject). They are listed so you know they exist. - **Deposit-unlock boost:** a one-time small top-up that permanently raises a free quota (OpenRouter: $10 → 1000 req/day ≈ +24M/mo). Reported as a separate figure, kept out of the steady headline. +- **Eligibility-gated quotas** (`eligibilityGate: "regional-identity"`): a real recurring quota that only opens after a region-bound identity check (mainland-China real-name verification today). Counted with the same pool-dedupe rule into a separate figure (`gatedRecurringTokens`), never into the steady headline. The regime (`freeType`) is unchanged, so routing is unchanged. - **Evidence classes.** The rule: a number in the catalog cites its source in an `// evidence:` comment next to the entry — `public-page` (a provider page anyone can read), `api-public` (an unauthenticated endpoint of the provider, e.g. NaraRouter's public plans endpoint at router.bynara.id), or `console-verified por ` (the figure is only visible inside an account console; the comment records who saw it and when, and the public page that says the cap exists). The state today: the five blocks re-audited on 2026-09-02 carry it (`gemini`, `groq`, `mistral`, `ollama-cloud`, `nara`); entries that predate the 2026-09-02 re-audit inherit the earlier research until they are touched; any **new or changed** number without an evidence comment is a bug. Today only `mistral` is console-verified. --- +## Why our number is smaller than other aggregators' + +Most "free tokens per month" figures in this space are sums of per-model labels. Ours is not, on purpose: + +- **Each shared pool is counted once.** Mistral's free plan is one 1B/month allowance per organization; listing it under five models does not make it 5B. Summed per model, our own catalog would read **~7.4B** (recomputed on 2026-09-03; this figure is not CI-gated — re-measure it whenever the catalog changes) — the headline says **~1.47B** because that is what one account of each provider can actually spend. +- **Daily caps are converted, rates are not.** A documented tokens/day cap becomes `× 30`; a documented requests/day cap becomes `RPD × ~800 tokens × 30`; a provider that only publishes requests-per-minute has **no** monthly figure and is listed as _uncapped_, never summed. Multiplying a rate limit by 24/7 is the inflation we refuse. +- **Quotas behind a regional identity check are shown apart** (`+~6M behind regional identity verification`), because most readers cannot use them. +- **Signup credits are first-month only** and reported as a second figure, never blended into the steady number. +- **The figure is enforced by CI.** `npm run check:docs-counts` recomputes the totals from the catalog and fails the build when this file, the README or the budget card drift from them. + +--- + ## ToS attention table > **ToS flag is advisory, not a routing gate.** Providers marked `tos` are still included in routing and combo/fallback by default; the flag only surfaces on `/dashboard/free-tiers` and `/api/free-tier/summary`. The `excludeTosAvoid` query parameter affects the summary view only, not global routing. The verdict lives in `open-sse/config/freeTierCatalog.ts` (informational, not read by routing engines). diff --git a/docs/screenshots/free-tier-budget-card.svg b/docs/screenshots/free-tier-budget-card.svg index 90ab6b0279..15e11774a9 100644 --- a/docs/screenshots/free-tier-budget-card.svg +++ b/docs/screenshots/free-tier-budget-card.svg @@ -1,94 +1,104 @@ - - - -OmniRoute · /dashboard/free-tiers · preview mockup + + + +OmniRoute · /dashboard/free-tiers · preview mockup Monthly free-token budget -20 free pools · 446 models · one endpoint +21 free pools · 444 models · one endpoint Steady / month -~1.48B +~1.47B First month (+ signup credits) ~2.10B ToS-flagged (you decide) 13 providers - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + Each segment = one free pool · widths floored so every provider shows · honest numbers in the grid. Mistral Large 3 1.00B -GPT-4o mini 150M +Agnes 2.0 Flash 210M -Tencent Hy3 150M +GPT-4o mini 150M -Gemini 2.5 Flash 60M +Llama 3.3 70B 30M -Llama 3.3 70B 30M +Grok-3 24M -Grok-3 24M +GPT-4o 7M -DeepSeek V4 Pro 20M +GPT-OSS 120B 6M -GPT-4o 7M +GPT-OSS 20B 6M -MiniMax-M2.7 6M +GPT-OSS Safeguard 20B 6M -Arcee Trinity Large Prev 5M +Qwen3.6 27B 6M -NavyAI free pool 5M +Qwen3.8 27B 6M -Auto Free 4M +MiniMax-M2.7 6M -Auto 1M +Arcee Trinity Large Prev 5M -Command A Reasoning 800K +NavyAI free pool 5M -ERNIE 4.5 VL 424B A47B B 500K +Auto Free 4M -morph-v3-large 400K +Auto 1M -Llama 3.1 8B 200K +Command A Reasoning 800K -Claude Sonnet 4.5 25K - -+ First month: one-time signup credits (~626M) - -vertex 300M - -agentrouter 200M - -predibase 25M - -together 25M - -glm-cn 20M - -doubao 15M - -ai21 10M - -longcat 10M +ERNIE 4.5 VL 424B A47B B 500K + +morph-v3-large 400K + +Llama 3.1 8B 200K + +Claude Sonnet 4.5 25K + ++ First month: one-time signup credits (~626M) -deepseek 5M - -Pool-deduped, honest counting — no inflated rate-limit ceilings. Some terms suggest personal-use only; we flag them so you decide. -+ 12 permanently-free, no-cap providers (e.g. baidu, glm-cn, opencode-zen) · OpenRouter $10 → +24M/mo. +vertex 300M + +agentrouter 200M + +predibase 25M + +together 25M + +glm-cn 20M + +doubao 15M + +ai21 10M + +longcat 10M + +deepseek 5M + +Pool-deduped, honest counting — no inflated rate-limit ceilings. Some terms suggest personal-use only; we flag them so you decide. ++ 15 permanently-free, no-cap providers (e.g. agnes, ainative, aion) · OpenRouter $10 → +24M/mo. ++ ~6M behind regional identity verification (modelscope) — real quota, never in the headline. diff --git a/open-sse/config/freeModelCatalog.data.ts b/open-sse/config/freeModelCatalog.data.ts index 08b54e455d..c6b8b133a3 100644 --- a/open-sse/config/freeModelCatalog.data.ts +++ b/open-sse/config/freeModelCatalog.data.ts @@ -287,6 +287,14 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "mistral", modelId: "mistral-small-latest", displayName: "Mistral Small 4", monthlyTokens: 1000000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "mistral", tos: "caution" }, { provider: "mistral", modelId: "devstral-latest", displayName: "Devstral 2", monthlyTokens: 1000000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "mistral", tos: "caution" }, { provider: "mistral", modelId: "codestral-latest", displayName: "Codestral", monthlyTokens: 1000000000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "mistral", tos: "caution" }, + // evidence: public-page https://modelscope.cn/docs/model-service/API-Inference/limits and + // https://modelscope.cn/docs/magicube/intro (2026-09-02) — API-Inference is free; calls are paid with + // 魔粒: "注册并登录 200 魔粒/日" + "绑定阿里云账号 50 魔粒/日", 1 魔粒 per call on "主流" models ⇒ ~250 calls/day + // ⇒ 250 × 800 × 30 = 6M/month, one balance per account (single pool). + // eligibilityGate: "账号注册后需绑定阿里云账号,并且通过实名认证后才可使用" (Alibaba Cloud binding + mainland + // real-name verification). The docs also call the product "非商业化,非盈利" — hence tos: caution. + { provider: "modelscope", modelId: "Qwen/Qwen3.5-397B-A17B", displayName: "Qwen3.5 397B A17B (ModelScope)", monthlyTokens: 6000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "modelscope-free", tos: "caution", eligibilityGate: "regional-identity" }, + { provider: "modelscope", modelId: "deepseek-ai/DeepSeek-V4-Pro", displayName: "DeepSeek V4 Pro (ModelScope)", monthlyTokens: 6000000, creditTokens: 0, freeType: "recurring-daily", poolKey: "modelscope-free", tos: "caution", eligibilityGate: "regional-identity" }, { provider: "monsterapi", modelId: "llama-3-8b-fuse", displayName: "Llama 3 8B Fuse", monthlyTokens: 0, creditTokens: 0, freeType: "one-time-initial", poolKey: "monsterapi", tos: "ambiguous" }, { provider: "morph", modelId: "morph-v3-large", displayName: "morph-v3-large", monthlyTokens: 400000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "morph", tos: "ok" }, { provider: "morph", modelId: "morph-v3-fast", displayName: "morph-v3-fast", monthlyTokens: 400000, creditTokens: 0, freeType: "recurring-monthly", poolKey: "morph", tos: "ok" }, diff --git a/open-sse/config/freeModelCatalog.ts b/open-sse/config/freeModelCatalog.ts index 041e393430..aa32a36464 100644 --- a/open-sse/config/freeModelCatalog.ts +++ b/open-sse/config/freeModelCatalog.ts @@ -11,6 +11,13 @@ export type FreeModelFreeType = | "keyless" | "discontinued"; +/** + * A real, recurring quota that only opens after an identity check tied to a + * region (e.g. 实名认证 with a mainland-China ID). One member today; extend the + * union when a second kind of gate is catalogued. + */ +export type FreeEligibilityGate = "regional-identity"; + export interface FreeModelBudget { provider: string; modelId: string; @@ -40,14 +47,28 @@ export interface FreeModelBudget { * `open-sse/services/autoCombo/strictZeroCostFilter.ts`. */ hardStopGuaranteed?: boolean; + /** + * Set when the quota is real and recurring but only reachable after a + * region-bound identity verification. Affects COUNTING only: the row + * leaves the steady headline and lands in `gatedRecurringTokens`. + * Routing, `isFreeModel` and STRICT_ZERO_COST read `freeType` alone. + * Put the gate's source in a comment next to the entry. + */ + eligibilityGate?: FreeEligibilityGate; } export interface FreeModelTotals { /** Pool-deduped recurring tokens/month — the headline "steady" number. */ steadyRecurringTokens: number; - /** Steady + recurring credit grants (e.g. monthly $-credit plans). */ + /** + * Steady + recurring credit grants (e.g. monthly $-credit plans). + * Eligibility-gated rows contribute nothing, exactly like the steady headline. + */ steadyWithRecurringCreditsTokens: number; - /** Steady + recurring + one-time signup credits — first-month only. */ + /** + * Steady + recurring + one-time signup credits — first-month only. + * Eligibility-gated rows contribute nothing, exactly like the steady headline. + */ firstMonthRealisticTokens: number; /** * Extra recurring tokens/month unlocked by a one-time small deposit @@ -59,8 +80,16 @@ export interface FreeModelTotals { * Providers that are permanently free but publish NO token cap * (rate/concurrency-limited). Real access, but un-quantifiable — listed, * never summed into the headline (avoids the rate-limit×24/7 inflation). + * Eligibility-gated rows are excluded: the list reads as "open to anyone". */ uncappedProviders: string[]; + /** + * Pool-deduped tokens/month behind an eligibility gate (same rule as the + * headline). Never summed into `steadyRecurringTokens`. + */ + gatedRecurringTokens: number; + /** Providers (sorted) contributing to `gatedRecurringTokens`. */ + gatedProviders: string[]; modelCount: number; poolCount: number; perModel: FreeModelBudget[]; @@ -224,41 +253,60 @@ export function computeFreeModelTotals( (m) => !(opts.excludeTosAvoid && m.tos === "avoid") && m.enabled !== false ); + const isGated = (m: FreeModelBudget) => m.eligibilityGate !== undefined; + const steadyRecurringTokens = dedupedSum( models, (m) => m.monthlyTokens, - (m) => STEADY_MONTHLY.has(m.freeType) + (m) => STEADY_MONTHLY.has(m.freeType) && !isGated(m) ); + const gatedRecurringTokens = dedupedSum( + models, + (m) => m.monthlyTokens, + (m) => STEADY_MONTHLY.has(m.freeType) && isGated(m) + ); + const gatedProviders = [ + ...new Set( + models.filter((m) => STEADY_MONTHLY.has(m.freeType) && isGated(m)).map((m) => m.provider) + ), + ].sort(); const recurringCredits = dedupedSum( models, (m) => m.creditTokens, - (m) => RECURRING_CREDIT.has(m.freeType) + (m) => RECURRING_CREDIT.has(m.freeType) && !isGated(m) ); const oneTimeCredits = dedupedSum( models, (m) => m.creditTokens, - (m) => ONE_TIME_CREDIT.has(m.freeType) + (m) => ONE_TIME_CREDIT.has(m.freeType) && !isGated(m) ); const steadyWithRecurringCreditsTokens = steadyRecurringTokens + recurringCredits; const firstMonthRealisticTokens = steadyWithRecurringCreditsTokens + oneTimeCredits; const poolCount = new Set( - models.filter((m) => STEADY_MONTHLY.has(m.freeType) && m.poolKey).map((m) => m.poolKey) + models + .filter((m) => STEADY_MONTHLY.has(m.freeType) && m.poolKey && !isGated(m)) + .map((m) => m.poolKey) ).size; // Deposit-unlock boost: sum the FREE_TIER_BOOSTS whose pool still has a live // recurring model in the (optionally ToS-filtered) set. const livePools = new Set( - models.filter((m) => STEADY_MONTHLY.has(m.freeType) && m.poolKey).map((m) => m.poolKey) + models + .filter((m) => STEADY_MONTHLY.has(m.freeType) && m.poolKey && !isGated(m)) + .map((m) => m.poolKey) ); const boostMonthlyTokens = Object.entries(FREE_TIER_BOOSTS) .filter(([pool]) => livePools.has(pool)) .reduce((s, [, b]) => s + b.boostMonthlyTokens, 0); // Permanently-free-but-uncapped providers (real access, no published cap). + // Gated rows are excluded: the list is read as "anyone can use this, forever". const uncappedProviders = [ - ...new Set(models.filter((m) => UNCAPPED.has(m.freeType)).map((m) => m.provider)), + ...new Set( + models.filter((m) => UNCAPPED.has(m.freeType) && !isGated(m)).map((m) => m.provider) + ), ].sort(); return { @@ -267,6 +315,8 @@ export function computeFreeModelTotals( firstMonthRealisticTokens, boostMonthlyTokens, uncappedProviders, + gatedRecurringTokens, + gatedProviders, modelCount: models.length, poolCount, perModel: models.slice().sort((a, b) => b.monthlyTokens - a.monthlyTokens), diff --git a/scripts/check/check-docs-counts-sync.mjs b/scripts/check/check-docs-counts-sync.mjs index 29a42fca2a..4e74d30464 100644 --- a/scripts/check/check-docs-counts-sync.mjs +++ b/scripts/check/check-docs-counts-sync.mjs @@ -218,12 +218,16 @@ function readCodeFacts() { "const t=computeFreeModelTotals();const cli=Object.values(CLI_TOOLS);", "const by=(c)=>cli.filter(x=>x.category===c).length;", // "Free forever" = every provider whose free access renews or needs no key at all. - // one-time-initial (signup credits) and discontinued pools are excluded on purpose. + // one-time-initial (signup credits) and discontinued pools are excluded on purpose, + // and so is every eligibility-gated row: a provider nobody can sign up for without + // clearing a gate is not "free forever" for the reader of the headline. "const FOREVER=new Set(['recurring-monthly','recurring-daily','recurring-uncapped',", "'recurring-credit','keyless']);", - "const ff=new Set();for(const m of t.perModel)if(FOREVER.has(m.freeType))ff.add(m.provider);", + "const ff=new Set();for(const m of t.perModel)", + "if(FOREVER.has(m.freeType)&&!m.eligibilityGate)ff.add(m.provider);", 'console.log("@@"+JSON.stringify({freeSteady:t.steadyRecurringTokens,entries:t.perModel.length,', - "freeFirst:t.firstMonthRealisticTokens,freePools:t.poolCount,engines:ENGINE_IDS.length,", + "freeFirst:t.firstMonthRealisticTokens,freeGated:t.gatedRecurringTokens,", + "freePools:t.poolCount,engines:ENGINE_IDS.length,", "cliTotal:cli.length,cliCode:by('code'),cliAgent:by('agent'),", "mcpTools:countUniqueMcpTools(cols),mcpScopes:sc.size,providers:pids.size,freeForever:ff.size,", "modePacks:Object.keys(MODE_PACKS),", @@ -271,6 +275,20 @@ export function extractHeadlineClaims(content) { return claims; } +// The eligibility-gated figure ("+~6M behind regional identity verification") is validated +// with its own anchor so it can neither drift nor be silently dropped once it exists. +const GATED_ANCHOR = /^\s*behind regional identity verification/i; + +export function extractGatedClaims(content) { + const claims = []; + for (const m of content.matchAll(/\+?~?(\d+(?:\.\d+)?)([BM])\b/g)) { + const after = content.slice(m.index + m[0].length, m.index + m[0].length + 60); + if (!GATED_ANCHOR.test(after)) continue; + claims.push({ tokens: Number(m[1]) * (m[2] === "B" ? 1e9 : 1e6), unit: m[2], text: m[0] }); + } + return claims; +} + export function checkFreeTierHeadline(content, totals) { const claims = extractHeadlineClaims(content); if (!claims.length) return { ok: true, detail: "no aggregate free-tier headline in this file" }; @@ -279,14 +297,31 @@ export function checkFreeTierHeadline(content, totals) { const stale = claims.filter( (c) => Math.abs(c.value - steady) >= 0.05 && Math.abs(c.value - first) >= 0.05 ); - if (!stale.length) - return { ok: true, detail: `${claims.length} headline claim(s) match the live catalog` }; - return { - ok: false, - detail: + const problems = []; + if (stale.length) { + problems.push( `stale headline ${[...new Set(stale.map((c) => c.text))].join(", ")} — live catalog ` + - `computes ~${steady.toFixed(2)}B steady / ~${first.toFixed(2)}B first month`, - }; + `computes ~${steady.toFixed(2)}B steady / ~${first.toFixed(2)}B first month` + ); + } + if (totals.g != null && totals.g > 0) { + const gated = extractGatedClaims(content); + const tol = (c) => (c.unit === "B" ? 0.05e9 : 0.5e6); + const gatedStale = gated.filter((c) => Math.abs(c.tokens - totals.g) >= tol(c)); + if (!gated.length) { + problems.push( + `missing gated figure — live catalog computes ${Math.round(totals.g / 1e6)}M behind regional identity verification` + ); + } else if (gatedStale.length) { + problems.push( + `stale gated figure ${[...new Set(gatedStale.map((c) => c.text))].join(", ")} — live catalog ` + + `computes ${Math.round(totals.g / 1e6)}M behind regional identity verification` + ); + } + } + if (!problems.length) + return { ok: true, detail: `${claims.length} headline claim(s) match the live catalog` }; + return { ok: false, detail: problems.join("; ") }; } // PURE: docs prose that names the product version ("OmniRoute v3.8.50 ·", @@ -599,12 +634,12 @@ export function buildChecks() { }, { label: "Free-tier headline (live catalog)", - actual: `~${(f.freeSteady / 1e9).toFixed(2)}B steady / ${f.freePools} pools`, + actual: `~${(f.freeSteady / 1e9).toFixed(2)}B steady / ${f.freePools} pools / ${Math.round(f.freeGated / 1e6)}M gated`, docKey: "free-tier headline", strict: true, files: ["README.md", "docs/reference/FREE_TIERS.md"], validate: (content) => - checkFreeTierHeadline(content, { s: f.freeSteady, m: f.freeFirst }), + checkFreeTierHeadline(content, { s: f.freeSteady, m: f.freeFirst, g: f.freeGated }), }, claim( f.engines, diff --git a/scripts/research/gen-budget-card-svg.mjs b/scripts/research/gen-budget-card-svg.mjs index f253ded074..f2c01b082c 100644 --- a/scripts/research/gen-budget-card-svg.mjs +++ b/scripts/research/gen-budget-card-svg.mjs @@ -1,56 +1,81 @@ -// Generates docs/screenshots/free-tier-budget-card.svg from the per-model catalog. -// Run: node scripts/research/gen-budget-card-svg.mjs +#!/usr/bin/env node +// Generates the free-tier budget card from the per-model catalog, through the +// same function the docs gate and the dashboard use — never by parsing the data +// file with a regex (that silently skipped every row carrying an extra field). +// Run from the repo root: +// node --import tsx/esm scripts/research/gen-budget-card-svg.mjs [--out path.svg] import fs from "node:fs"; +import { computeFreeModelTotals } from "../../open-sse/config/freeModelCatalog.ts"; -const txt = fs.readFileSync("open-sse/config/freeModelCatalog.data.ts", "utf8"); -const recs = [ - ...txt.matchAll( - /\{ provider: "([^"]+)", modelId: "([^"]+)", displayName: "([^"]+)", monthlyTokens: (\d+), creditTokens: (\d+), freeType: "([^"]+)", poolKey: (null|"[^"]+"), tos: "([^"]+)" \}/g - ), -].map((m) => ({ - provider: m[1], - modelId: m[2], - displayName: m[3], - monthlyTokens: +m[4], - creditTokens: +m[5], - freeType: m[6], - poolKey: m[7] === "null" ? null : m[7].slice(1, -1), - tos: m[8], -})); +const outIdx = process.argv.indexOf("--out"); +if (outIdx >= 0 && !process.argv[outIdx + 1]) throw new Error("--out requires a path"); +const OUT = outIdx >= 0 ? process.argv[outIdx + 1] : "docs/screenshots/free-tier-budget-card.svg"; +const t = computeFreeModelTotals(); +const STEADY_TYPES = new Set(["recurring-daily", "recurring-monthly", "keyless"]); const fmt = (n) => - n >= 1e9 ? (n / 1e9).toFixed(2) + "B" : n >= 1e6 ? Math.round(n / 1e6) + "M" : Math.round(n / 1e3) + "K"; + n >= 1e9 + ? (n / 1e9).toFixed(2) + "B" + : n >= 1e6 + ? Math.round(n / 1e6) + "M" + : Math.round(n / 1e3) + "K"; +// One bar segment per steady pool (largest member), gated rows excluded like the headline. const poolMap = new Map(); -for (const r of recs) { - if (!["recurring-daily", "recurring-monthly", "keyless"].includes(r.freeType)) continue; +for (const r of t.perModel) { + if (!STEADY_TYPES.has(r.freeType) || r.eligibilityGate) continue; const k = r.poolKey || `${r.provider}:${r.modelId}`; const cur = poolMap.get(k); if (!cur || r.monthlyTokens > cur.monthlyTokens) poolMap.set(k, r); } -const pools = [...poolMap.values()].filter((r) => r.monthlyTokens > 0).sort((a, b) => b.monthlyTokens - a.monthlyTokens); -const steady = pools.reduce((s, r) => s + r.monthlyTokens, 0); +const pools = [...poolMap.values()] + .filter((r) => r.monthlyTokens > 0) + .sort((a, b) => b.monthlyTokens - a.monthlyTokens); +const steady = t.steadyRecurringTokens; +const firstMonth = t.firstMonthRealisticTokens; +const gated = t.gatedRecurringTokens; const otMap = new Map(); -for (const r of recs) { - if (r.freeType !== "one-time-initial" || r.creditTokens <= 0) continue; +for (const r of t.perModel) { + // Gated rows are excluded here too — they are absent from firstMonthRealisticTokens. + if (r.freeType !== "one-time-initial" || r.creditTokens <= 0 || r.eligibilityGate) continue; const k = r.poolKey || r.provider; otMap.set(k, { provider: r.provider, v: Math.max(otMap.get(k)?.v || 0, r.creditTokens) }); } const oneTime = [...otMap.values()].sort((a, b) => b.v - a.v); const oneTimeSum = oneTime.reduce((s, r) => s + r.v, 0); -const firstMonth = steady + oneTimeSum; -const avoidProviders = [...new Set(recs.filter((r) => r.tos === "avoid").map((r) => r.provider))].length; -const uncappedProviders = [...new Set(recs.filter((r) => r.freeType === "recurring-uncapped").map((r) => r.provider))]; +const avoidProviders = new Set(t.perModel.filter((r) => r.tos === "avoid").map((r) => r.provider)) + .size; +const uncappedProviders = t.uncappedProviders; const GRID = pools.slice(0, 28); const STRIP = oneTime.slice(0, 9); -const PAL = ["#6c5ce7","#00b894","#0984e3","#e17055","#fdcb6e","#e84393","#00cec9","#d63031","#a29bfe","#55efc4","#74b9ff","#ffeaa7","#fab1a0","#81ecec"]; +const PAL = [ + "#6c5ce7", + "#00b894", + "#0984e3", + "#e17055", + "#fdcb6e", + "#e84393", + "#00cec9", + "#d63031", + "#a29bfe", + "#55efc4", + "#74b9ff", + "#ffeaa7", + "#fab1a0", + "#81ecec", +]; const color = (i) => PAL[i % PAL.length]; -const cleanName = (r) => (r.displayName || r.provider).replace(/\s*\(.*$/, "").replace(/ —.*$/, "").slice(0, 24); +const cleanName = (r) => + (r.displayName || r.provider) + .replace(/\s*\(.*$/, "") + .replace(/ —.*$/, "") + .slice(0, 24); -// bar segments (min width so every pool shows) -const BAR_X = 32, BAR_W = 836, MIN = 7; +const BAR_X = 32, + BAR_W = 836, + MIN = 7; const extra = BAR_W - MIN * GRID.length; let bx = BAR_X; const segs = GRID.map((r, i) => { @@ -60,11 +85,13 @@ const segs = GRID.map((r, i) => { return s; }); -const B = []; // body elements -// title -B.push(`Monthly free-token budget`); -B.push(`${pools.length} free pools · ${recs.length} models · one endpoint`); -// stats +const B = []; +B.push( + `Monthly free-token budget` +); +B.push( + `${pools.length} free pools · ${t.modelCount} models · one endpoint` +); const stat = (sx, label, val, vc) => { B.push(`${label}`); B.push(`${val}`); @@ -72,50 +99,90 @@ const stat = (sx, label, val, vc) => { stat(32, "Steady / month", `~${fmt(steady)}`, "#e6edf3"); stat(330, "First month (+ signup credits)", `~${fmt(firstMonth)}`, "#3fb950"); stat(700, "ToS-flagged (you decide)", `${avoidProviders} providers`, "#d29922"); -// bar -B.push(``); -B.push(``); -for (const s of segs) B.push(``); +B.push( + `` +); +B.push( + `` +); +for (const s of segs) + B.push( + `` + ); B.push(``); -B.push(`Each segment = one free pool · widths floored so every provider shows · honest numbers in the grid.`); -// model grid 4 cols -const COLS = 4, COLW = 213, GX = 32, GY = 200, RH = 30; +B.push( + `Each segment = one free pool · widths floored so every provider shows · honest numbers in the grid.` +); +const COLS = 4, + COLW = 213, + GX = 32, + GY = 200, + RH = 30; GRID.forEach((r, i) => { - const col = i % COLS, row = (i / COLS) | 0; - const cx = GX + col * COLW, cy = GY + row * RH; + const col = i % COLS, + row = (i / COLS) | 0; + const cx = GX + col * COLW, + cy = GY + row * RH; B.push(``); - B.push(`${cleanName(r)} ${fmt(r.monthlyTokens)}`); + B.push( + `${cleanName(r)} ${fmt(r.monthlyTokens)}` + ); }); let y = GY + Math.ceil(GRID.length / COLS) * RH + 6; -// first-month strip (wrapping) B.push(``); y += 26; -B.push(`+ First month: one-time signup credits (~${fmt(oneTimeSum)})`); +B.push( + `+ First month: one-time signup credits (~${fmt(oneTimeSum)})` +); y += 24; let sxp = 32; for (const r of STRIP) { const label = `${r.provider} ${fmt(r.v)}`; const w = 16 + label.length * 6.7; - if (sxp + w > 862) { sxp = 32; y += 30; } - B.push(``); - B.push(`${label}`); + if (sxp + w > 862) { + sxp = 32; + y += 30; + } + B.push( + `` + ); + B.push( + `${label}` + ); sxp += w + 8; } y += 26; -// ToS note (softened) -B.push(``); -B.push(`Pool-deduped, honest counting — no inflated rate-limit ceilings. Some terms suggest personal-use only; we flag them so you decide.`); -B.push(`+ ${uncappedProviders.length} permanently-free, no-cap providers (e.g. ${uncappedProviders.slice(0, 3).join(", ")}) · OpenRouter $10 → +24M/mo.`); -y += 34; -const H = y + 24; // card content bottom +const noteH = gated > 0 ? 48 : 34; +B.push( + `` +); +B.push( + `Pool-deduped, honest counting — no inflated rate-limit ceilings. Some terms suggest personal-use only; we flag them so you decide.` +); +B.push( + `+ ${uncappedProviders.length} permanently-free, no-cap providers (e.g. ${uncappedProviders.slice(0, 3).join(", ")}) · OpenRouter $10 → +${fmt(t.boostMonthlyTokens)}/mo.` +); +if (gated > 0) { + B.push( + `+ ~${fmt(gated)} behind regional identity verification (${t.gatedProviders.join(", ")}) — real quota, never in the headline.` + ); +} +y += noteH; +const H = y + 24; const CANVAS = H + 16; const out = []; -out.push(``); +out.push( + `` +); out.push(``); out.push(``); -out.push(`OmniRoute · /dashboard/free-tiers · preview mockup`); +out.push( + `OmniRoute · /dashboard/free-tiers · preview mockup` +); out.push(...B); out.push(``); -fs.writeFileSync("docs/screenshots/free-tier-budget-card.svg", out.join("\n") + "\n"); -console.log(`SVG: ${GRID.length} models, ${STRIP.length} first-month chips, canvas ${CANVAS}px. steady=${fmt(steady)} firstMonth=${fmt(firstMonth)} oneTime=${fmt(oneTimeSum)}`); +fs.writeFileSync(OUT, out.join("\n") + "\n"); +console.log( + `SVG → ${OUT}: ${GRID.length} pools, ${STRIP.length} first-month chips, canvas ${CANVAS}px. steady=${fmt(steady)} firstMonth=${fmt(firstMonth)} gated=${fmt(gated)} oneTime=${fmt(oneTimeSum)}` +); diff --git a/src/app/(dashboard)/dashboard/usage/components/FreeBudgetCard.tsx b/src/app/(dashboard)/dashboard/usage/components/FreeBudgetCard.tsx index 4fd8dff8ed..4e5ecaadca 100644 --- a/src/app/(dashboard)/dashboard/usage/components/FreeBudgetCard.tsx +++ b/src/app/(dashboard)/dashboard/usage/components/FreeBudgetCard.tsx @@ -33,6 +33,10 @@ export interface FreeBudgetData { boostMonthlyTokens?: number; /** Providers that are permanently free but publish no token cap (rate/concurrency-limited). */ uncappedProviders?: string[]; + /** Pool-deduped tokens/mo behind a regional identity check — real quota, never in the headline. */ + gatedRecurringTokens?: number; + /** Providers behind that check. */ + gatedProviders?: string[]; headline?: string; /** ISO timestamp of the last catalog update. Absent/null → freshness is not shown. */ catalogUpdatedAt?: string | null; @@ -88,6 +92,7 @@ interface FreeBudgetLabels { segmentHint: string; boost: (tokens: string) => string; uncapped: string; + gated: (tokens: string) => string; tosRestricted: (count: number) => string; provider: string; model: string; @@ -111,6 +116,8 @@ const DEFAULT_LABELS: FreeBudgetLabels = { `Unlock ~${tokens} more/mo with a one-time $10 OpenRouter top-up (50 → 1000 req/day)`, uncapped: "Permanently free, no published cap (rate-limited) — real access, not counted in the headline:", + gated: (tokens) => + `~${tokens}/mo more behind a regional identity check — real quota, not counted in the headline:`, tosRestricted: (count) => `${count} model${count === 1 ? "" : "s"} flagged as ToS-restricted — you decide`, provider: "Provider", @@ -339,6 +346,8 @@ export function FreeBudgetView({ perModel, boostMonthlyTokens = 0, uncappedProviders = [], + gatedRecurringTokens = 0, + gatedProviders = [], catalogUpdatedAt, noCredentialProviders = [], } = data; @@ -429,9 +438,7 @@ export function FreeBudgetView({ lock_open - - {labels.noApiKey} - + {labels.noApiKey} ({keylessModels.length}个模型 · {keylessProviders.length}个提供者) @@ -475,6 +482,23 @@ export function FreeBudgetView({
)} + {gatedRecurringTokens > 0 && ( +
+ + {labels.gated(fmt(gatedRecurringTokens))} + +
+ {gatedProviders.map((p) => ( + + {p} + + ))} +
+
+ )} {/* ToS-restricted callout */} {avoidModels.length > 0 && ( @@ -669,6 +693,7 @@ export default function FreeBudgetCard() { segmentHint: t("segmentHint"), boost: (tokens) => t("boost", { tokens }), uncapped: t("uncapped"), + gated: (tokens) => t("gated", { tokens }), tosRestricted: (count) => t("tosRestricted", { count }), provider: t("provider"), model: t("model"), diff --git a/src/app/api/free-tier/summary/route.ts b/src/app/api/free-tier/summary/route.ts index 52c38f8673..f391634f53 100644 --- a/src/app/api/free-tier/summary/route.ts +++ b/src/app/api/free-tier/summary/route.ts @@ -46,6 +46,7 @@ function toBudgetEntry(entry: MergedEntry): FreeModelBudget & { enabled?: boolea poolKey: entry.poolKey, tos: entry.tos, trainsOnPrompts: entry.trainsOnPrompts, + eligibilityGate: entry.eligibilityGate, hardStopGuaranteed: HARD_STOP_BY_KEY.get(`${entry.provider}:${entry.modelId}`), enabled: entry.enabled, }; diff --git a/src/i18n/messages/ar.json b/src/i18n/messages/ar.json index dcaaed82f8..47efa8dc95 100644 --- a/src/i18n/messages/ar.json +++ b/src/i18n/messages/ar.json @@ -13133,6 +13133,7 @@ "segmentHint": "كل شريحة = مجمع مجاني واحد · مجمع مفرود من التكرار، عدّ صادق (بدون حدود قصوى مضخمة لمعدل الطلبات).", "boost": "افتح حوالي {tokens} إضافية/شهرياً بشحن رصيد OpenRouter لمرة واحدة بقيمة 10$ (50 ← 1000 طلب/يوم)", "uncapped": "مجاني بشكل دائم، بدون حد أقصى معلن (محدود بمعدل الطلبات) — وصول حقيقي، لا يُحتسب في العنوان الرئيسي:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# نموذج تم وضع علامة عليه كمقيد بشروط الخدمة} other {# نماذج تم وضع علامة عليها كمقيدة بشروط الخدمة}} — القرار لك", "provider": "المزود", "model": "النموذج", diff --git a/src/i18n/messages/az.json b/src/i18n/messages/az.json index e13bd45903..9f9efb9640 100644 --- a/src/i18n/messages/az.json +++ b/src/i18n/messages/az.json @@ -13133,6 +13133,7 @@ "segmentHint": "Hər seqment = bir pulsuz hovuz · hovuz üzrə təkrarlanmayan, dürüst sayım (şişirdilmiş sorğu limiti tavanları olmadan).", "boost": "Birdəfəlik $10 OpenRouter balans artımı ilə ayda ~{tokens} daha çox əldə edin (50 → 1000 sorğu/gün)", "uncapped": "Həmişəlik pulsuz, dərc edilmiş limit yoxdur (sorğu sayı məhdudlaşdırılıb) — real giriş, əsas göstəricidə sayılmır:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# model} other {# model}} ToS ilə məhdudlaşdırılmış kimi qeyd edilib — qərar sizindir", "provider": "Provayder", "model": "Model", diff --git a/src/i18n/messages/bg.json b/src/i18n/messages/bg.json index f8662e5db3..e442c29e0d 100644 --- a/src/i18n/messages/bg.json +++ b/src/i18n/messages/bg.json @@ -13133,6 +13133,7 @@ "segmentHint": "Всеки сегмент = един безплатен пул · дедупликиран пул, честно отчитане (без изкуствено завишени тавани на лимитите за скорост).", "boost": "Отключете още ~{tokens}/месец с еднократно допълване от $10 в OpenRouter (50 → 1000 заявки/ден)", "uncapped": "Постоянно безплатно, без публикуван лимит (с ограничение на скоростта) — реален достъп, който не се отчита в заглавието:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# модел е маркиран като ограничен от Условията за ползване — вие решавате} other {# модела са маркирани като ограничени от Условията за ползване — вие решавате}}", "provider": "Доставчик", "model": "Модел", diff --git a/src/i18n/messages/bn.json b/src/i18n/messages/bn.json index 0b768593ef..fb1e088781 100644 --- a/src/i18n/messages/bn.json +++ b/src/i18n/messages/bn.json @@ -13133,6 +13133,7 @@ "segmentHint": "প্রতিটি সেগমেন্ট = একটি ফ্রি পুল · পুল-ডিডুপ্লিকেটেড, সঠিক গণনা (কোনো অতিরঞ্জিত রেট-লিমিট সিলিং নেই)।", "boost": "এককালীন $10 OpenRouter টপ-আপের মাধ্যমে প্রতি মাসে আরও ~{tokens} আনলক করুন (50 → 1000 req/day)", "uncapped": "স্থায়ীভাবে বিনামূল্যে, কোনো প্রকাশিত সীমা নেই (রেট-সীমিত) — প্রকৃত অ্যাক্সেস, হেডলাইনে গণনা করা হয়নি:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {#টি মডেল} other {#টি মডেল}} ToS-সীমাবদ্ধ হিসেবে চিহ্নিত — সিদ্ধান্ত আপনার", "provider": "প্রোভাইডার", "model": "মডেল", diff --git a/src/i18n/messages/cs.json b/src/i18n/messages/cs.json index ec3e8ce094..14bd25e496 100644 --- a/src/i18n/messages/cs.json +++ b/src/i18n/messages/cs.json @@ -13133,6 +13133,7 @@ "segmentHint": "Každý segment = jeden bezplatný pool · deduplikovaný pool, poctivé počítání (žádné uměle navýšené stropy limitů).", "boost": "Odemkněte o ~{tokens} více/měs. jednorázovým dobitím $10 na OpenRouteru (50 → 1000 požadavků/den)", "uncapped": "Trvale zdarma, bez zveřejněného limitu (omezená rychlost) — reálný přístup, nepočítá se do hlavního přehledu:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# model označený jako omezený ToS} few {# modely označené jako omezené ToS} other {# modelů označených jako omezené ToS}} — rozhodnutí je na vás", "provider": "Poskytovatel", "model": "Model", diff --git a/src/i18n/messages/da.json b/src/i18n/messages/da.json index 9587febd10..7309787fd2 100644 --- a/src/i18n/messages/da.json +++ b/src/i18n/messages/da.json @@ -13133,6 +13133,7 @@ "segmentHint": "Hvert segment = én gratis pulje · pulje-dedupliceret, ærlig optælling (ingen oppustede hastighedsgrænselofter).", "boost": "Lås op for ~{tokens} mere/md. med en engangsoptankning på $10 hos OpenRouter (50 → 1000 anm./dag)", "uncapped": "Permanent gratis, intet offentliggjort loft (hastighedsbegrænset) — reel adgang, ikke talt med i overskriften:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# model} other {# modeller}} markeret som ToS-begrænset — du bestemmer", "provider": "Udbyder", "model": "Model", diff --git a/src/i18n/messages/de.json b/src/i18n/messages/de.json index d233f0dba4..41ea050e7b 100644 --- a/src/i18n/messages/de.json +++ b/src/i18n/messages/de.json @@ -13140,6 +13140,7 @@ "segmentHint": "Jedes Segment = ein kostenloser Pool · Pool-dedupliziert, ehrliche Zählung (keine künstlich erhöhten Rate-Limit-Obergrenzen).", "boost": "Schalten Sie ~{tokens} mehr/Monat mit einer einmaligen $10 OpenRouter-Aufladung frei (50 → 1000 Anfr./Tag)", "uncapped": "Dauerhaft kostenlos, kein veröffentlichtes Limit (ratenbegrenzt) — echter Zugang, nicht in der Überschrift gezählt:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# Modell} other {# Modelle}} als ToS-eingeschränkt markiert — Sie entscheiden", "provider": "Anbieter", "model": "Modell", diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index 88ba6a9bdc..ae4b3b28fe 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -13143,6 +13143,7 @@ "segmentHint": "Each segment = one free pool · pool-deduped, honest counting (no inflated rate-limit ceilings).", "boost": "Unlock ~{tokens} more/mo with a one-time $10 OpenRouter top-up (50 → 1000 req/day)", "uncapped": "Permanently free, no published cap (rate-limited) — real access, not counted in the headline:", + "gated": "~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# model} other {# models}} flagged as ToS-restricted — you decide", "provider": "Provider", "model": "Model", diff --git a/src/i18n/messages/es.json b/src/i18n/messages/es.json index 24dd42291a..348db0b960 100644 --- a/src/i18n/messages/es.json +++ b/src/i18n/messages/es.json @@ -13133,6 +13133,7 @@ "segmentHint": "Each segment = one free pool · pool-deduped, honest counting (no inflated rate-limit ceilings).", "boost": "Unlock ~{tokens} more/mo with a one-time $10 OpenRouter top-up (50 → 1000 req/day)", "uncapped": "Permanently free, no published cap (rate-limited) — real access, not counted in the headline:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# model} other {# models}} flagged as ToS-restricted — you decide", "provider": "Provider", "model": "Model", diff --git a/src/i18n/messages/fa.json b/src/i18n/messages/fa.json index 1b643eed99..0fa364611b 100644 --- a/src/i18n/messages/fa.json +++ b/src/i18n/messages/fa.json @@ -13133,6 +13133,7 @@ "segmentHint": "هر بخش = یک استخر رایگان · حذف تکرار استخر، شمارش واقعی (بدون سقف‌های محدودیت نرخ کاذب).", "boost": "با یک‌بار شارژ ۱۰ دلاری OpenRouter، حدود ~{tokens} بیشتر در ماه آزاد کنید (۵۰ → ۱۰۰۰ درخواست/روز)", "uncapped": "دائماً رایگان، بدون سقف اعلام‌شده (دارای محدودیت نرخ) — دسترسی واقعی، بدون احتساب در عنوان اصلی:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# مدل} other {# مدل}} دارای محدودیت ToS علامت‌گذاری شده‌اند — تصمیم با شماست", "provider": "ارائه‌دهنده", "model": "مدل", diff --git a/src/i18n/messages/fi.json b/src/i18n/messages/fi.json index ea6c058f9a..e5a09ea7dc 100644 --- a/src/i18n/messages/fi.json +++ b/src/i18n/messages/fi.json @@ -13133,6 +13133,7 @@ "segmentHint": "Jokainen segmentti = yksi ilmainen pooli · poolikohtaisesti duplikaatit poistettu, rehellinen laskenta (ei paisutettuja nopeusrajoituskattoja).", "boost": "Avaa ~{tokens} lisää/kk kertaluonteisella 10 dollarin OpenRouter-latauksella (50 → 1000 pyyntöä/päivä)", "uncapped": "Pysyvästi ilmainen, ei julkaistua ylärajaa (nopeusrajoitettu) — todellinen käyttöoikeus, ei lasketa pääotsikkoon:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# malli} other {# mallia}} merkitty käyttöehtojen vastaiseksi — sinä päätät", "provider": "Tarjoaja", "model": "Malli", diff --git a/src/i18n/messages/fr.json b/src/i18n/messages/fr.json index b1dc36a641..48bcdae45a 100644 --- a/src/i18n/messages/fr.json +++ b/src/i18n/messages/fr.json @@ -13133,6 +13133,7 @@ "segmentHint": "Chaque segment = un pool gratuit · dédoublonné par pool, décompte honnête (sans plafonds de limite de débit gonflés).", "boost": "Débloquez ~{tokens} de plus/mois avec une recharge unique de 10 $ sur OpenRouter (50 → 1000 req/jour)", "uncapped": "Gratuit en permanence, sans plafond publié (limité en débit) — accès réel, non comptabilisé dans le total :", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# modèle} other {# modèles}} signalés comme restreints par les CGU — à vous de décider", "provider": "Fournisseur", "model": "Modèle", diff --git a/src/i18n/messages/gu.json b/src/i18n/messages/gu.json index b8f03d0f23..c9014de70b 100644 --- a/src/i18n/messages/gu.json +++ b/src/i18n/messages/gu.json @@ -13133,6 +13133,7 @@ "segmentHint": "દરેક સેગમેન્ટ = એક મફત પૂલ · પૂલ-ડિડુપ્લિકેટ, પ્રમાણિક ગણતરી (કોઈ ફૂલેલી રેટ-મર્યાદા સીમાઓ નહીં).", "boost": "એક વખતના $10 OpenRouter ટોપ-અપ સાથે દર મહિને ~{tokens} વધુ અનલૉક કરો (50 → 1000 req/day)", "uncapped": "કાયમી ધોરણે મફત, કોઈ પ્રકાશિત મર્યાદા નથી (રેટ-મર્યાદિત) — વાસ્તવિક ઍક્સેસ, હેડલાઇનમાં ગણવામાં આવતી નથી:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# મોડેલ} other {# મોડેલો}} ToS-પ્રતિબંધિત તરીકે ચિહ્નિત થયેલ છે — તમે નક્કી કરો", "provider": "પ્રદાતા", "model": "મોડેલ", diff --git a/src/i18n/messages/he.json b/src/i18n/messages/he.json index cb5619f767..8305333bb0 100644 --- a/src/i18n/messages/he.json +++ b/src/i18n/messages/he.json @@ -13133,6 +13133,7 @@ "segmentHint": "כל מקטע = מאגר חינמי אחד · מניעת כפילויות במאגר, ספירה הוגנת (ללא תקרות מגבלת קצב מנופחות).", "boost": "פתחו עוד כ-{tokens}/חודש עם טעינה חד-פעמית של $10 ב-OpenRouter ‏(50 → 1000 בקשות/יום)", "uncapped": "חינם לצמיתות, ללא מגבלה מפורסמת (מוגבל בקצב) — גישה אמיתית, לא נספר בכותרת הראשית:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {מודל #} other {# מודלים}} מסומנים כחסומים לפי תנאי השימוש — ההחלטה בידיך", "provider": "ספק", "model": "מודל", diff --git a/src/i18n/messages/hi.json b/src/i18n/messages/hi.json index 3cc6c5038a..4a7226cbfe 100644 --- a/src/i18n/messages/hi.json +++ b/src/i18n/messages/hi.json @@ -13133,6 +13133,7 @@ "segmentHint": "प्रत्येक सेगमेंट = एक मुफ़्त पूल · पूल-डीडुप्लिकेटेड, सटीक गणना (कोई बढ़ी हुई रेट-लिमिट सीमा नहीं)।", "boost": "एकमुश्त $10 OpenRouter टॉप-अप के साथ ~{tokens} अधिक/माह अनलॉक करें (50 → 1000 req/day)", "uncapped": "स्थायी रूप से मुफ़्त, कोई प्रकाशित सीमा नहीं (रेट-लिमिटेड) — वास्तविक एक्सेस, हेडलाइन में नहीं गिना गया:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# मॉडल} other {# मॉडल}} ToS-प्रतिबंधित के रूप में चिह्नित — आप तय करें", "provider": "प्रदाता", "model": "मॉडल", diff --git a/src/i18n/messages/hu.json b/src/i18n/messages/hu.json index abcd906753..99d95becd5 100644 --- a/src/i18n/messages/hu.json +++ b/src/i18n/messages/hu.json @@ -13133,6 +13133,7 @@ "segmentHint": "Each segment = one free pool · pool-deduped, honest counting (no inflated rate-limit ceilings).", "boost": "Oldjon fel további ~{tokens}/hó-t egy egyszeri 10 dolláros OpenRouter feltöltéssel (50 → 1000 req/day)", "uncapped": "Tartósan ingyenes, nincs közzétett korlát (sebességkorlátozott) — valós hozzáférés, nem számít bele a főcímbe:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# modell} other {# modell}} ÁSZF-korlátozottként megjelölve — Ön dönt", "provider": "Provider", "model": "Model", diff --git a/src/i18n/messages/id.json b/src/i18n/messages/id.json index 62a822f6f0..6ac4539837 100644 --- a/src/i18n/messages/id.json +++ b/src/i18n/messages/id.json @@ -13133,6 +13133,7 @@ "segmentHint": "Setiap segmen = satu pool gratis · deduplikasi pool, penghitungan jujur (tanpa batas rate-limit yang digelembungkan).", "boost": "Buka ~{tokens} tambahan/bln dengan top-up OpenRouter $10 satu kali (50 → 1000 req/hari)", "uncapped": "Gratis permanen, tanpa batas yang dipublikasikan (dibatasi rate-limit) — akses nyata, tidak dihitung dalam tajuk utama:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# model} other {# model}} ditandai sebagai dibatasi ToS — Anda yang menentukan", "provider": "Penyedia", "model": "Model", diff --git a/src/i18n/messages/it.json b/src/i18n/messages/it.json index 8c8d28641c..7bf00a4c77 100644 --- a/src/i18n/messages/it.json +++ b/src/i18n/messages/it.json @@ -13133,6 +13133,7 @@ "segmentHint": "Ogni segmento = un pool gratuito · pool deduplicato, conteggio onesto (nessun limite massimo di rate-limit gonfiato).", "boost": "Sblocca ~{tokens} in più al mese con una ricarica una tantum di $10 su OpenRouter (50 → 1000 rich/giorno)", "uncapped": "Permanentemente gratuito, nessun limite pubblicato (soggetto a rate-limit) — accesso reale, non conteggiato nel titolo:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# modello contrassegnato come limitato dai ToS} other {# modelli contrassegnati come limitati dai ToS}} — decidi tu", "provider": "Provider", "model": "Modello", diff --git a/src/i18n/messages/ja.json b/src/i18n/messages/ja.json index 2281cdb29e..0948496009 100644 --- a/src/i18n/messages/ja.json +++ b/src/i18n/messages/ja.json @@ -13133,6 +13133,7 @@ "segmentHint": "各セグメント = 1つの無料プール · プール重複排除、誠実なカウント(誇張されたレート制限上限なし)。", "boost": "1回限りの$10のOpenRouterチャージで、月あたりさらに約{tokens}をアンロック(50 → 1000 リクエスト/日)", "uncapped": "恒久的に無料、公開された上限なし(レート制限あり) — 見出しにはカウントされない、実際のアクセス:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# 個のモデル} other {# 個のモデル}}が利用規約制限としてフラグ立てされています — ご自身で判断してください", "provider": "プロバイダー", "model": "モデル", diff --git a/src/i18n/messages/ko.json b/src/i18n/messages/ko.json index 8fe59e947e..63d0b70434 100644 --- a/src/i18n/messages/ko.json +++ b/src/i18n/messages/ko.json @@ -13133,6 +13133,7 @@ "segmentHint": "각 세그먼트 = 하나의 무료 풀 · 풀 중복 제거, 정직한 집계 (부풀려진 요율 제한 한도 없음).", "boost": "일회성 $10 OpenRouter 충전으로 월 약 {tokens}개 추가 잠금 해제 (일일 50 → 1000회 요청)", "uncapped": "영구 무료, 공개된 제한 없음 (요율 제한됨) — 헤드라인에 집계되지 않는 실제 액세스:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {#개 모델} other {#개 모델}}이 이용약관(ToS) 제한으로 표시됨 — 귀하가 결정하세요", "provider": "제공업체", "model": "모델", diff --git a/src/i18n/messages/mr.json b/src/i18n/messages/mr.json index 0e41c94813..140afbdda6 100644 --- a/src/i18n/messages/mr.json +++ b/src/i18n/messages/mr.json @@ -13133,6 +13133,7 @@ "segmentHint": "प्रत्येक विभाग = एक मोफत पूल · पूल-डिड्युप केलेले, प्रामाणिक मोजणी (कोणतीही फुगवलेली रेट-लिमिट कमाल मर्यादा नाही).", "boost": "एकवेळच्या $10 OpenRouter टॉप-अपसह आणखी ~{tokens}/महिना अनलॉक करा (50 → 1000 req/day)", "uncapped": "कायमस्वरूपी मोफत, कोणतीही प्रकाशित मर्यादा नाही (रेट-लिमिटेड) — खरा ॲक्सेस, हेडलाइनमध्ये मोजला जात नाही:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# मॉडेल} other {# मॉडेल्स}} ToS-प्रतिबंधित म्हणून चिन्हांकित — तुम्ही ठरवा", "provider": "प्रदाता", "model": "मॉडेल", diff --git a/src/i18n/messages/ms.json b/src/i18n/messages/ms.json index 23b56c84a1..66e04377f7 100644 --- a/src/i18n/messages/ms.json +++ b/src/i18n/messages/ms.json @@ -13133,6 +13133,7 @@ "segmentHint": "Setiap segmen = satu kolam percuma · kolam dinyahduplikasi, pengiraan jujur (tiada siling had kadar yang melambung).", "boost": "Nyahkunci ~{tokens} lagi/bln dengan tambah nilai OpenRouter $10 sekali sahaja (50 → 1000 perm/hari)", "uncapped": "Percuma selama-lamanya, tiada had diterbitkan (had kadar dikenakan) — akses sebenar, tidak dikira dalam tajuk utama:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# model} other {# model}} ditandakan sebagai disekat ToS — anda tentukan", "provider": "Penyedia", "model": "Model", diff --git a/src/i18n/messages/nl.json b/src/i18n/messages/nl.json index 0f2f81bbea..b16e43b7d8 100644 --- a/src/i18n/messages/nl.json +++ b/src/i18n/messages/nl.json @@ -13133,6 +13133,7 @@ "segmentHint": "Elk segment = één gratis pool · pool-ontdubbeld, eerlijke telling (geen opgeblazen rate-limit-plafonds).", "boost": "Ontgrendel ~{tokens} extra/mnd met een eenmalige OpenRouter-opwaardering van $10 (50 → 1000 req/dag)", "uncapped": "Permanent gratis, geen gepubliceerde limiet (rate-limited) — echte toegang, niet meegeteld in de kop:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# model} other {# modellen}} gemarkeerd als ToS-beperkt — jij bepaalt", "provider": "Provider", "model": "Model", diff --git a/src/i18n/messages/no.json b/src/i18n/messages/no.json index 371319f17a..3e0aa602a7 100644 --- a/src/i18n/messages/no.json +++ b/src/i18n/messages/no.json @@ -13133,6 +13133,7 @@ "segmentHint": "Hvert segment = én gratis pool · pool-deduplisert, ærlig telling (ingen oppblåste tak for hastighetsbegrensning).", "boost": "Lås opp ~{tokens} mer/mnd med en engangs $10 OpenRouter-påfylling (50 → 1000 forespørsler/dag)", "uncapped": "Permanent gratis, ingen publisert grense (hastighetsbegrenset) — reell tilgang, ikke talt med i overskriften:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# modell} other {# modeller}} flagget som ToS-begrenset — du bestemmer", "provider": "Leverandør", "model": "Modell", diff --git a/src/i18n/messages/phi.json b/src/i18n/messages/phi.json index 04fd631ba1..282dc3c6f2 100644 --- a/src/i18n/messages/phi.json +++ b/src/i18n/messages/phi.json @@ -13133,6 +13133,7 @@ "segmentHint": "Bawat segment = isang libreng pool · pool-deduped, tapat na pagbibilang (walang pinalobong mga ceiling ng rate-limit).", "boost": "I-unlock ang ~{tokens} pa/buwan gamit ang isang beses na $10 OpenRouter top-up (50 → 1000 req/araw)", "uncapped": "Permanenteng libre, walang nai-publish na limitasyon (rate-limited) — totoong access, hindi binibilang sa headline:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# modelo} other {# na modelo}} ang na-flag bilang ToS-restricted — ikaw ang magpasya", "provider": "Provider", "model": "Modelo", diff --git a/src/i18n/messages/pl.json b/src/i18n/messages/pl.json index 9b0c208058..39b053ca78 100644 --- a/src/i18n/messages/pl.json +++ b/src/i18n/messages/pl.json @@ -13133,6 +13133,7 @@ "segmentHint": "Każdy segment = jedna darmowa pula · deduplikacja puli, rzetelne zliczanie (bez zawyżonych limitów zapytań).", "boost": "Odblokuj ~{tokens} więcej/mies. dzięki jednorazowemu doładowaniu OpenRouter za 10 $ (50 → 1000 żądań/dzień)", "uncapped": "Trwale bezpłatne, brak opublikowanego limitu (z ograniczeniem zapytań) — rzeczywisty dostęp, nieuwzględniony w nagłówku:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# model} few {# modele} many {# modeli} other {# modeli}} oznaczono jako ograniczone przez ToS — Ty decydujesz", "provider": "Dostawca", "model": "Model", diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index 840a2d5656..c50d940f5e 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -13144,6 +13144,7 @@ "segmentHint": "Cada segmento = um pool gratuito · deduplicado por pool, contagem honesta (sem tetos de limite de taxa inflados).", "boost": "Desbloqueie ~{tokens} a mais/mês com uma recarga única de $10 na OpenRouter (50 → 1000 solicitações/dia)", "uncapped": "Permanentemente gratuito, sem teto publicado (limitado por taxa) — acesso real, não contabilizado no total principal:", + "gated": "~{tokens}/mês a mais atrás de verificação de identidade regional (ex.: real-name da China continental) — quota real, fora do headline:", "tosRestricted": "{count, plural, one {# modelo} other {# modelos}} sinalizado(s) como restrito(s) pelos Termos de Serviço — você decide", "provider": "Provedor", "model": "Modelo", diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index baa6d1fe1c..aa75826e5f 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -13133,6 +13133,7 @@ "segmentHint": "Cada segmento = um pool gratuito · deduplicado por pool, contagem honesta (sem limites de taxa inflacionados).", "boost": "Desbloqueie mais ~{tokens}/mês com um carregamento único de $10 no OpenRouter (50 → 1000 ped/dia)", "uncapped": "Permanentemente gratuito, sem limite publicado (com limite de taxa) — acesso real, não contabilizado no destaque:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# modelo assinalado} other {# modelos assinalados}} com restrições de ToS — você decide", "provider": "Fornecedor", "model": "Modelo", diff --git a/src/i18n/messages/ro.json b/src/i18n/messages/ro.json index 65c1129bed..4475a38325 100644 --- a/src/i18n/messages/ro.json +++ b/src/i18n/messages/ro.json @@ -13133,6 +13133,7 @@ "segmentHint": "Fiecare segment = un pool gratuit · pool deduplicat, contorizare corectă (fără plafoane de limită de rată umflate).", "boost": "Deblochează încă ~{tokens}/lună cu o reîncărcare unică de 10 $ pe OpenRouter (50 → 1000 cereri/zi)", "uncapped": "Permanent gratuit, fără limită publicată (limitat ca rată) — acces real, necontorizat în titlu:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# model marcat ca restricționat prin ToS — tu decizi} few {# modele marcate ca restricționate prin ToS — tu decizi} other {# de modele marcate ca restricționate prin ToS — tu decizi}}", "provider": "Furnizor", "model": "Model", diff --git a/src/i18n/messages/ru.json b/src/i18n/messages/ru.json index 2d011d4c1d..d5ca918520 100644 --- a/src/i18n/messages/ru.json +++ b/src/i18n/messages/ru.json @@ -13133,6 +13133,7 @@ "segmentHint": "Каждый сегмент = один бесплатный пул · дедупликация пулов, честный подсчет (без завышенных лимитов частоты запросов).", "boost": "Разблокируйте еще ~{tokens}/мес. с помощью разового пополнения OpenRouter на $10 (50 → 1000 запр./день)", "uncapped": "Навсегда бесплатно, без опубликованного лимита (с ограничением частоты) — реальный доступ, не учитывается в заголовке:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# модель помечена как ограниченная Условиями использования} few {# модели помечены как ограниченные Условиями использования} many {# моделей помечено как ограниченные Условиями использования} other {# моделей помечено как ограниченные Условиями использования}} — решать вам", "provider": "Провайдер", "model": "Модель", diff --git a/src/i18n/messages/sk.json b/src/i18n/messages/sk.json index db24a48116..bf82f9a7ea 100644 --- a/src/i18n/messages/sk.json +++ b/src/i18n/messages/sk.json @@ -13133,6 +13133,7 @@ "segmentHint": "Každý segment = jeden bezplatný pool · deduplikované podľa poolov, poctivé počítanie (žiadne nafúknuté stropy limitov).", "boost": "Odomknite o ~{tokens} viac/mes. jednorazovým dobitím 10 $ na OpenRouter (50 → 1000 požiadaviek/deň)", "uncapped": "Trvalo zadarmo, bez zverejneného limitu (obmedzená rýchlosť) — skutočný prístup, nezapočítaný v hlavnom prehľade:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# model označený ako obmedzený podmienkami ToS} few {# modely označené ako obmedzené podmienkami ToS} other {# modelov označených ako obmedzené podmienkami ToS}} — rozhodnutie je na vás", "provider": "Poskytovateľ", "model": "Model", diff --git a/src/i18n/messages/sv.json b/src/i18n/messages/sv.json index 0ee1fc3362..42716bd720 100644 --- a/src/i18n/messages/sv.json +++ b/src/i18n/messages/sv.json @@ -13133,6 +13133,7 @@ "segmentHint": "Varje segment = en gratispool · pool-deduplicerad, ärlig räkning (inga uppblåsta hastighetsbegränsningstak).", "boost": "Lås upp ~{tokens} fler/mån med en engångspåfyllning på $10 hos OpenRouter (50 → 1000 förfrågn./dag)", "uncapped": "Permanent gratis, inget publicerat tak (hastighetsbegränsad) — verklig åtkomst, räknas inte i rubriken:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# modell} other {# modeller}} flaggad som ToS-begränsad — du bestämmer", "provider": "Leverantör", "model": "Modell", diff --git a/src/i18n/messages/sw.json b/src/i18n/messages/sw.json index 72e87fd1ef..50462df6b9 100644 --- a/src/i18n/messages/sw.json +++ b/src/i18n/messages/sw.json @@ -13133,6 +13133,7 @@ "segmentHint": "Kila sehemu = pool moja ya bure · pool isiyo na marudio, hesabu ya uaminifu (hakuna dari zilizoongezwa za kikomo cha kasi).", "boost": "Fungua takriban ~{tokens} zaidi/mwezi kwa kuongeza salio la mara moja la $10 la OpenRouter (maombi 50 → 1000/siku)", "uncapped": "Bure kabisa, hakuna kikomo kilichochapishwa (kasi imedhibitiwa) — ufikiaji halisi, haujahesabiwa kwenye kichwa cha habari:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# mfano} other {# mifano}} imewekewa alama kama iliyozuiliwa na ToS — unaamua", "provider": "Mtoa huduma", "model": "Mfano", diff --git a/src/i18n/messages/ta.json b/src/i18n/messages/ta.json index 8ba8a84e87..05f3494c76 100644 --- a/src/i18n/messages/ta.json +++ b/src/i18n/messages/ta.json @@ -13133,6 +13133,7 @@ "segmentHint": "ஒவ்வொரு பகுதியும் = ஒரு இலவச பூல் · பூல்-நகல் நீக்கப்பட்டது, நேர்மையான எண்ணிக்கை (அதிகரித்த விகித-வரம்பு உச்சவரம்புகள் இல்லை).", "boost": "ஒரு முறை $10 OpenRouter டாப்-அப் மூலம் மாதத்திற்கு மேலும் ~{tokens} ஐ அன்லாக் செய்யவும் (50 → 1000 கோரிக்கைகள்/நாள்)", "uncapped": "நிரந்தரமாக இலவசம், வெளியிடப்பட்ட வரம்பு இல்லை (விகிதம் வரையறுக்கப்பட்டது) — உண்மையான அணுகல், தலைப்புச் செய்தியில் கணக்கிடப்படவில்லை:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# மாடல்} other {# மாடல்கள்}} சேவை விதிமுறைகள் (ToS) கட்டுப்படுத்தப்பட்டதாகக் குறிக்கப்பட்டுள்ளது — நீங்களே முடிவு செய்யுங்கள்", "provider": "வழங்குநர்", "model": "மாடல்", diff --git a/src/i18n/messages/te.json b/src/i18n/messages/te.json index f13ab45bde..6aee70dfa5 100644 --- a/src/i18n/messages/te.json +++ b/src/i18n/messages/te.json @@ -13133,6 +13133,7 @@ "segmentHint": "ప్రతి విభాగం = ఒక ఉచిత పూల్ · పూల్-డూప్లికేట్ తీసివేసినది, నిజాయితీ గల లెక్కింపు (ఎక్కువ చేసి చూపిన రేట్-పరిమితి గరిష్టాలు లేవు).", "boost": "ఒకేసారి $10 OpenRouter టాప్-అప్‌తో నెలకు మరో ~{tokens} అన్‌లాక్ చేయండి (రోజుకు 50 → 1000 అభ్యర్థనలు)", "uncapped": "శాశ్వతంగా ఉచితం, ప్రచురించిన పరిమితి లేదు (రేట్-పరిమితం చేయబడింది) — నిజమైన యాక్సెస్, హెడ్‌లైన్‌లో లెక్కించబడదు:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# మోడల్} other {# మోడల్స్}} ToS-పరిమితం చేయబడినట్లుగా ఫ్లాగ్ చేయబడ్డాయి — మీరే నిర్ణయించుకోండి", "provider": "ప్రొవైడర్", "model": "మోడల్", diff --git a/src/i18n/messages/th.json b/src/i18n/messages/th.json index 1e4b2d4356..b8ea2db512 100644 --- a/src/i18n/messages/th.json +++ b/src/i18n/messages/th.json @@ -13133,6 +13133,7 @@ "segmentHint": "แต่ละส่วน = หนึ่งพูลฟรี · ลดข้อมูลซ้ำในพูล, นับตามจริง (ไม่มีการเพิ่มเพดานจำกัดอัตราการใช้งานเกินจริง)", "boost": "ปลดล็อกเพิ่มอีกประมาณ ~{tokens}/เดือน ด้วยการเติมเงิน OpenRouter $10 ครั้งเดียว (50 → 1000 คำขอ/วัน)", "uncapped": "ฟรีถาวร ไม่มีขีดจำกัดที่เผยแพร่ (จำกัดอัตราการใช้งาน) — เข้าถึงได้จริง ไม่นับรวมในหัวข้อหลัก:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# โมเดล} other {# โมเดล}} ถูกทำเครื่องหมายว่าจำกัดตามข้อกำหนดการให้บริการ — คุณเป็นผู้ตัดสินใจ", "provider": "ผู้ให้บริการ", "model": "โมเดล", diff --git a/src/i18n/messages/tr.json b/src/i18n/messages/tr.json index 35bec39cfa..9875c62e73 100644 --- a/src/i18n/messages/tr.json +++ b/src/i18n/messages/tr.json @@ -13133,6 +13133,7 @@ "segmentHint": "Her segment = bir ücretsiz havuz · havuzdan tekilleştirilmiş, dürüst sayım (şişirilmiş istek sınırı tavanları yok).", "boost": "Tek seferlik 10 $'lık OpenRouter bakiye yüklemesi ile ayda yaklaşık ~{tokens} daha fazlasının kilidini açın (50 → 1000 istek/gün)", "uncapped": "Kalıcı olarak ücretsiz, yayınlanmış bir sınır yok (istek sınırlı) — gerçek erişim, başlıkta sayılmaz:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# model} other {# model}} Hizmet Şartları kısıtlamalı olarak işaretlendi — karar sizin", "provider": "Sağlayıcı", "model": "Model", diff --git a/src/i18n/messages/uk-UA.json b/src/i18n/messages/uk-UA.json index ca35b7c104..3acb5c19e9 100644 --- a/src/i18n/messages/uk-UA.json +++ b/src/i18n/messages/uk-UA.json @@ -13133,6 +13133,7 @@ "segmentHint": "Кожен сегмент = один безкоштовний пул · дедуплікований пул, чесний підрахунок (без завищених лімітів частоти запитів).", "boost": "Розблокуйте ще ~{tokens}/міс за допомогою одноразового поповнення OpenRouter на $10 (50 → 1000 зап./день)", "uncapped": "Постійно безкоштовно, без опублікованого ліміту (з обмеженням частоти) — реальний доступ, не враховується в заголовку:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# модель позначена як обмежена ToS — рішення за вами} few {# моделі позначені як обмежені ToS — рішення за вами} many {# моделей позначено як обмежені ToS — рішення за вами} other {# моделі позначено як обмежені ToS — рішення за вами}}", "provider": "Провайдер", "model": "Модель", diff --git a/src/i18n/messages/ur.json b/src/i18n/messages/ur.json index 98a41c76ae..61baa9bfb6 100644 --- a/src/i18n/messages/ur.json +++ b/src/i18n/messages/ur.json @@ -13133,6 +13133,7 @@ "segmentHint": "ہر حصہ = ایک مفت پول · پول ڈی ڈپلیکیٹڈ، ایماندارانہ گنتی (بغیر کسی بڑھی ہوئی ریٹ لمٹ کی حد کے)۔", "boost": "ایک بار $10 OpenRouter ٹاپ اپ کے ساتھ مزید ~{tokens}/ماہ ان لاک کریں (50 → 1000 req/day)", "uncapped": "مستقل طور پر مفت، کوئی شائع شدہ حد نہیں (ریٹ لمیٹڈ) — حقیقی رسائی، ہیڈ لائن میں شمار نہیں:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# ماڈل} other {# ماڈلز}} ToS-محدود کے طور پر نشان زد — آپ فیصلہ کریں", "provider": "فراہم کنندہ", "model": "ماڈل", diff --git a/src/i18n/messages/vi.json b/src/i18n/messages/vi.json index 0f775b637c..45fb862ce3 100644 --- a/src/i18n/messages/vi.json +++ b/src/i18n/messages/vi.json @@ -13144,6 +13144,7 @@ "segmentHint": "Mỗi đoạn là một nhóm miễn phí · đã khử trùng lặp theo nhóm, đếm đúng thực tế (không thổi phồng trần giới hạn tốc độ).", "boost": "Mở khóa thêm khoảng {tokens}/tháng bằng một lần nạp $10 vào OpenRouter (50 → 1000 yêu cầu/ngày)", "uncapped": "Miễn phí vĩnh viễn, không công bố giới hạn (bị giới hạn tốc độ) — quyền truy cập thực, không tính vào tổng nổi bật:", + "gated": "~{tokens}/tháng nữa nằm sau bước xác minh danh tính theo khu vực (ví dụ: xác thực tên thật ở Trung Quốc đại lục) — hạn mức thật, không tính vào con số chính:", "tosRestricted": "{count, plural, one {# model} other {# models}} flagged as ToS-restricted — you decide", "provider": "Nhà cung cấp", "model": "Mô hình", diff --git a/src/i18n/messages/zh-CN.json b/src/i18n/messages/zh-CN.json index 7283008b91..282545b274 100644 --- a/src/i18n/messages/zh-CN.json +++ b/src/i18n/messages/zh-CN.json @@ -13133,6 +13133,7 @@ "segmentHint": "每个分段 = 一个免费池 · 池去重,真实统计(无虚高的速率限制上限)。", "boost": "一次性充值 $10 OpenRouter 即可每月多解锁约 ~{tokens}(50 → 1000 次请求/天)", "uncapped": "永久免费,无公开上限(受速率限制)— 实际可用,未计入总览:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# 个模型} other {# 个模型}}被标记为受 ToS 限制 — 由您决定", "provider": "提供者", "model": "模型", diff --git a/src/i18n/messages/zh-TW.json b/src/i18n/messages/zh-TW.json index 426b3f2544..e6c513e06e 100644 --- a/src/i18n/messages/zh-TW.json +++ b/src/i18n/messages/zh-TW.json @@ -13133,6 +13133,7 @@ "segmentHint": "每個區段 = 一個免費池 · 池間去重、誠實計數(無膨脹的速率限制上限)。", "boost": "一次性 $10 OpenRouter 充值即可解鎖約 {tokens}/月(50 → 1000 請求/天)", "uncapped": "永久免費,無公佈上限(有速率限制)——真實存取,不計入標題數字:", + "gated": "__MISSING__:~{tokens}/mo more behind a regional identity check (e.g. mainland-China real-name verification) — real quota, not counted in the headline:", "tosRestricted": "{count, plural, one {# model} other {# models}} flagged as ToS-restricted — you decide", "provider": "提供者", "model": "模型", diff --git a/src/lib/radar/applyFeed.ts b/src/lib/radar/applyFeed.ts index 6bfcf3e846..a041cd9de3 100644 --- a/src/lib/radar/applyFeed.ts +++ b/src/lib/radar/applyFeed.ts @@ -42,6 +42,8 @@ export interface MergedEntry { poolKey: string | null; tos: "ok" | "caution" | "ambiguous" | "avoid" | "unknown"; trainsOnPrompts?: boolean; + /** Set when the quota only opens after a region-bound identity check; counted apart. */ + eligibilityGate?: "regional-identity"; /** Whether the entry is enabled for use. Defaults to true. */ enabled?: boolean; /** @@ -113,6 +115,8 @@ export interface FeedModel { metadataEvidenceUrls?: string[]; trainsOnPrompts: boolean | null; tosRisk: MergedEntry["tos"]; + /** Absent = the feed does not know; null = explicitly no gate. */ + eligibilityGate?: "regional-identity" | null; setup: { keyUrl: string | null; steps: RadarLocalizedText[]; @@ -299,6 +303,11 @@ function mergeOne( if (!overriddenKeys.has("trainsOnPrompts")) { result.trainsOnPrompts = feed.trainsOnPrompts ?? undefined; } + // Absent means "this feed predates the field": keep whatever the baseline says. + // An explicit null is the feed clearing the gate. + if (!overriddenKeys.has("eligibilityGate") && feed.eligibilityGate !== undefined) { + result.eligibilityGate = feed.eligibilityGate ?? undefined; + } if (!overriddenKeys.has("creditTokens")) { // Feed doesn't have creditTokens; keep baseline } @@ -329,6 +338,7 @@ function mergeOne( if (overrides.poolKey !== undefined) result.poolKey = overrides.poolKey; if (overrides.tos !== undefined) result.tos = overrides.tos; if (overrides.trainsOnPrompts !== undefined) result.trainsOnPrompts = overrides.trainsOnPrompts; + if (overrides.eligibilityGate !== undefined) result.eligibilityGate = overrides.eligibilityGate; if (overrides.enabled !== undefined) result.enabled = overrides.enabled; if (overrides.contextWindow !== undefined) result.contextWindow = overrides.contextWindow; if (overrides.capabilities !== undefined) result.capabilities = overrides.capabilities; @@ -368,6 +378,7 @@ function feedModelToMerged( creditTokens: overrides?.creditTokens ?? 0, freeType: overrides?.freeType ?? feed.freeType, poolKey: overrides?.poolKey ?? feedBudgetToPoolKey(feed.budget), + eligibilityGate: overrides?.eligibilityGate ?? feed.eligibilityGate ?? undefined, tos: overrides?.tos ?? feed.tosRisk, trainsOnPrompts: overrides?.trainsOnPrompts ?? feed.trainsOnPrompts ?? undefined, enabled: feed.enabled ? (overrides?.enabled ?? true) : false, diff --git a/src/lib/radar/feedSchema.ts b/src/lib/radar/feedSchema.ts index e8895f6cc7..7c2e42ccb9 100644 --- a/src/lib/radar/feedSchema.ts +++ b/src/lib/radar/feedSchema.ts @@ -182,6 +182,8 @@ const ModelV1Schema = z.object({ capabilities: CapabilitiesV1Schema, trainsOnPrompts: z.boolean().nullable(), tosRisk: TosRiskEnum, + /** Real quota, but behind a regional identity verification (counted apart on the client). */ + eligibilityGate: z.enum(["regional-identity"]).nullable().optional(), setup: SetupSchema, enabled: z.boolean(), }); diff --git a/src/lib/radar/index.ts b/src/lib/radar/index.ts index af0c80d3c9..d8f2d88cf5 100644 --- a/src/lib/radar/index.ts +++ b/src/lib/radar/index.ts @@ -86,6 +86,7 @@ export function baselineToMergedEntries(budgets: typeof FREE_MODEL_BUDGETS): Mer poolKey: b.poolKey ?? null, tos: b.tos, trainsOnPrompts: b.trainsOnPrompts, + eligibilityGate: b.eligibilityGate, enabled: true, origin: "baseline" as const, })); diff --git a/tests/unit/check-docs-counts-sync.test.ts b/tests/unit/check-docs-counts-sync.test.ts index 2cea8b52db..7bafffec8c 100644 --- a/tests/unit/check-docs-counts-sync.test.ts +++ b/tests/unit/check-docs-counts-sync.test.ts @@ -106,16 +106,20 @@ test("the gate exits 0 against the current (synced) repo state", () => { // down to 1.37B, because no gate watched that number. import { checkFreeTierHeadline, + extractGatedClaims, extractHeadlineClaims, } from "../../scripts/check/check-docs-counts-sync.mjs"; const checkHeadline = checkFreeTierHeadline as ( content: string, - totals: { s: number; m: number; p: number } + totals: { s: number; m: number; p: number; g?: number } ) => { ok: boolean; detail: string }; const extractClaims = extractHeadlineClaims as ( content: string ) => { value: number; text: string }[]; +const extractGated = extractGatedClaims as ( + content: string +) => { tokens: number; unit: "B" | "M"; text: string }[]; const TOTALS = { s: 1_371_725_000, m: 1_998_225_000, p: 39 }; @@ -147,6 +151,45 @@ test("free-tier gate passes when a file carries no headline at all", () => { assert.equal(checkHeadline("no figures here", TOTALS).ok, true); }); +// --- Eligibility-gated bucket ("+~6M behind regional identity verification") -- +// The gated figure sits next to the headline and is validated with its own anchor, +// so it can neither drift nor be silently dropped once the catalog reports one. +const TOTALS_G = { s: 1_503_225_000, m: 2_129_725_000, p: 35, g: 6_000_000 }; + +test("free-tier gate validates the gated figure that sits next to the headline", () => { + const ok = + "~1.5B free tokens per month … +~6M behind regional identity verification (ModelScope)"; + assert.equal(checkHeadline(ok, TOTALS_G).ok, true); + const stale = "~1.5B free tokens per month … +~60M behind regional identity verification"; + assert.equal(checkHeadline(stale, TOTALS_G).ok, false); + assert.match(checkHeadline(stale, TOTALS_G).detail, /gated/); +}); + +test("free-tier gate rejects a file that carries the headline but omits the gated line", () => { + assert.equal(checkHeadline("~1.5B free tokens per month", TOTALS_G).ok, false); + // a file with no headline at all is still fine (per-provider tables, changelogs) + assert.equal(checkHeadline("no figures here", TOTALS_G).ok, true); + // and nothing changes for callers that pass no gated total + assert.equal( + checkHeadline("~1.5B free tokens per month", { s: TOTALS_G.s, m: TOTALS_G.m, p: 35 }).ok, + true + ); +}); + +test("gated claims are read in M or B and need the anchor phrase", () => { + assert.deepEqual(extractGated("~6M of unrelated text"), []); + assert.deepEqual(extractGated("+~6M behind regional identity verification"), [ + { tokens: 6_000_000, unit: "M", text: "+~6M" }, + ]); + assert.equal( + checkHeadline("~1.5B free tokens per month · ~1.2B behind regional identity verification", { + ...TOTALS_G, + g: 1_230_000_000, + }).ok, + true + ); +}); + // --- Generic numeric-claim gate (engines / MCP tools / scopes / CLI) -------- // Extends the same drift guard to the counts that silently drifted in v3.8.49: // 11→12 engines, 94→109 MCP tools, 30→33 scopes, 26→33 CLI tools. diff --git a/tests/unit/free-catalog-no-confidence-field.test.ts b/tests/unit/free-catalog-no-confidence-field.test.ts index bb546b1668..39be8d8635 100644 --- a/tests/unit/free-catalog-no-confidence-field.test.ts +++ b/tests/unit/free-catalog-no-confidence-field.test.ts @@ -20,6 +20,9 @@ const here = path.dirname(fileURLToPath(import.meta.url)); const CATALOG_ENTRY_KEYS = [ "creditTokens", "displayName", + // Who may claim a quota (e.g. a regional identity check), not how much the row can + // be trusted — the totals split it into its own gated bucket instead of rating it. + "eligibilityGate", "freeType", "hardStopGuaranteed", "modelId", diff --git a/tests/unit/free-model-catalog-gated-bucket.test.ts b/tests/unit/free-model-catalog-gated-bucket.test.ts new file mode 100644 index 0000000000..73cbd0fea5 --- /dev/null +++ b/tests/unit/free-model-catalog-gated-bucket.test.ts @@ -0,0 +1,165 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { + computeFreeModelTotals, + FREE_MODEL_BUDGETS, + type FreeModelBudget, +} from "@omniroute/open-sse/config/freeModelCatalog.ts"; + +/** + * `eligibilityGate` changes COUNTING only: a gated row leaves the steady + * headline (and the pool count) and lands in `gatedRecurringTokens`, + * pool-deduped exactly like the headline. The regime stays what it is. + */ +const base = { creditTokens: 0, tos: "caution" as const }; +const entries: Array = [ + { + provider: "a", + modelId: "a1", + displayName: "A1", + monthlyTokens: 100, + freeType: "recurring-daily", + poolKey: "a-pool", + ...base, + }, + { + provider: "a", + modelId: "a2", + displayName: "A2", + monthlyTokens: 100, + freeType: "recurring-daily", + poolKey: "a-pool", + ...base, + }, + { + provider: "g", + modelId: "g1", + displayName: "G1", + monthlyTokens: 60, + freeType: "recurring-daily", + poolKey: "g-pool", + eligibilityGate: "regional-identity", + ...base, + }, + { + provider: "g", + modelId: "g2", + displayName: "G2", + monthlyTokens: 60, + freeType: "recurring-daily", + poolKey: "g-pool", + eligibilityGate: "regional-identity", + ...base, + }, + { + provider: "h", + modelId: "h1", + displayName: "H1", + monthlyTokens: 7, + freeType: "recurring-monthly", + poolKey: null, + eligibilityGate: "regional-identity", + ...base, + }, + { + provider: "u", + modelId: "u1", + displayName: "U1", + monthlyTokens: 0, + freeType: "recurring-uncapped", + poolKey: "u-pool", + eligibilityGate: "regional-identity", + ...base, + }, +]; + +test("gated entries leave the steady headline and the pool count", () => { + const t = computeFreeModelTotals({ entries }); + assert.equal(t.steadyRecurringTokens, 100); + assert.equal(t.poolCount, 1); + assert.equal(t.firstMonthRealisticTokens, 100); +}); + +test("gated entries are summed apart, pool-deduped, with their providers listed", () => { + const t = computeFreeModelTotals({ entries }); + assert.equal(t.gatedRecurringTokens, 67); // g-pool once (60) + h1 (7); the uncapped u1 adds 0 + assert.deepEqual(t.gatedProviders, ["g", "h"]); +}); + +test("a disabled gated entry contributes nothing", () => { + const t = computeFreeModelTotals({ + entries: entries.map((e) => (e.provider === "h" ? { ...e, enabled: false } : e)), + }); + assert.equal(t.gatedRecurringTokens, 60); + assert.deepEqual(t.gatedProviders, ["g"]); +}); + +test("excludeTosAvoid applies to gated entries too", () => { + const t = computeFreeModelTotals({ + excludeTosAvoid: true, + entries: entries.map((e) => (e.provider === "g" ? { ...e, tos: "avoid" as const } : e)), + }); + assert.equal(t.gatedRecurringTokens, 7); + assert.deepEqual(t.gatedProviders, ["h"]); +}); + +test("a gated signup credit never enters the first-month figure", () => { + const baseline = computeFreeModelTotals({ entries }); + const withGatedCredit = computeFreeModelTotals({ + entries: [ + ...entries, + { + provider: "c", + modelId: "c1", + displayName: "C1", + monthlyTokens: 0, + freeType: "one-time-initial", + poolKey: null, + creditTokens: 1000, + tos: "caution", + eligibilityGate: "regional-identity", + }, + ], + }); + assert.equal( + withGatedCredit.firstMonthRealisticTokens, + baseline.firstMonthRealisticTokens, + "a gated one-time credit must not inflate the first-month headline" + ); + assert.equal( + withGatedCredit.steadyWithRecurringCreditsTokens, + baseline.steadyWithRecurringCreditsTokens + ); +}); + +test("a gated uncapped provider is not advertised as permanently free", () => { + const t = computeFreeModelTotals({ entries }); + // `u` is the gated recurring-uncapped row in the fixture above. + assert.ok(!t.uncappedProviders.includes("u"), "gated rows must stay out of uncappedProviders"); + assert.deepEqual(t.uncappedProviders, []); +}); + +test("the shipped catalog exposes the two new fields", () => { + const t = computeFreeModelTotals(); + assert.equal(typeof t.gatedRecurringTokens, "number"); + assert.ok(Array.isArray(t.gatedProviders)); +}); + +/** + * Invariant: the gated bucket only ever accounts for STEADY tokens. A gated row + * carrying credits would silently drop them from every figure (credits are filtered + * out of the credit sums, and `gatedRecurringTokens` only sums `monthlyTokens`), so + * such a row must not exist in the shipped catalog without the totals growing a + * matching gated-credit figure first. + */ +test("shipped gated rows carry no credit tokens", () => { + for (const m of FREE_MODEL_BUDGETS) { + if (!m.eligibilityGate) continue; + assert.equal( + m.creditTokens, + 0, + `${m.provider}/${m.modelId} is eligibility-gated but declares creditTokens=${m.creditTokens}` + ); + } +}); diff --git a/tests/unit/free-tier-reaudit-2026-09.test.ts b/tests/unit/free-tier-reaudit-2026-09.test.ts index f227616349..32323bf893 100644 --- a/tests/unit/free-tier-reaudit-2026-09.test.ts +++ b/tests/unit/free-tier-reaudit-2026-09.test.ts @@ -115,3 +115,37 @@ test("legacy provider-level catalog agrees with the per-model catalog for the re assert.equal(FREE_TIER_BUDGETS.groq, 30_000_000); assert.equal(FREE_TIER_BUDGETS.nara, 210_000_000); }); + +test("modelscope: 250 魔粒/day → 6M/month, one pool, behind mainland real-name verification", () => { + const m = rows("modelscope"); + assert.ok(m.length >= 1); + for (const r of m) { + assert.equal(r.monthlyTokens, 6_000_000, r.modelId); + assert.equal(r.poolKey, "modelscope-free"); + assert.equal(r.freeType, "recurring-daily"); + assert.equal(r.eligibilityGate, "regional-identity"); + assert.equal(r.tos, "caution"); + } + const t = computeFreeModelTotals(); + assert.equal(t.gatedRecurringTokens, 6_000_000); + assert.deepEqual(t.gatedProviders, ["modelscope"]); + assert.ok(!t.uncappedProviders.includes("modelscope")); +}); + +test("a shared pool never mixes gated and ungated entries", () => { + const byPool = new Map>(); + for (const r of FREE_MODEL_BUDGETS) { + if (!r.poolKey) continue; + const gate = r.eligibilityGate ?? "none"; + const seen = byPool.get(r.poolKey) ?? new Set(); + seen.add(gate); + byPool.set(r.poolKey, seen); + } + for (const [poolKey, gates] of byPool) { + assert.equal( + gates.size, + 1, + `pool ${poolKey} mixes eligibility gates: ${[...gates].join(", ")}` + ); + } +}); diff --git a/tests/unit/free-tier-summary-radar-overlay.test.ts b/tests/unit/free-tier-summary-radar-overlay.test.ts index 59090b1c3c..b63bdab68d 100644 --- a/tests/unit/free-tier-summary-radar-overlay.test.ts +++ b/tests/unit/free-tier-summary-radar-overlay.test.ts @@ -54,6 +54,7 @@ const STALE_GEN_AT = "2026-01-02T12:00:00.000Z"; const FETCHED_AT = `${FREE_CATALOG_CURATED_AT}T18:00:00.000Z`; const OVERLAY_TOKENS = 1_234_567; const DISABLED_TOKENS = 9_999_999; +const GATED_TOKENS = 4_242_000; async function authCookieHeader(): Promise { const secret = new TextEncoder().encode(process.env.JWT_SECRET); @@ -115,6 +116,34 @@ function feedPayload(tier: "community" | "live") { }; } +function seedGatedCache() { + const payload = feedPayload("community"); + payload.models.push({ + provider: "test-radar", + modelId: "overlay-gated-model", + displayName: "Overlay Gated Model", + familyId: null, + freeType: "recurring-daily", + budget: { kind: "shared_pool", poolId: "overlay-gated", tokensPerMonth: GATED_TOKENS }, + limits: { rpm: null, rpd: null, tpm: null, tpd: null }, + contextWindow: null, + capabilities: { tools: false, vision: false, thinking: false }, + trainsOnPrompts: null, + tosRisk: "ok", + setup: null, + enabled: true, + eligibilityGate: "regional-identity", + } as (typeof payload.models)[number]); + radarDb.setRadarCache({ + version: "2026.08.25.1", + generatedAt: GEN_AT, + tier: "community", + payload: JSON.stringify(payload), + signature: "test-signature-not-verified-on-read", + fetchedAt: FETCHED_AT, + }); +} + function resetState() { core.resetDbInstance(); try { @@ -364,3 +393,25 @@ test("stale overlay => a locally disabled model stays out of the totals", async "a locally disabled model must not be counted when the feed is withheld" ); }); + +// --- eligibility-gated entries from the feed stay out of the headline -------- + +test("a gated overlay entry lands in gatedRecurringTokens, never in the steady headline", async () => { + resetState(); + setFeatureFlagOverride("RADAR_ENABLED", "true"); + seedGatedCache(); + + const body = await getBody(false); + + assert.equal(body.catalogSource, "radar-overlay"); + assert.equal( + body.gatedRecurringTokens, + (computeFreeModelTotals().gatedRecurringTokens as number) + GATED_TOKENS, + "the feed-only gated pool joins the gated bucket on top of the baseline's own" + ); + assert.equal( + body.steadyRecurringTokens, + (computeFreeModelTotals().steadyRecurringTokens as number) + OVERLAY_TOKENS, + "the gated pool must not move the steady headline" + ); +}); diff --git a/tests/unit/gen-budget-card-svg.test.ts b/tests/unit/gen-budget-card-svg.test.ts new file mode 100644 index 0000000000..a976ad3b6d --- /dev/null +++ b/tests/unit/gen-budget-card-svg.test.ts @@ -0,0 +1,50 @@ +import assert from "node:assert/strict"; +import { execFileSync } from "node:child_process"; +import { mkdtempSync, readFileSync } from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; + +import { computeFreeModelTotals } from "@omniroute/open-sse/config/freeModelCatalog.ts"; + +const fmt = (n: number) => (n >= 1e9 ? (n / 1e9).toFixed(2) + "B" : Math.round(n / 1e6) + "M"); + +test("the budget card prints the catalog's own totals (no regex-parsed subset)", () => { + const out = path.join(mkdtempSync(path.join(os.tmpdir(), "budget-card-")), "card.svg"); + execFileSync( + process.execPath, + ["--import", "tsx/esm", "scripts/research/gen-budget-card-svg.mjs", "--out", out], + { stdio: "pipe" } + ); + const svg = readFileSync(out, "utf8"); + const t = computeFreeModelTotals(); + assert.ok(svg.includes(`~${fmt(t.steadyRecurringTokens)}`), "steady figure"); + assert.ok(svg.includes(`~${fmt(t.firstMonthRealisticTokens)}`), "first-month figure"); + assert.ok(svg.includes(`${t.uncappedProviders.length} permanently-free`), "uncapped count"); + if (t.gatedRecurringTokens > 0) { + assert.ok(svg.includes("behind regional identity verification"), "gated line"); + } + + // The committed card is a generated artifact: it must be exactly what the + // generator produces today, or the docs ship a stale picture of the totals. + const committed = readFileSync( + path.join(import.meta.dirname, "../../docs/screenshots/free-tier-budget-card.svg"), + "utf8" + ); + assert.equal( + svg, + committed, + "docs/screenshots/free-tier-budget-card.svg is stale — regenerate with " + + "`node --import tsx/esm scripts/research/gen-budget-card-svg.mjs`" + ); +}); + +test("--out without a path fails loudly instead of writing to undefined", () => { + assert.throws(() => + execFileSync( + process.execPath, + ["--import", "tsx/esm", "scripts/research/gen-budget-card-svg.mjs", "--out"], + { stdio: "pipe" } + ) + ); +}); diff --git a/tests/unit/radar-feed-eligibility-gate.test.ts b/tests/unit/radar-feed-eligibility-gate.test.ts new file mode 100644 index 0000000000..e93158079f --- /dev/null +++ b/tests/unit/radar-feed-eligibility-gate.test.ts @@ -0,0 +1,128 @@ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import test from "node:test"; + +import { RadarFeedSchema } from "../../src/lib/radar/feedSchema.ts"; +import { applyFeed, type FeedModel, type MergedEntry } from "../../src/lib/radar/applyFeed.ts"; +import { baselineToMergedEntries } from "../../src/lib/radar/index.ts"; + +const fixture = JSON.parse( + readFileSync(new URL("../fixtures/radar-feed-canonical.json", import.meta.url), "utf8") +) as { models: Array> }; + +test("feed schema: eligibilityGate is optional, nullable and closed to unknown values", () => { + const absent = RadarFeedSchema.parse(structuredClone(fixture)); + assert.equal(absent.models[0]!.eligibilityGate, undefined); + const gated = structuredClone(fixture); + gated.models[0]!.eligibilityGate = "regional-identity"; + assert.equal(RadarFeedSchema.parse(gated).models[0]!.eligibilityGate, "regional-identity"); + const nulled = structuredClone(fixture); + nulled.models[0]!.eligibilityGate = null; + assert.equal(RadarFeedSchema.parse(nulled).models[0]!.eligibilityGate, null); + const bad = structuredClone(fixture); + bad.models[0]!.eligibilityGate = "vip"; + assert.throws(() => RadarFeedSchema.parse(bad)); +}); + +function feedModel(overrides: Partial = {}): FeedModel { + return { + provider: "modelscope", + modelId: "Qwen/Qwen3.5-397B-A17B", + displayName: "Qwen3.5 397B A17B (ModelScope)", + familyId: null, + freeType: "recurring-daily", + budget: { kind: "shared_pool", poolId: "modelscope-free", tokensPerMonth: 6_000_000 }, + limits: { rpm: null, rpd: null, tpm: null, tpd: null }, + contextWindow: null, + capabilities: { tools: null, vision: null, thinking: null }, + trainsOnPrompts: null, + tosRisk: "caution", + setup: null, + enabled: true, + ...overrides, + }; +} +const KEY = "modelscope:Qwen/Qwen3.5-397B-A17B"; +const noLocal = () => ({ + localOverrides: new Map>(), + tombstones: new Set(), +}); +const gatedBaseline = () => + baselineToMergedEntries([ + { + provider: "modelscope", + modelId: "Qwen/Qwen3.5-397B-A17B", + displayName: "Qwen3.5 397B A17B (ModelScope)", + monthlyTokens: 6_000_000, + creditTokens: 0, + freeType: "recurring-daily", + poolKey: "modelscope-free", + tos: "caution", + eligibilityGate: "regional-identity", + }, + ]); + +test("baseline entries keep their gate through baselineToMergedEntries", () => { + assert.equal(gatedBaseline()[0]!.eligibilityGate, "regional-identity"); +}); + +test("a feed-only entry carries its gate into the merged catalog", () => { + const [m] = applyFeed({ + baseline: [], + feed: [feedModel({ eligibilityGate: "regional-identity" })], + ...noLocal(), + }); + assert.equal(m!.eligibilityGate, "regional-identity"); + assert.equal(m!.origin, "radar"); +}); + +test("a feed that does not know the field preserves the baseline gate; an explicit null clears it", () => { + const kept = applyFeed({ baseline: gatedBaseline(), feed: [feedModel()], ...noLocal() }); + assert.equal(kept[0]!.eligibilityGate, "regional-identity"); + const cleared = applyFeed({ + baseline: gatedBaseline(), + feed: [feedModel({ eligibilityGate: null })], + ...noLocal(), + }); + assert.equal(cleared[0]!.eligibilityGate, undefined); +}); + +test("a local override on the gate wins over the feed", () => { + const localOverrides = new Map>([ + [KEY, { eligibilityGate: "regional-identity" }], + ]); + const [m] = applyFeed({ + baseline: [], + feed: [feedModel({ eligibilityGate: null })], + localOverrides, + tombstones: new Set(), + }); + assert.equal(m!.eligibilityGate, "regional-identity"); + assert.equal(m!.origin, "local"); +}); + +test("a local override sets the gate even when the feed clears it on a baseline entry", () => { + const ungatedBaseline = baselineToMergedEntries([ + { + provider: "modelscope", + modelId: "Qwen/Qwen3.5-397B-A17B", + displayName: "Qwen3.5 397B A17B (ModelScope)", + monthlyTokens: 6_000_000, + creditTokens: 0, + freeType: "recurring-daily", + poolKey: "modelscope-free", + tos: "caution", + }, + ]); + assert.equal(ungatedBaseline[0]!.eligibilityGate, undefined); + const [m] = applyFeed({ + baseline: ungatedBaseline, + feed: [feedModel({ eligibilityGate: null })], + localOverrides: new Map>([ + [KEY, { eligibilityGate: "regional-identity" }], + ]), + tombstones: new Set(), + }); + assert.equal(m!.eligibilityGate, "regional-identity"); + assert.equal(m!.origin, "local"); +}); diff --git a/tests/unit/ui/free-budget-card-gated.test.tsx b/tests/unit/ui/free-budget-card-gated.test.tsx new file mode 100644 index 0000000000..55be1c908e --- /dev/null +++ b/tests/unit/ui/free-budget-card-gated.test.tsx @@ -0,0 +1,77 @@ +// @vitest-environment jsdom +import React, { act } from "react"; +import { createRoot } from "react-dom/client"; +import { afterEach, beforeAll, describe, expect, it, vi } from "vitest"; + +vi.mock("next-intl", () => ({ + useTranslations: () => (key: string) => key, +})); + +const { default: FreeBudgetCard } = + await import("../../../src/app/(dashboard)/dashboard/usage/components/FreeBudgetCard"); + +const summary = { + steadyRecurringTokens: 1_503_225_000, + steadyWithRecurringCreditsTokens: 1_504_225_000, + firstMonthRealisticTokens: 2_129_725_000, + usedThisMonth: 0, + remaining: 1_503_225_000, + modelCount: 453, + poolCount: 35, + perModel: [], + boostMonthlyTokens: 24_000_000, + uncappedProviders: ["gemini"], + gatedRecurringTokens: 6_000_000, + gatedProviders: ["modelscope"], + catalogUpdatedAt: null, + noCredentialProviders: [], +}; + +describe("FreeBudgetCard — eligibility-gated line", () => { + beforeAll(() => { + ( + globalThis as typeof globalThis & { IS_REACT_ACT_ENVIRONMENT?: boolean } + ).IS_REACT_ACT_ENVIRONMENT = true; + }); + + afterEach(() => { + document.body.innerHTML = ""; + vi.unstubAllGlobals(); + }); + + it("renders the gated callout with its providers when gatedRecurringTokens > 0", async () => { + vi.stubGlobal( + "fetch", + vi.fn(async () => new Response(JSON.stringify(summary), { status: 200 })) + ); + const el = document.createElement("div"); + document.body.appendChild(el); + const root = createRoot(el); + await act(async () => { + root.render(); + }); + await act(async () => {}); + expect(el.textContent).toContain("gated"); + expect(el.textContent).toContain("modelscope"); + }); + + it("hides the callout when the gated total is zero", async () => { + vi.stubGlobal( + "fetch", + vi.fn( + async () => + new Response( + JSON.stringify({ ...summary, gatedRecurringTokens: 0, gatedProviders: [] }), + { status: 200 } + ) + ) + ); + const el = document.createElement("div"); + document.body.appendChild(el); + await act(async () => { + createRoot(el).render(); + }); + await act(async () => {}); + expect(el.textContent).not.toContain("gated"); + }); +}); From 008da6d19a0399e71b685eeb68f805d5fc240221 Mon Sep 17 00:00:00 2001 From: Markus Hartung Date: Fri, 4 Sep 2026 08:39:09 +0200 Subject: [PATCH 101/143] feat(dashboard): link a log entry's Conversation Context to its owning conversation (#12646) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validado sobre o tip de `release/v3.8.51`, com duas coisas resolvidas antes do merge. **A falha de CI era stale.** O job `No new ESLint warnings` deste PR apontava `react-hooks/set-state-in-effect` em `src/app/(dashboard)/dashboard/combos/page.tsx:774` — arquivo que este PR não toca, e o mesmo erro aparecia em #12668 e #12672, que também não o tocam. A linha do tempo: o #12355 introduziu a violação de manhã, os CIs rodaram nessa janela, e o #12607 acrescentou a entrada de supressão à tarde. Medido no tip atual com o comando exato do job: **0 ocorrências não suprimidas**. A supressão sobrevivente é "unpruned", e o script passa `--pass-on-unpruned-suppressions` justamente para isso não bloquear. **Faltava o teste que a regra do projeto exige** para mudanças em `src/`. Acrescentei `tests/unit/ui/log-detail-conversation-link-12646.test.tsx`, verificado **RED-then-GREEN** em vez de escrito contra o código pronto: revertendo `RequestLoggerDetail.sections.tsx` para o tip, 2 dos 3 casos falham; com a mudança deste PR, 3/3 passam. Detalhe que valeu a pena descobrir: a seção curto-circuita em `allTurns.length === 0`, então o fixture precisa de um `requestBody` que normalize em pelo menos um turno — sem isso o cabeçalho inteiro nunca monta e as asserções passariam pelo motivo errado. O teste fixa três coisas: o href para um `sessionTag` simples, o percent-encoding para um que não é URL-safe, e a ausência de link quando não há `sessionTag`. Obrigado, @hartmark. --- .../RequestLoggerDetail.sections.tsx | 15 +++ ...og-detail-conversation-link-12646.test.tsx | 106 ++++++++++++++++++ 2 files changed, 121 insertions(+) create mode 100644 tests/unit/ui/log-detail-conversation-link-12646.test.tsx diff --git a/src/shared/components/RequestLoggerDetail.sections.tsx b/src/shared/components/RequestLoggerDetail.sections.tsx index 78527ab495..cbdce62a01 100644 --- a/src/shared/components/RequestLoggerDetail.sections.tsx +++ b/src/shared/components/RequestLoggerDetail.sections.tsx @@ -273,6 +273,21 @@ export function ConversationContextSection({ log, detail }) { continues from parent )} + {liveDetail?.sessionTag && ( + // /dashboard/conversations reads its own `?tree=` deep-link param + // from a fresh mount too (useState(() => searchParams.get("tree")) in + // that page) -- same full-navigation reasoning as the parent-log link + // above. sessionTag is the same conv_ the conversations list and + // /api/conversations/[id]/tree both key on. + + forum + view conversation + + )} {open && (
diff --git a/tests/unit/ui/log-detail-conversation-link-12646.test.tsx b/tests/unit/ui/log-detail-conversation-link-12646.test.tsx new file mode 100644 index 0000000000..9a8ad2eeba --- /dev/null +++ b/tests/unit/ui/log-detail-conversation-link-12646.test.tsx @@ -0,0 +1,106 @@ +// @vitest-environment jsdom +/** + * Guard for #12646: a log entry's Conversation Context header links to the + * conversation that owns it. + * + * The link is a plain `` on purpose, not a client-side route push: + * /dashboard/conversations reads its own `?tree=` deep-link param from a + * fresh mount (`useState(() => searchParams.get("tree"))`), so it only picks + * the param up on a full navigation. A future refactor to a Next `` + * would silently stop opening the right tree — which is what this test pins. + */ +import React from "react"; +import { act } from "react"; +import { createRoot, type Root } from "react-dom/client"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; + +const RequestLoggerDetail = (await import("../../../src/shared/components/RequestLoggerDetail.tsx")) + .default; + +let container: HTMLElement; +let root: Root; + +function baseLog(overrides: Record = {}) { + return { + id: "log-12646", + status: 200, + method: "POST", + path: "/v1/chat/completions", + model: "gpt-test", + provider: "openai", + timestamp: new Date().toISOString(), + duration: 42, + tokens: { in: 1, out: 2 }, + ...overrides, + }; +} + +const noop = () => {}; + +// The section short-circuits with `if (allTurns.length === 0) return null`, so the +// fixture needs a request body that normalizes into at least one turn — otherwise +// the whole header, link included, never mounts and the assertions would pass or +// fail for the wrong reason. +const REQUEST_BODY_WITH_A_TURN = { + model: "gpt-test", + messages: [{ role: "user", content: "hello" }], +}; + +async function renderDetail(detail: Record) { + const log = baseLog(); + await act(async () => { + root.render( + true} + /> + ); + }); +} + +function conversationLink(): HTMLAnchorElement | null { + return container.querySelector('a[href^="/dashboard/conversations?tree="]'); +} + +beforeEach(() => { + container = document.createElement("div"); + document.body.appendChild(container); + root = createRoot(container); +}); + +afterEach(async () => { + if (root) { + await act(async () => { + root.unmount(); + }); + } + container?.remove(); +}); + +describe("log detail — Conversation Context link (#12646)", () => { + it("deep-links to the owning conversation when the detail carries a sessionTag", async () => { + await renderDetail({ sessionTag: "conv_abc123" }); + + const link = conversationLink(); + expect(link).not.toBeNull(); + expect(link!.getAttribute("href")).toBe("/dashboard/conversations?tree=conv_abc123"); + }); + + it("percent-encodes a sessionTag that is not URL-safe", async () => { + await renderDetail({ sessionTag: "conv_a b/c?d" }); + + expect(conversationLink()!.getAttribute("href")).toBe( + `/dashboard/conversations?tree=${encodeURIComponent("conv_a b/c?d")}` + ); + }); + + it("renders no conversation link when the detail has no sessionTag", async () => { + await renderDetail({}); + + expect(conversationLink()).toBeNull(); + }); +}); From c3945a724c4a44a413410a642c95ca54ee908a5c Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Fri, 4 Sep 2026 04:07:43 -0300 Subject: [PATCH 102/143] fix(ci): security-tier gate must honor ALWAYS_PROTECTED_API_PATTERNS too (+ file-size rebaseline) (#12605) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(ci): mirror isLocalOnlyPath in the security-tier gate and rebaseline four merged-growth file caps Two base-reds on release/v3.8.51 (#12581), both drained at the source. 1) check:openapi-security-tiers reported six CORRECTLY annotated routes as unprotected and demanded the removal of their x-loopback-only annotation — pushing the fix in the unsafe direction. The gate re-reads routeGuard.ts as text (it cannot import the module: routeGuard pulls the server runtime and the gate runs on plain node), but it only read the FIRST half of isLocalOnlyPath(): LOCAL_ONLY_API_PREFIXES.some(...) || LOCAL_ONLY_API_PATTERNS.some(...) so every route gated by a regex (/api/providers/volcengine-plan/connect/*) or by an imported constant (VNC_ROUTE_PREFIX, which the text parse turned into the literal string "VNC_ROUTE_PREFIX") looked open. Proven with isLocalOnlyPath() at runtime: all six return true; the control /api/providers/{id}/refresh stays false. New scripts/check/routeGuardConstants.mjs reads BOTH arrays, resolves imported identifiers by following the import, and THROWS on an unresolvable token instead of silently degrading it into a literal. Its array scanner is hand-rolled because regex literals carry the brackets and commas a \[([^\]]+)\] capture plus a naive comma split break on ([^/] and {1,3}). The reverse pass (missing-annotation warnings) now uses the same predicate. 2) check:file-size: four frozen files grew past their cap through merged PRs — chat.ts +10 (#12427/#12503 video-transcript redaction, derived from the post-guardrail payload at the single dispatch point) and stream.ts / accountFallback.ts / codex.ts +17 total (#12179 hot-path regex hoisting, bounded caches, quadratic-buffering fix). All cohesive at existing chokepoints; rebaselined with the rationale recorded in the baseline file. Refs #12581 * fix(ci): security-tier gate must honor ALWAYS_PROTECTED_API_PATTERNS too #12350 fixed the LOCAL_ONLY half of the checker (prefixes + patterns + imported consts). isAlwaysProtectedPath() is two-armed the same way: ALWAYS_PROTECTED_API_PATHS.some(...) || ALWAYS_PROTECTED_API_PATTERNS.some(...) but the checker still read only the path array, so the four credential routes gated by the GHSA-5926-2w35-7h4q pattern (#12600) — /api/providers/{id}/{claude,codex}-auth/{export,apply-local} — reported as 'has x-always-protected but is NOT in ALWAYS_PROTECTED_API_PATHS', asking for the removal of a CORRECT annotation on a credential-export route. Verified with the real predicate: all four isAlwaysProtectedPath() → true; control /api/providers/{id}/models → false. tests/unit/openapi-security-tiers.test.ts already checks BOTH arrays (#12600 updated the test but not the gate script) and stays green — this commit makes the gate agree with the test and with the runtime. Also carries the file-size rebaseline for four caps grown by merged PRs (chat.ts +10 from #12427/#12503; stream.ts / accountFallback.ts / codex.ts +17 from #12179), rationale recorded in the baseline file. Refs #12581 * fix(ci): re-anchor the zcodeProtocol public-creds allowlist entry (302 -> 313) The check:public-creds allowlist pins each frozen literal by FILE:LINE, so #12179 (hot-path regex hoisting in the same file) shifted the ZCode handshake id from L302 to L313 and broke the gate twice over: the old entry went stale ('a violação foi corrigida; REMOVA a entrada') while the literal itself, now at L313, was no longer covered. The literal is unchanged and still not a credential: `omniroute-${process.pid}` is a per-process handshake id for the local ZCode app-server, already audited and frozen with that justification. Only the anchor moves. Refs #12581 * test(ci): re-anchor the ZCode allowlist test to L313 alongside the gate entry The allowlist key is file:LINE:value, so the synthetic source in this test pads to the exact line the entry pins. Re-anchoring the entry 302 -> 313 (previous commit) without moving the padding left the test asserting the old line — caught by Unit Tests fast-path (4/4) on #12605. Both halves now sit at 313, and the test still proves the allowlist does NOT weaken detection: swapping the value for 'upstream-client-' is still flagged. Refs #12581 * docs(ci): changelog fragment for #12605 * chore(ci): trim #12605 to the one fix the base still needs The base drained fast while this PR was open. Re-verified on 008da6d19a and dropped everything already covered there: - check-public-creds.mjs: the base already re-anchors the ZCode entry to L313 (my commit only added a comment on top) -> reverted to the base version. - file-size-baseline.json: the base rebaselined chat.ts/codex.ts/ accountFallback.ts to HIGHER caps than mine, and stream.ts measures 3064 against the base cap of 3072 — my 3078 bump would have loosened a cap for no reason -> reverted to the base version. What the base still does NOT have, verified on its current tip: node scripts/check/check-openapi-security-tiers.mjs -> EXIT=1, 4 mismatches so the ALWAYS_PROTECTED_API_PATTERNS half stays, plus its changelog entry. Refs #12581 --- .../fixes/12605-openapi-tiers-public-creds.md | 1 + .../check/check-openapi-security-tiers.mjs | 27 ++++++++++++------- 2 files changed, 19 insertions(+), 9 deletions(-) create mode 100644 changelog.d/fixes/12605-openapi-tiers-public-creds.md diff --git a/changelog.d/fixes/12605-openapi-tiers-public-creds.md b/changelog.d/fixes/12605-openapi-tiers-public-creds.md new file mode 100644 index 0000000000..f484d00adb --- /dev/null +++ b/changelog.d/fixes/12605-openapi-tiers-public-creds.md @@ -0,0 +1 @@ +- **CI:** the OpenAPI security-tier gate now mirrors `isAlwaysProtectedPath()` in full — it also reads `ALWAYS_PROTECTED_API_PATTERNS`, so the pattern-gated credential routes (`/api/providers/{id}/{claude,codex}-auth/{export,apply-local}`, GHSA-5926-2w35-7h4q) no longer report as unannotated. (#12605) diff --git a/scripts/check/check-openapi-security-tiers.mjs b/scripts/check/check-openapi-security-tiers.mjs index 89d17c0482..bd0096015e 100644 --- a/scripts/check/check-openapi-security-tiers.mjs +++ b/scripts/check/check-openapi-security-tiers.mjs @@ -106,6 +106,11 @@ function parsePatterns(name) { const LOCAL_ONLY_PREFIXES = parsePrefixes("LOCAL_ONLY_API_PREFIXES"); const LOCAL_ONLY_PATTERNS = parsePatterns("LOCAL_ONLY_API_PATTERNS"); const ALWAYS_PROTECTED_PATHS = parsePrefixes("ALWAYS_PROTECTED_API_PATHS"); +// isAlwaysProtectedPath() is ALSO two-armed (paths || patterns) — reading only the +// path array repeated, on this half, the very bug #12350 fixed on the LOCAL_ONLY +// half: the pattern-gated credential routes (…/{claude,codex}-auth/{export, +// apply-local}, #12600) read as unannotated even though they are protected. +const ALWAYS_PROTECTED_PATTERNS = parsePatterns("ALWAYS_PROTECTED_API_PATTERNS"); if ( LOCAL_ONLY_PREFIXES.length === 0 || @@ -135,6 +140,15 @@ function coveredByLocalOnly(pathStr) { return matchesPrefix(concrete) || LOCAL_ONLY_PATTERNS.some((re) => re.test(concrete)); } +/** Mirror of routeGuard.isAlwaysProtectedPath() — both arms, same order. */ +function coveredByAlwaysProtected(pathStr) { + const concrete = concretize(pathStr); + return ( + ALWAYS_PROTECTED_PATHS.some((p) => concrete === p || concrete.startsWith(`${p}/`)) || + ALWAYS_PROTECTED_PATTERNS.some((re) => re.test(concrete)) + ); +} + const raw = yaml.load(fs.readFileSync(OPENAPI_PATH, "utf-8")); const paths = raw.paths || {}; const errors = []; @@ -151,16 +165,11 @@ for (const [pathStr, methods] of Object.entries(paths)) { ); } - if (spec["x-always-protected"] === true) { - const matchesPath = ALWAYS_PROTECTED_PATHS.some( - (p) => pathStr === p || pathStr.startsWith(`${p}/`) + if (spec["x-always-protected"] === true && !coveredByAlwaysProtected(pathStr)) { + errors.push( + `${method.toUpperCase()} ${pathStr}: has x-always-protected but is NOT covered by ` + + `ALWAYS_PROTECTED_API_PATHS or ALWAYS_PROTECTED_API_PATTERNS` ); - if (!matchesPath) { - errors.push( - `${method.toUpperCase()} ${pathStr}: has x-always-protected but is NOT in ` + - `ALWAYS_PROTECTED_API_PATHS [${ALWAYS_PROTECTED_PATHS.join(", ")}]` - ); - } } } } From 891cb26b2cc647333893aed32c72a897dd184459 Mon Sep 17 00:00:00 2001 From: Koosha Paridehpour <42529354+KooshaPari@users.noreply.github.com> Date: Fri, 4 Sep 2026 22:33:16 -0700 Subject: [PATCH 103/143] fix(db): back-fill last_ping_at + last_pinged_reset_key on provider_connections (#12470) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged. Clean, surgical fix with its own regression guard. `ensureProviderConnectionsColumns()` reconciles the base columns that later data migrations assume, but `last_ping_at` / `last_pinged_reset_key` were only ever created by `123_quota_auto_ping` — so a lineage that skipped it kept a table that the quota auto-ping writes cannot target. Adding them to the reconciliation list is exactly the right place. Validated on `release/v3.8.51`: `tests/unit/db-schema-columns-split.test.ts` 10/10, including your new `back-fills last_ping columns on a pre-123 lineage` case and the idempotency re-run. `typecheck:core` clean, `check-file-size` OK. The `changelog.d/fixes/` fragment was already correct. Thank you — this is the shape a fix should have: root cause named, minimal diff, test that fails without it. --- .../fixes/12470-last-ping-at-backfill.md | 1 + src/lib/db/schemaColumns.ts | 2 ++ tests/unit/db-schema-columns-split.test.ts | 19 +++++++++++++++++++ 3 files changed, 22 insertions(+) create mode 100644 changelog.d/fixes/12470-last-ping-at-backfill.md diff --git a/changelog.d/fixes/12470-last-ping-at-backfill.md b/changelog.d/fixes/12470-last-ping-at-backfill.md new file mode 100644 index 0000000000..09d45f4acd --- /dev/null +++ b/changelog.d/fixes/12470-last-ping-at-backfill.md @@ -0,0 +1 @@ +- **fix(db):** back-fill `last_ping_at` and `last_pinged_reset_key` on `provider_connections` during schema reconciliation so divergent lineages that skipped `123_quota_auto_ping` still accept quota auto-ping writes ([#12470](https://github.com/diegosouzapw/OmniRoute/pull/12470) — thanks @KooshaPari) diff --git a/src/lib/db/schemaColumns.ts b/src/lib/db/schemaColumns.ts index 2b58470948..068c140c69 100644 --- a/src/lib/db/schemaColumns.ts +++ b/src/lib/db/schemaColumns.ts @@ -27,6 +27,8 @@ export function ensureProviderConnectionsColumns(db: SqliteDatabase) { ["rate_limit_protection", "INTEGER DEFAULT 0"], ["last_used_at", "TEXT"], ["default_model", "TEXT"], // legacy-schema hole; later data migrations read it + ["last_ping_at", "TEXT"], // added by 123_quota_auto_ping; back-filled here for divergent lineages + ["last_pinged_reset_key", "TEXT"], // added by 123_quota_auto_ping; back-filled here for divergent lineages ]) { if (!columnNames.has(column)) { db.exec(`ALTER TABLE provider_connections ADD COLUMN ${column} ${type}`); diff --git a/tests/unit/db-schema-columns-split.test.ts b/tests/unit/db-schema-columns-split.test.ts index 9adebe3b93..9e7e249a6a 100644 --- a/tests/unit/db-schema-columns-split.test.ts +++ b/tests/unit/db-schema-columns-split.test.ts @@ -136,6 +136,8 @@ test("ensureProviderConnectionsColumns restores base columns required by later m assert.equal(hasColumn(db, "provider_connections", "provider_specific_data"), true); assert.equal(hasColumn(db, "provider_connections", "default_model"), true); + assert.equal(hasColumn(db, "provider_connections", "last_ping_at"), true); + assert.equal(hasColumn(db, "provider_connections", "last_pinged_reset_key"), true); const columnsAfterFirstRun = getTableColumns(db, "provider_connections").sort(); const indexesAfterFirstRun = ( db.prepare("PRAGMA index_list(provider_connections)").all() as Array<{ name: string }> @@ -163,3 +165,20 @@ test("ensureProviderConnectionsColumns restores base columns required by later m db.close?.(); } }); + +test("ensureProviderConnectionsColumns back-fills last_ping columns on a pre-123 lineage", () => { + const db = openMemoryDb(); + try { + db.exec("CREATE TABLE provider_connections (id TEXT PRIMARY KEY, provider TEXT NOT NULL)"); + assert.equal(hasColumn(db, "provider_connections", "last_ping_at"), false); + assert.equal(hasColumn(db, "provider_connections", "last_pinged_reset_key"), false); + + ensureProviderConnectionsColumns(db); + + assert.equal(hasColumn(db, "provider_connections", "last_ping_at"), true); + assert.equal(hasColumn(db, "provider_connections", "last_pinged_reset_key"), true); + assert.doesNotThrow(() => ensureProviderConnectionsColumns(db)); + } finally { + db.close?.(); + } +}); From 82f78b3b3b8cdb861f5bc3d804494dbd51687bb3 Mon Sep 17 00:00:00 2001 From: Koosha Paridehpour <42529354+KooshaPari@users.noreply.github.com> Date: Fri, 4 Sep 2026 22:33:33 -0700 Subject: [PATCH 104/143] fix(api/pricing): surface validation error message as string, not raw object (#12494) (#12771) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged, with one adjustment. Confirmed the bug end to end: `PricingTab.tsx:369` types the payload as `{ error?: string }` and feeds it to `new Error(errorPayload.error || ...)`, so the `{ message, details }` object landed in the toast as `[object Object]` — exactly what #12494 reported. The one change I made before merging: `validation.error.message` is the fixed constant `"Invalid request"` (see `validateBody` in `src/shared/validation/helpers.ts:44`), so it would have swapped an unreadable toast for an uninformative one. The repo already has `formatValidationMessage()`, added in #10849 for precisely this case — it returns `"field: reason"` naming the first offending field. Merged with that instead, so a bad pricing value now says which field it was. Validated on `release/v3.8.51`: `typecheck:core` clean, `check-file-size` OK. Rebased onto the release branch — the PR was cut from `main`, which is ~3695 commits behind the active branch. Thank you for the report and the fix. --- src/app/api/pricing/route.ts | 16 ++++++++++++++-- 1 file changed, 14 insertions(+), 2 deletions(-) diff --git a/src/app/api/pricing/route.ts b/src/app/api/pricing/route.ts index e1849e672b..e062ef3530 100644 --- a/src/app/api/pricing/route.ts +++ b/src/app/api/pricing/route.ts @@ -8,7 +8,11 @@ import { resetAllPricing, } from "@/lib/db/settings"; import { updatePricingSchema } from "@/shared/validation/schemas"; -import { isValidationFailure, validateBody } from "@/shared/validation/helpers"; +import { + formatValidationMessage, + isValidationFailure, + validateBody, +} from "@/shared/validation/helpers"; /** * GET /api/pricing @@ -59,7 +63,15 @@ export async function PATCH(request) { try { const validation = validateBody(updatePricingSchema, rawBody); if (isValidationFailure(validation)) { - return NextResponse.json({ error: validation.error }, { status: 400 }); + // #12494: PricingTab reads this payload as `{ error?: string }` and feeds it + // straight to `new Error(...)`, so handing back the `{ message, details }` + // object rendered as "Falha ao salvar preços: [object Object]". Send a string. + // `formatValidationMessage` names the offending field ("field: reason") instead + // of the bare "Invalid request" constant, so the toast stays actionable (#10849). + return NextResponse.json( + { error: formatValidationMessage(validation.error) }, + { status: 400 } + ); } const body = validation.data; From 366099a08cadf8789437e95355238807ac831eb6 Mon Sep 17 00:00:00 2001 From: Koosha Paridehpour <42529354+KooshaPari@users.noreply.github.com> Date: Fri, 4 Sep 2026 22:33:37 -0700 Subject: [PATCH 105/143] fix(i18n): quote placeholder in OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES description (#12505) (#12769) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged, with the fix moved to where the bug actually lives — and thank you, because the issue analysis in #12505 is what made that possible. The diagnosis was right: `FeatureFlagsGrid.tsx:422-428` renders descriptions through a plain `t()`, so next-intl compiles the value as ICU and a bare `` parses as an unknown rich-text tag. But the branch changed `src/shared/constants/featureFlagDefinitions.ts` — the TS default, which is the `flag.description` **fallback** rendered raw, never through ICU. Two consequences: the reported bug stayed live (all 42 locale files still carried the raw tag — `grep -l "profiles//settings.json" src/i18n/messages/*.json` returned 42, and 0 for the escaped form), and the quotes would have shown up literally in the one place that string does render. So this merge reverts the TS default to the raw path and applies the ICU escape to the 42 locale files instead — follow-up 1 from your issue, inverted to hit the file that matters. I also added follow-up 2 as a real guard: `tests/unit/feature-flag-description-icu-parse-12505.test.ts` compiles every `featureFlags.definitions.*` message in every locale through `intl-messageformat` (the parser next-intl uses) and asserts the placeholder renders as a literal ``. Verified red-then-green — reverting `en.json` alone fails both cases; restored, 2/2 pass. Validated on `release/v3.8.51`: all locale files re-parse as valid JSON, `typecheck:core` clean, `check-file-size` OK. `i18n:check` drift is pre-existing on the tip, unrelated. Closes #12505. --- src/i18n/messages/ar.json | 2 +- src/i18n/messages/az.json | 2 +- src/i18n/messages/bg.json | 2 +- src/i18n/messages/bn.json | 2 +- src/i18n/messages/cs.json | 2 +- src/i18n/messages/da.json | 2 +- src/i18n/messages/de.json | 2 +- src/i18n/messages/en.json | 2 +- src/i18n/messages/es.json | 2 +- src/i18n/messages/fa.json | 2 +- src/i18n/messages/fi.json | 2 +- src/i18n/messages/fr.json | 2 +- src/i18n/messages/gu.json | 2 +- src/i18n/messages/he.json | 2 +- src/i18n/messages/hi.json | 2 +- src/i18n/messages/hu.json | 2 +- src/i18n/messages/id.json | 2 +- src/i18n/messages/it.json | 2 +- src/i18n/messages/ja.json | 2 +- src/i18n/messages/ko.json | 2 +- src/i18n/messages/mr.json | 2 +- src/i18n/messages/ms.json | 2 +- src/i18n/messages/nl.json | 2 +- src/i18n/messages/no.json | 2 +- src/i18n/messages/phi.json | 2 +- src/i18n/messages/pl.json | 2 +- src/i18n/messages/pt-BR.json | 2 +- src/i18n/messages/pt.json | 2 +- src/i18n/messages/ro.json | 2 +- src/i18n/messages/ru.json | 2 +- src/i18n/messages/sk.json | 2 +- src/i18n/messages/sv.json | 2 +- src/i18n/messages/sw.json | 2 +- src/i18n/messages/ta.json | 2 +- src/i18n/messages/te.json | 2 +- src/i18n/messages/th.json | 2 +- src/i18n/messages/tr.json | 2 +- src/i18n/messages/uk-UA.json | 2 +- src/i18n/messages/ur.json | 2 +- src/i18n/messages/vi.json | 2 +- src/i18n/messages/zh-CN.json | 2 +- src/i18n/messages/zh-TW.json | 2 +- ...e-flag-description-icu-parse-12505.test.ts | 68 +++++++++++++++++++ 43 files changed, 110 insertions(+), 42 deletions(-) create mode 100644 tests/unit/feature-flag-description-icu-parse-12505.test.ts diff --git a/src/i18n/messages/ar.json b/src/i18n/messages/ar.json index 47efa8dc95..0f44ff6a8d 100644 --- a/src/i18n/messages/ar.json +++ b/src/i18n/messages/ar.json @@ -12982,7 +12982,7 @@ "description": "بعد مزامنة المزود والنموذج، أعد إنشاء ملفات التعريف ~/.codex/*.config.toml من الكتالوج المباشر. لا يغير هذا أبداً تكوين Codex النشط أو الافتراضي ويكون معطلاً بشكل افتراضي." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "بعد مزامنة المزود والنموذج، أعد إنشاء ملفات التعريف ~/.claude/profiles//settings.json من الكتالوج المباشر. لا يغير هذا أبداً تكوين Claude النشط أو الافتراضي ويكون معطلاً بشكل افتراضي." + "description": "بعد مزامنة المزود والنموذج، أعد إنشاء ملفات التعريف ~/.claude/profiles/''/settings.json من الكتالوج المباشر. لا يغير هذا أبداً تكوين Claude النشط أو الافتراضي ويكون معطلاً بشكل افتراضي." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "تعطيل نقطة نهاية فحص صحة المثيل المحلي." diff --git a/src/i18n/messages/az.json b/src/i18n/messages/az.json index 9f9efb9640..f054911477 100644 --- a/src/i18n/messages/az.json +++ b/src/i18n/messages/az.json @@ -12982,7 +12982,7 @@ "description": "Provayder-model sinxronizasiyasından sonra canlı kataloqdan ~/.codex/*.config.toml profillərini yenidən yaradın. Bu, heç vaxt aktiv və ya defolt Codex konfiqurasiyasını dəyişmir və defolt olaraq qapalıdır." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Provayder-model sinxronizasiyasından sonra canlı kataloqdan ~/.claude/profiles//settings.json profillərini yenidən yaradın. Bu, heç vaxt aktiv və ya defolt Claude konfiqurasiyasını dəyişmir və defolt olaraq qapalıdır." + "description": "Provayder-model sinxronizasiyasından sonra canlı kataloqdan ~/.claude/profiles/''/settings.json profillərini yenidən yaradın. Bu, heç vaxt aktiv və ya defolt Claude konfiqurasiyasını dəyişmir və defolt olaraq qapalıdır." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Yerli instansiyanın sağlamlıq yoxlaması son nöqtəsini sıradan çıxarın." diff --git a/src/i18n/messages/bg.json b/src/i18n/messages/bg.json index e442c29e0d..ced22ca308 100644 --- a/src/i18n/messages/bg.json +++ b/src/i18n/messages/bg.json @@ -12982,7 +12982,7 @@ "description": "След синхронизиране на доставчик-модел, регенериране на ~/.codex/*.config.toml профили от каталога на живо. Това никога не променя активната или подразбиращата се конфигурация на Codex и е изключено по подразбиране." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "After provider-model synchronization, regenerate ~/.claude/profiles//settings.json profiles from the live catalog. This never changes the active or default Claude configuration and is off by default." + "description": "After provider-model synchronization, regenerate ~/.claude/profiles/''/settings.json profiles from the live catalog. This never changes the active or default Claude configuration and is off by default." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Деактивиране на крайната точка за проверка на състоянието на локалния екземпляр." diff --git a/src/i18n/messages/bn.json b/src/i18n/messages/bn.json index fb1e088781..f592a899d4 100644 --- a/src/i18n/messages/bn.json +++ b/src/i18n/messages/bn.json @@ -12982,7 +12982,7 @@ "description": "প্রোভাইডার-মডেল সিঙ্ক্রোনাইজেশনের পরে, লাইভ ক্যাটালগ থেকে ~/.codex/*.config.toml প্রোফাইলগুলি পুনরায় তৈরি করুন। এটি সক্রিয় বা ডিফল্ট Codex কনফিগারেশন কখনই পরিবর্তন করে না এবং ডিফল্টভাবে বন্ধ থাকে।" }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "প্রোভাইডার-মডেল সিঙ্ক্রোনাইজেশনের পরে, লাইভ ক্যাটালগ থেকে ~/.claude/profiles//settings.json প্রোফাইলগুলি পুনরায় তৈরি করুন। এটি সক্রিয় বা ডিফল্ট Claude কনফিগারেশন কখনই পরিবর্তন করে না এবং ডিফল্টভাবে বন্ধ থাকে।" + "description": "প্রোভাইডার-মডেল সিঙ্ক্রোনাইজেশনের পরে, লাইভ ক্যাটালগ থেকে ~/.claude/profiles/''/settings.json প্রোফাইলগুলি পুনরায় তৈরি করুন। এটি সক্রিয় বা ডিফল্ট Claude কনফিগারেশন কখনই পরিবর্তন করে না এবং ডিফল্টভাবে বন্ধ থাকে।" }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "স্থানীয় ইনস্ট্যান্স হেলথ-চেক এন্ডপয়েন্ট নিষ্ক্রিয় করুন।" diff --git a/src/i18n/messages/cs.json b/src/i18n/messages/cs.json index 14bd25e496..c2ef9a2d68 100644 --- a/src/i18n/messages/cs.json +++ b/src/i18n/messages/cs.json @@ -12982,7 +12982,7 @@ "description": "Po synchronizaci poskytovatelů a modelů regenerovat profily ~/.codex/*.config.toml z živého katalogu. Toto nikdy nemění aktivní nebo výchozí konfiguraci Codexu a je ve výchozím nastavení vypnuto." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Po synchronizaci poskytovatelů a modelů regenerovat profily ~/.claude/profiles//settings.json z živého katalogu. Toto nikdy nemění aktivní nebo výchozí konfiguraci Claude a je ve výchozím nastavení vypnuto." + "description": "Po synchronizaci poskytovatelů a modelů regenerovat profily ~/.claude/profiles/''/settings.json z živého katalogu. Toto nikdy nemění aktivní nebo výchozí konfiguraci Claude a je ve výchozím nastavení vypnuto." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Zakázat koncový bod kontroly stavu lokální instance." diff --git a/src/i18n/messages/da.json b/src/i18n/messages/da.json index 7309787fd2..2533f08ad3 100644 --- a/src/i18n/messages/da.json +++ b/src/i18n/messages/da.json @@ -12982,7 +12982,7 @@ "description": "Efter udbyder-model-synkronisering regenereres ~/.codex/*.config.toml-profiler fra det aktive katalog. Dette ændrer aldrig den aktive eller standardmæssige Codex-konfiguration og er deaktiveret som standard." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Efter udbyder-model-synkronisering regenereres ~/.claude/profiles//settings.json-profiler fra det aktive katalog. Dette ændrer aldrig den aktive eller standardmæssige Claude-konfiguration og er deaktiveret som standard." + "description": "Efter udbyder-model-synkronisering regenereres ~/.claude/profiles/''/settings.json-profiler fra det aktive katalog. Dette ændrer aldrig den aktive eller standardmæssige Claude-konfiguration og er deaktiveret som standard." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Deaktivér slutpunktet for den lokale instans' tilstandstjek." diff --git a/src/i18n/messages/de.json b/src/i18n/messages/de.json index 41ea050e7b..62e80a3082 100644 --- a/src/i18n/messages/de.json +++ b/src/i18n/messages/de.json @@ -12989,7 +12989,7 @@ "description": "Nach der Anbieter-Modell-Synchronisierung ~/.codex/*.config.toml-Profile aus dem Live-Katalog neu generieren. Dies ändert niemals die aktive oder Standard-Codex-Konfiguration und ist standardmäßig deaktiviert." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Nach der Anbieter-Modell-Synchronisierung ~/.claude/profiles//settings.json-Profile aus dem Live-Katalog neu generieren. Dies ändert niemals die aktive oder Standard-Claude-Konfiguration und ist standardmäßig deaktiviert." + "description": "Nach der Anbieter-Modell-Synchronisierung ~/.claude/profiles/''/settings.json-Profile aus dem Live-Katalog neu generieren. Dies ändert niemals die aktive oder Standard-Claude-Konfiguration und ist standardmäßig deaktiviert." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Den Health-Check-Endpunkt der lokalen Instanz deaktivieren." diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index ae4b3b28fe..6d89d67153 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -12992,7 +12992,7 @@ "description": "After provider-model synchronization, regenerate ~/.codex/*.config.toml profiles from the live catalog. This never changes the active or default Codex configuration and is off by default." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "After provider-model synchronization, regenerate ~/.claude/profiles//settings.json profiles from the live catalog. This never changes the active or default Claude configuration and is off by default." + "description": "After provider-model synchronization, regenerate ~/.claude/profiles/''/settings.json profiles from the live catalog. This never changes the active or default Claude configuration and is off by default." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Disable the local instance health-check endpoint." diff --git a/src/i18n/messages/es.json b/src/i18n/messages/es.json index 348db0b960..6dc13a0fde 100644 --- a/src/i18n/messages/es.json +++ b/src/i18n/messages/es.json @@ -12982,7 +12982,7 @@ "description": "After provider-model synchronization, regenerate ~/.codex/*.config.toml profiles from the live catalog. This never changes the active or default Codex configuration and is off by default." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "After provider-model synchronization, regenerate ~/.claude/profiles//settings.json profiles from the live catalog. This never changes the active or default Claude configuration and is off by default." + "description": "After provider-model synchronization, regenerate ~/.claude/profiles/''/settings.json profiles from the live catalog. This never changes the active or default Claude configuration and is off by default." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Disable the local instance health-check endpoint." diff --git a/src/i18n/messages/fa.json b/src/i18n/messages/fa.json index 0fa364611b..6f9678b74f 100644 --- a/src/i18n/messages/fa.json +++ b/src/i18n/messages/fa.json @@ -12982,7 +12982,7 @@ "description": "پس از همگام‌سازی ارائه‌دهنده-مدل، پروفایل‌های ~/.codex/*.config.toml را از کاتالوگ زنده بازسازی کنید. این کار هرگز پیکربندی فعال یا پیش‌فرض Codex را تغییر نمی‌دهد و به طور پیش‌فرض غیرفعال است." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "پس از همگام‌سازی ارائه‌دهنده-مدل، پروفایل‌های ~/.claude/profiles//settings.json را از کاتالوگ زنده بازسازی کنید. این کار هرگز پیکربندی فعال یا پیش‌فرض Claude را تغییر نمی‌دهد و به طور پیش‌فرض غیرفعال است." + "description": "پس از همگام‌سازی ارائه‌دهنده-مدل، پروفایل‌های ~/.claude/profiles/''/settings.json را از کاتالوگ زنده بازسازی کنید. این کار هرگز پیکربندی فعال یا پیش‌فرض Claude را تغییر نمی‌دهد و به طور پیش‌فرض غیرفعال است." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "غیرفعال کردن نقطه پایانی بررسی سلامت نمونه محلی." diff --git a/src/i18n/messages/fi.json b/src/i18n/messages/fi.json index e5a09ea7dc..b7a930e02c 100644 --- a/src/i18n/messages/fi.json +++ b/src/i18n/messages/fi.json @@ -12982,7 +12982,7 @@ "description": "Tarjoajamallien synkronoinnin jälkeen luo ~/.codex/*.config.toml -profiilit uudelleen reaaliaikaisesta luettelosta. Tämä ei koskaan muuta aktiivista tai oletusarvoista Codex-konfiguraatiota ja on oletuksena pois päältä." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Tarjoajamallien synkronoinnin jälkeen luo ~/.claude/profiles//settings.json -profiilit uudelleen reaaliaikaisesta luettelosta. Tämä ei koskaan muuta aktiivista tai oletusarvoista Claude-konfiguraatiota ja on oletuksena pois päältä." + "description": "Tarjoajamallien synkronoinnin jälkeen luo ~/.claude/profiles/''/settings.json -profiilit uudelleen reaaliaikaisesta luettelosta. Tämä ei koskaan muuta aktiivista tai oletusarvoista Claude-konfiguraatiota ja on oletuksena pois päältä." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Poista käytöstä paikallisen instanssin terveystarkistuksen päätepiste." diff --git a/src/i18n/messages/fr.json b/src/i18n/messages/fr.json index 48bcdae45a..f3cb535451 100644 --- a/src/i18n/messages/fr.json +++ b/src/i18n/messages/fr.json @@ -12982,7 +12982,7 @@ "description": "Après la synchronisation fournisseur-modèle, régénérer les profils ~/.codex/*.config.toml à partir du catalogue en direct. Cela ne modifie jamais la configuration Codex active ou par défaut et est désactivé par défaut." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Après la synchronisation fournisseur-modèle, régénérer les profils ~/.claude/profiles//settings.json à partir du catalogue en direct. Cela ne modifie jamais la configuration Claude active ou par défaut et est désactivé par défaut." + "description": "Après la synchronisation fournisseur-modèle, régénérer les profils ~/.claude/profiles/''/settings.json à partir du catalogue en direct. Cela ne modifie jamais la configuration Claude active ou par défaut et est désactivé par défaut." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Désactiver le point de terminaison de vérification de l'état de l'instance locale." diff --git a/src/i18n/messages/gu.json b/src/i18n/messages/gu.json index c9014de70b..abe98e5397 100644 --- a/src/i18n/messages/gu.json +++ b/src/i18n/messages/gu.json @@ -12982,7 +12982,7 @@ "description": "પ્રદાતા-મોડલ સિંક્રનાઇઝેશન પછી, લાઇવ કૅટેલોગમાંથી ~/.codex/*.config.toml પ્રોફાઇલ્સ ફરીથી જનરેટ કરો. આ ક્યારેય સક્રિય અથવા ડિફોલ્ટ Codex રૂપરેખાંકનને બદલતું નથી અને ડિફોલ્ટ રૂપે off હોય છે." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "પ્રદાતા-મોડલ સિંક્રનાઇઝેશન પછી, લાઇવ કૅટેલોગમાંથી ~/.claude/profiles//settings.json પ્રોફાઇલ્સ ફરીથી જનરેટ કરો. આ ક્યારેય સક્રિય અથવા ડિફોલ્ટ Claude રૂપરેખાંકનને બદલતું નથી અને ડિફોલ્ટ રૂપે off હોય છે." + "description": "પ્રદાતા-મોડલ સિંક્રનાઇઝેશન પછી, લાઇવ કૅટેલોગમાંથી ~/.claude/profiles/''/settings.json પ્રોફાઇલ્સ ફરીથી જનરેટ કરો. આ ક્યારેય સક્રિય અથવા ડિફોલ્ટ Claude રૂપરેખાંકનને બદલતું નથી અને ડિફોલ્ટ રૂપે off હોય છે." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "સ્થાનિક ઇન્સ્ટન્સ હેલ્થ-ચેક એન્ડપોઇન્ટ નિષ્ક્રિય કરો." diff --git a/src/i18n/messages/he.json b/src/i18n/messages/he.json index 8305333bb0..0c6c5933e7 100644 --- a/src/i18n/messages/he.json +++ b/src/i18n/messages/he.json @@ -12982,7 +12982,7 @@ "description": "לאחר סנכרון ספק-מודל, יצירה מחדש של פרופילי ~/.codex/*.config.toml מתוך הקטלוג הפעיל. פעולה זו אינה משנה לעולם את תצורת Codex הפעילה או כברירת מחדל, והיא כבויה כברירת מחדל." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "לאחר סנכרון ספק-מודל, יצירה מחדש של פרופילי ~/.claude/profiles//settings.json מתוך הקטלוג הפעיל. פעולה זו אינה משנה לעולם את תצורת Claude הפעילה או כברירת מחדל, והיא כבויה כברירת מחדל." + "description": "לאחר סנכרון ספק-מודל, יצירה מחדש של פרופילי ~/.claude/profiles/''/settings.json מתוך הקטלוג הפעיל. פעולה זו אינה משנה לעולם את תצורת Claude הפעילה או כברירת מחדל, והיא כבויה כברירת מחדל." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "השבתת נקודת הקצה לבדיקת תקינות של המופע המקומי." diff --git a/src/i18n/messages/hi.json b/src/i18n/messages/hi.json index 4a7226cbfe..0d761462ca 100644 --- a/src/i18n/messages/hi.json +++ b/src/i18n/messages/hi.json @@ -12982,7 +12982,7 @@ "description": "प्रदाता-मॉडल सिंक्रनाइज़ेशन के बाद, लाइव कैटलॉग से ~/.codex/*.config.toml प्रोफाइल को पुनरुत्पादित करें। यह सक्रिय या डिफ़ॉल्ट Codex कॉन्फ़िगरेशन को कभी नहीं बदलता है और डिफ़ॉल्ट रूप से बंद रहता है।" }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "प्रदाता-मॉडल सिंक्रनाइज़ेशन के बाद, लाइव कैटलॉग से ~/.claude/profiles//settings.json प्रोफाइल को पुनरुत्पादित करें। यह सक्रिय या डिफ़ॉल्ट Claude कॉन्फ़िगरेशन को कभी नहीं बदलता है और डिफ़ॉल्ट रूप से बंद रहता है।" + "description": "प्रदाता-मॉडल सिंक्रनाइज़ेशन के बाद, लाइव कैटलॉग से ~/.claude/profiles/''/settings.json प्रोफाइल को पुनरुत्पादित करें। यह सक्रिय या डिफ़ॉल्ट Claude कॉन्फ़िगरेशन को कभी नहीं बदलता है और डिफ़ॉल्ट रूप से बंद रहता है।" }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "स्थानीय इंस्टेंस स्वास्थ्य-जांच एंडपॉइंट को अक्षम करें।" diff --git a/src/i18n/messages/hu.json b/src/i18n/messages/hu.json index 99d95becd5..192aab7262 100644 --- a/src/i18n/messages/hu.json +++ b/src/i18n/messages/hu.json @@ -12982,7 +12982,7 @@ "description": "A szolgáltató-modell szinkronizálás után a ~/.codex/*.config.toml profilok újragenerálása az élő katalógusból. Ez soha nem változtatja meg az aktív vagy alapértelmezett Codex konfigurációt, és alapértelmezés szerint ki van kapcsolva." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "A szolgáltató-modell szinkronizálás után a ~/.claude/profiles//settings.json profilok újragenerálása az élő katalógusból. Ez soha nem változtatja meg az aktív vagy alapértelmezett Claude konfigurációt, és alapértelmezés szerint ki van kapcsolva." + "description": "A szolgáltató-modell szinkronizálás után a ~/.claude/profiles/''/settings.json profilok újragenerálása az élő katalógusból. Ez soha nem változtatja meg az aktív vagy alapértelmezett Claude konfigurációt, és alapértelmezés szerint ki van kapcsolva." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "A helyi példány állapotellenőrző végpontjának letiltása." diff --git a/src/i18n/messages/id.json b/src/i18n/messages/id.json index 6ac4539837..749c54878f 100644 --- a/src/i18n/messages/id.json +++ b/src/i18n/messages/id.json @@ -12982,7 +12982,7 @@ "description": "Setelah sinkronisasi model penyedia, buat ulang profil ~/.codex/*.config.toml dari katalog langsung. Ini tidak pernah mengubah konfigurasi Codex yang aktif atau default dan nonaktif secara default." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Setelah sinkronisasi model penyedia, buat ulang profil ~/.claude/profiles//settings.json dari katalog langsung. Ini tidak pernah mengubah konfigurasi Claude yang aktif atau default dan nonaktif secara default." + "description": "Setelah sinkronisasi model penyedia, buat ulang profil ~/.claude/profiles/''/settings.json dari katalog langsung. Ini tidak pernah mengubah konfigurasi Claude yang aktif atau default dan nonaktif secara default." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Nonaktifkan titik akhir health-check instans lokal." diff --git a/src/i18n/messages/it.json b/src/i18n/messages/it.json index 7bf00a4c77..548717ca56 100644 --- a/src/i18n/messages/it.json +++ b/src/i18n/messages/it.json @@ -12982,7 +12982,7 @@ "description": "Dopo la sincronizzazione provider-modello, rigenera i profili ~/.codex/*.config.toml dal catalogo live. Questa operazione non modifica mai la configurazione attiva o predefinita di Codex ed è disattivata per impostazione predefinita." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Dopo la sincronizzazione provider-modello, rigenera i profili ~/.claude/profiles//settings.json dal catalogo live. Questa operazione non modifica mai la configurazione attiva o predefinita di Claude ed è disattivata per impostazione predefinita." + "description": "Dopo la sincronizzazione provider-modello, rigenera i profili ~/.claude/profiles/''/settings.json dal catalogo live. Questa operazione non modifica mai la configurazione attiva o predefinita di Claude ed è disattivata per impostazione predefinita." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Disabilita l'endpoint di health check dell'istanza locale." diff --git a/src/i18n/messages/ja.json b/src/i18n/messages/ja.json index 0948496009..8ab091a78c 100644 --- a/src/i18n/messages/ja.json +++ b/src/i18n/messages/ja.json @@ -12982,7 +12982,7 @@ "description": "After provider-model synchronization, regenerate ~/.codex/*.config.toml profiles from the live catalog. This never changes the active or default Codex configuration and is off by default." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "After provider-model synchronization, regenerate ~/.claude/profiles//settings.json profiles from the live catalog. This never changes the active or default Claude configuration and is off by default." + "description": "After provider-model synchronization, regenerate ~/.claude/profiles/''/settings.json profiles from the live catalog. This never changes the active or default Claude configuration and is off by default." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Disable the local instance health-check endpoint." diff --git a/src/i18n/messages/ko.json b/src/i18n/messages/ko.json index 63d0b70434..7a768219e7 100644 --- a/src/i18n/messages/ko.json +++ b/src/i18n/messages/ko.json @@ -12982,7 +12982,7 @@ "description": "공급자-모델 동기화 후 라이브 카탈로그에서 ~/.codex/*.config.toml 프로필을 재생성합니다. 이는 활성 또는 기본 Codex 구성을 변경하지 않으며 기본적으로 꺼져 있습니다." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "공급자-모델 동기화 후 라이브 카탈로그에서 ~/.claude/profiles//settings.json 프로필을 재생성합니다. 이는 활성 또는 기본 Claude 구성을 변경하지 않으며 기본적으로 꺼져 있습니다." + "description": "공급자-모델 동기화 후 라이브 카탈로그에서 ~/.claude/profiles/''/settings.json 프로필을 재생성합니다. 이는 활성 또는 기본 Claude 구성을 변경하지 않으며 기본적으로 꺼져 있습니다." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "로컬 인스턴스 상태 확인 엔드포인트를 비활성화합니다." diff --git a/src/i18n/messages/mr.json b/src/i18n/messages/mr.json index 140afbdda6..37934d34a8 100644 --- a/src/i18n/messages/mr.json +++ b/src/i18n/messages/mr.json @@ -12982,7 +12982,7 @@ "description": "प्रदाता-मॉडेल सिंक्रोनाइझेशननंतर, लाइव्ह कॅटलॉगमधून ~/.codex/*.config.toml प्रोफाइल्स पुन्हा तयार करा. हे सक्रिय किंवा डीफॉल्ट Codex कॉन्फिगरेशन कधीही बदलत नाही आणि डीफॉल्टनुसार बंद असते." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "प्रदाता-मॉडेल सिंक्रोनाइझेशननंतर, लाइव्ह कॅटलॉगमधून ~/.claude/profiles//settings.json प्रोफाइल्स पुन्हा तयार करा. हे सक्रिय किंवा डीफॉल्ट Claude कॉन्फिगरेशन कधीही बदलत नाही आणि डीफॉल्टनुसार बंद असते." + "description": "प्रदाता-मॉडेल सिंक्रोनाइझेशननंतर, लाइव्ह कॅटलॉगमधून ~/.claude/profiles/''/settings.json प्रोफाइल्स पुन्हा तयार करा. हे सक्रिय किंवा डीफॉल्ट Claude कॉन्फिगरेशन कधीही बदलत नाही आणि डीफॉल्टनुसार बंद असते." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "स्थानिक इन्स्टन्स हेल्थ-चेक एंडपॉइंट अक्षम करा." diff --git a/src/i18n/messages/ms.json b/src/i18n/messages/ms.json index 66e04377f7..8580748629 100644 --- a/src/i18n/messages/ms.json +++ b/src/i18n/messages/ms.json @@ -12982,7 +12982,7 @@ "description": "Selepas penyegerakan penyedia-model, jana semula profil ~/.codex/*.config.toml daripada katalog langsung. Ini tidak pernah mengubah konfigurasi Codex yang aktif atau lalai dan dinyahdayakan secara lalai." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Selepas penyegerakan penyedia-model, jana semula profil ~/.claude/profiles//settings.json daripada katalog langsung. Ini tidak pernah mengubah konfigurasi Claude yang aktif atau lalai dan dinyahdayakan secara lalai." + "description": "Selepas penyegerakan penyedia-model, jana semula profil ~/.claude/profiles/''/settings.json daripada katalog langsung. Ini tidak pernah mengubah konfigurasi Claude yang aktif atau lalai dan dinyahdayakan secara lalai." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Nyahdayakan titik akhir pemeriksaan kesihatan tika tempatan." diff --git a/src/i18n/messages/nl.json b/src/i18n/messages/nl.json index b16e43b7d8..bc13adaabe 100644 --- a/src/i18n/messages/nl.json +++ b/src/i18n/messages/nl.json @@ -12982,7 +12982,7 @@ "description": "Genereer na provider-modelsynchronisatie ~/.codex/*.config.toml-profielen opnieuw vanuit de live catalogus. Dit wijzigt nooit de actieve of standaard Codex-configuratie en is standaard uitgeschakeld." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Genereer na provider-modelsynchronisatie ~/.claude/profiles//settings.json-profielen opnieuw vanuit de live catalogus. Dit wijzigt nooit de actieve of standaard Claude-configuratie en is standaard uitgeschakeld." + "description": "Genereer na provider-modelsynchronisatie ~/.claude/profiles/''/settings.json-profielen opnieuw vanuit de live catalogus. Dit wijzigt nooit de actieve of standaard Claude-configuratie en is standaard uitgeschakeld." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Schakel het health-check-eindpunt van de lokale instantie uit." diff --git a/src/i18n/messages/no.json b/src/i18n/messages/no.json index 3e0aa602a7..c84dcf1a37 100644 --- a/src/i18n/messages/no.json +++ b/src/i18n/messages/no.json @@ -12982,7 +12982,7 @@ "description": "Etter synkronisering av leverandørmodell, regenerer ~/.codex/*.config.toml-profiler fra den aktive katalogen. Dette endrer aldri den aktive eller standard Codex-konfigurasjonen og er av som standard." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Etter synkronisering av leverandørmodell, regenerer ~/.claude/profiles//settings.json-profiler fra den aktive katalogen. Dette endrer aldri den aktive eller standard Claude-konfigurasjonen og er av som standard." + "description": "Etter synkronisering av leverandørmodell, regenerer ~/.claude/profiles/''/settings.json-profiler fra den aktive katalogen. Dette endrer aldri den aktive eller standard Claude-konfigurasjonen og er av som standard." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Deaktiver helsesjekk-endepunktet for den lokale instansen." diff --git a/src/i18n/messages/phi.json b/src/i18n/messages/phi.json index 282dc3c6f2..00a40118e9 100644 --- a/src/i18n/messages/phi.json +++ b/src/i18n/messages/phi.json @@ -12982,7 +12982,7 @@ "description": "Pagkatapos ng pag-synchronize ng provider-model, muling buuin ang mga profile ng ~/.codex/*.config.toml mula sa live catalog. Hindi nito kailanman binabago ang aktibo o default na configuration ng Codex at naka-off bilang default." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Pagkatapos ng pag-synchronize ng provider-model, muling buuin ang mga profile ng ~/.claude/profiles//settings.json mula sa live catalog. Hindi nito kailanman binabago ang aktibo o default na configuration ng Claude at naka-off bilang default." + "description": "Pagkatapos ng pag-synchronize ng provider-model, muling buuin ang mga profile ng ~/.claude/profiles/''/settings.json mula sa live catalog. Hindi nito kailanman binabago ang aktibo o default na configuration ng Claude at naka-off bilang default." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "I-disable ang health-check endpoint ng lokal na instance." diff --git a/src/i18n/messages/pl.json b/src/i18n/messages/pl.json index 39b053ca78..680137f445 100644 --- a/src/i18n/messages/pl.json +++ b/src/i18n/messages/pl.json @@ -12982,7 +12982,7 @@ "description": "Po synchronizacji dostawców i modeli wygeneruj ponownie profile ~/.codex/*.config.toml z aktywnego katalogu. To nigdy nie zmienia aktywnej ani domyślnej konfiguracji Codex i jest domyślnie wyłączone." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Po synchronizacji dostawców i modeli wygeneruj ponownie profile ~/.claude/profiles//settings.json z aktywnego katalogu. To nigdy nie zmienia aktywnej ani domyślnej konfiguracji Claude i jest domyślnie wyłączone." + "description": "Po synchronizacji dostawców i modeli wygeneruj ponownie profile ~/.claude/profiles/''/settings.json z aktywnego katalogu. To nigdy nie zmienia aktywnej ani domyślnej konfiguracji Claude i jest domyślnie wyłączone." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Wyłącz punkt końcowy sprawdzania stanu lokalnej instancji." diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index c50d940f5e..c4c0bab8c4 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -12993,7 +12993,7 @@ "description": "Após a sincronização de modelos do provedor, regenera os perfis ~/.codex/*.config.toml a partir do catálogo ativo. Isso nunca altera a configuração ativa ou padrão do Codex e está desativado por padrão." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Após a sincronização de modelos do provedor, regenera os perfis ~/.claude/profiles//settings.json a partir do catálogo ativo. Isso nunca altera a configuração ativa ou padrão do Claude e está desativado por padrão." + "description": "Após a sincronização de modelos do provedor, regenera os perfis ~/.claude/profiles/''/settings.json a partir do catálogo ativo. Isso nunca altera a configuração ativa ou padrão do Claude e está desativado por padrão." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Desativa o endpoint de verificação de saúde da instância local." diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index aa75826e5f..43a27138b1 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -12982,7 +12982,7 @@ "description": "Após a sincronização de fornecedor-modelo, regenerar os perfis ~/.codex/*.config.toml a partir do catálogo ativo. Isto nunca altera a configuração ativa ou predefinida do Codex e está desativado por predefinição." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Após a sincronização de fornecedor-modelo, regenerar os perfis ~/.claude/profiles//settings.json a partir do catálogo ativo. Isto nunca altera a configuração ativa ou predefinida do Claude e está desativado por predefinição." + "description": "Após a sincronização de fornecedor-modelo, regenerar os perfis ~/.claude/profiles/''/settings.json a partir do catálogo ativo. Isto nunca altera a configuração ativa ou predefinida do Claude e está desativado por predefinição." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Desativar o endpoint de health-check da instância local." diff --git a/src/i18n/messages/ro.json b/src/i18n/messages/ro.json index 4475a38325..87f1f5f75f 100644 --- a/src/i18n/messages/ro.json +++ b/src/i18n/messages/ro.json @@ -12982,7 +12982,7 @@ "description": "După sincronizarea furnizor-model, regenerează profilurile ~/.codex/*.config.toml din catalogul live. Acest lucru nu modifică niciodată configurația Codex activă sau implicită și este dezactivat în mod implicit." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "După sincronizarea furnizor-model, regenerează profilurile ~/.claude/profiles//settings.json din catalogul live. Acest lucru nu modifică niciodată configurația Claude activă sau implicită și este dezactivat în mod implicit." + "description": "După sincronizarea furnizor-model, regenerează profilurile ~/.claude/profiles/''/settings.json din catalogul live. Acest lucru nu modifică niciodată configurația Claude activă sau implicită și este dezactivat în mod implicit." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Dezactivează endpoint-ul de health-check al instanței locale." diff --git a/src/i18n/messages/ru.json b/src/i18n/messages/ru.json index d5ca918520..1a8ff09e40 100644 --- a/src/i18n/messages/ru.json +++ b/src/i18n/messages/ru.json @@ -12982,7 +12982,7 @@ "description": "After provider-model synchronization, regenerate ~/.codex/*.config.toml profiles from the live catalog. This never changes the active or default Codex configuration and is off by default." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "After provider-model synchronization, regenerate ~/.claude/profiles//settings.json profiles from the live catalog. This never changes the active or default Claude configuration and is off by default." + "description": "After provider-model synchronization, regenerate ~/.claude/profiles/''/settings.json profiles from the live catalog. This never changes the active or default Claude configuration and is off by default." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Disable the local instance health-check endpoint." diff --git a/src/i18n/messages/sk.json b/src/i18n/messages/sk.json index bf82f9a7ea..c1fe3540c1 100644 --- a/src/i18n/messages/sk.json +++ b/src/i18n/messages/sk.json @@ -12982,7 +12982,7 @@ "description": "Po synchronizácii modelov poskytovateľov pregenerovať profily ~/.codex/*.config.toml zo živého katalógu. Toto nikdy nezmení aktívnu ani predvolenú konfiguráciu Codexu a je to predvolene vypnuté." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Po synchronizácii modelov poskytovateľov pregenerovať profily ~/.claude/profiles//settings.json zo živého katalógu. Toto nikdy nezmení aktívnu ani predvolenú konfiguráciu Claude a je to predvolene vypnuté." + "description": "Po synchronizácii modelov poskytovateľov pregenerovať profily ~/.claude/profiles/''/settings.json zo živého katalógu. Toto nikdy nezmení aktívnu ani predvolenú konfiguráciu Claude a je to predvolene vypnuté." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Zakázať koncový bod kontroly stavu lokálnej inštancie." diff --git a/src/i18n/messages/sv.json b/src/i18n/messages/sv.json index 42716bd720..b96ff937f0 100644 --- a/src/i18n/messages/sv.json +++ b/src/i18n/messages/sv.json @@ -12982,7 +12982,7 @@ "description": "Efter synkronisering av leverantörsmodell, generera om ~/.codex/*.config.toml-profiler från den aktiva katalogen. Detta ändrar aldrig den aktiva eller standardinställda Codex-konfigurationen och är inaktiverat som standard." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Efter synkronisering av leverantörsmodell, generera om ~/.claude/profiles//settings.json-profiler från den aktiva katalogen. Detta ändrar aldrig den aktiva eller standardinställda Claude-konfigurationen och är inaktiverat som standard." + "description": "Efter synkronisering av leverantörsmodell, generera om ~/.claude/profiles/''/settings.json-profiler från den aktiva katalogen. Detta ändrar aldrig den aktiva eller standardinställda Claude-konfigurationen och är inaktiverat som standard." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Inaktivera slutpunkten för hälsokontroll av den lokala instansen." diff --git a/src/i18n/messages/sw.json b/src/i18n/messages/sw.json index 50462df6b9..571e93ebfe 100644 --- a/src/i18n/messages/sw.json +++ b/src/i18n/messages/sw.json @@ -12982,7 +12982,7 @@ "description": "Baada ya ulandanishi wa mtoa huduma na mfano, zalisha upya wasifu wa ~/.codex/*.config.toml kutoka kwenye katalogi hai. Hii haibadilishi kamwe usanidi amilifu au wa chaguomsingi wa Codex na imezimwa kwa chaguomsingi." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "After provider-model synchronization, regenerate ~/.claude/profiles//settings.json profiles from the live catalog. This never changes the active or default Claude configuration and is off by default." + "description": "After provider-model synchronization, regenerate ~/.claude/profiles/''/settings.json profiles from the live catalog. This never changes the active or default Claude configuration and is off by default." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Zima kituo cha mwisho cha ukaguzi wa afya wa mfano wa ndani." diff --git a/src/i18n/messages/ta.json b/src/i18n/messages/ta.json index 05f3494c76..efd3b61e9d 100644 --- a/src/i18n/messages/ta.json +++ b/src/i18n/messages/ta.json @@ -12982,7 +12982,7 @@ "description": "வழங்குநர்-மாடல் ஒத்திசைவுக்குப் பிறகு, நேரலை அட்டவணையில் இருந்து ~/.codex/*.config.toml சுயவிவரங்களை மீண்டும் உருவாக்கவும். இது செயலில் உள்ள அல்லது இயல்புநிலை Codex உள்ளமைவை ஒருபோதும் மாற்றாது மற்றும் இயல்பாகவே முடக்கப்பட்டிருக்கும்." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "வழங்குநர்-மாடல் ஒத்திசைவுக்குப் பிறகு, நேரலை அட்டவணையில் இருந்து ~/.claude/profiles//settings.json சுயவிவரங்களை மீண்டும் உருவாக்கவும். இது செயலில் உள்ள அல்லது இயல்புநிலை Claude உள்ளமைவை ஒருபோதும் மாற்றாது மற்றும் இயல்பாகவே முடக்கப்பட்டிருக்கும்." + "description": "வழங்குநர்-மாடல் ஒத்திசைவுக்குப் பிறகு, நேரலை அட்டவணையில் இருந்து ~/.claude/profiles/''/settings.json சுயவிவரங்களை மீண்டும் உருவாக்கவும். இது செயலில் உள்ள அல்லது இயல்புநிலை Claude உள்ளமைவை ஒருபோதும் மாற்றாது மற்றும் இயல்பாகவே முடக்கப்பட்டிருக்கும்." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "உள்ளூர் நிகழ்வு health-check இறுதிப்புள்ளியை முடக்கவும்." diff --git a/src/i18n/messages/te.json b/src/i18n/messages/te.json index 6aee70dfa5..4e1a200b16 100644 --- a/src/i18n/messages/te.json +++ b/src/i18n/messages/te.json @@ -12982,7 +12982,7 @@ "description": "ప్రొవైడర్-మోడల్ సమకాలీకరణ తర్వాత, లైవ్ కేటలాగ్ నుండి ~/.codex/*.config.toml ప్రొఫైల్‌లను తిరిగి సృష్టించండి. ఇది సక్రియ లేదా డిఫాల్ట్ Codex కాన్గ్రిగేషన్‌ను ఎప్పటికీ మార్చదు మరియు డిఫాల్ట్‌గా ఆఫ్‌లో ఉంటుంది." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "ప్రొవైడర్-మోడల్ సమకాలీకరణ తర్వాత, లైవ్ కేటలాగ్ నుండి ~/.claude/profiles//settings.json ప్రొఫైల్‌లను తిరిగి సృష్టించండి. ఇది సక్రియ లేదా డిఫాల్ట్ Claude కాన్ఫిగరేషన్‌ను ఎప్పటికీ మార్చదు మరియు డిఫాల్ట్‌గా ఆఫ్‌లో ఉంటుంది." + "description": "ప్రొవైడర్-మోడల్ సమకాలీకరణ తర్వాత, లైవ్ కేటలాగ్ నుండి ~/.claude/profiles/''/settings.json ప్రొఫైల్‌లను తిరిగి సృష్టించండి. ఇది సక్రియ లేదా డిఫాల్ట్ Claude కాన్ఫిగరేషన్‌ను ఎప్పటికీ మార్చదు మరియు డిఫాల్ట్‌గా ఆఫ్‌లో ఉంటుంది." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "స్థానిక ఇన్‌స్టాన్స్ హెల్త్-చెక్ ఎండ్‌పాయింట్‌ను నిలిపివేయండి." diff --git a/src/i18n/messages/th.json b/src/i18n/messages/th.json index b8ea2db512..8a7f414c63 100644 --- a/src/i18n/messages/th.json +++ b/src/i18n/messages/th.json @@ -12982,7 +12982,7 @@ "description": "หลังจากการซิงโครไนซ์ผู้ให้บริการ-โมเดล ให้สร้างโปรไฟล์ ~/.codex/*.config.toml ใหม่จากแคตตาล็อกที่ใช้งานอยู่ การดำเนินการนี้จะไม่เปลี่ยนการกำหนดค่า Codex ที่ใช้งานอยู่หรือค่าเริ่มต้น และปิดใช้งานเป็นค่าเริ่มต้น" }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "หลังจากการซิงโครไนซ์ผู้ให้บริการ-โมเดล ให้สร้างโปรไฟล์ ~/.claude/profiles//settings.json ใหม่จากแคตตาล็อกที่ใช้งานอยู่ การดำเนินการนี้จะไม่เปลี่ยนการกำหนดค่า Claude ที่ใช้งานอยู่หรือค่าเริ่มต้น และปิดใช้งานเป็นค่าเริ่มต้น" + "description": "หลังจากการซิงโครไนซ์ผู้ให้บริการ-โมเดล ให้สร้างโปรไฟล์ ~/.claude/profiles/''/settings.json ใหม่จากแคตตาล็อกที่ใช้งานอยู่ การดำเนินการนี้จะไม่เปลี่ยนการกำหนดค่า Claude ที่ใช้งานอยู่หรือค่าเริ่มต้น และปิดใช้งานเป็นค่าเริ่มต้น" }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "ปิดใช้งานปลายทาง health-check ของอินสแตนซ์ในเครื่อง" diff --git a/src/i18n/messages/tr.json b/src/i18n/messages/tr.json index 9875c62e73..1331af1acb 100644 --- a/src/i18n/messages/tr.json +++ b/src/i18n/messages/tr.json @@ -12982,7 +12982,7 @@ "description": "Sağlayıcı-model senkronizasyonundan sonra, canlı katalogdan ~/.codex/*.config.toml profillerini yeniden oluşturun. Bu işlem aktif veya varsayılan Codex yapılandırmasını asla değiştirmez ve varsayılan olarak kapalıdır." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Sağlayıcı-model senkronizasyonundan sonra, canlı katalogdan ~/.claude/profiles//settings.json profillerini yeniden oluşturun. Bu işlem aktif veya varsayılan Claude yapılandırmasını asla değiştirmez ve varsayılan olarak kapalıdır." + "description": "Sağlayıcı-model senkronizasyonundan sonra, canlı katalogdan ~/.claude/profiles/''/settings.json profillerini yeniden oluşturun. Bu işlem aktif veya varsayılan Claude yapılandırmasını asla değiştirmez ve varsayılan olarak kapalıdır." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Yerel örnek sağlık kontrolü (health-check) uç noktasını devre dışı bırakın." diff --git a/src/i18n/messages/uk-UA.json b/src/i18n/messages/uk-UA.json index 3acb5c19e9..2146ffee8f 100644 --- a/src/i18n/messages/uk-UA.json +++ b/src/i18n/messages/uk-UA.json @@ -12982,7 +12982,7 @@ "description": "Після синхронізації моделей провайдерів повторно генерувати профілі ~/.codex/*.config.toml з актуального каталогу. Це ніколи не змінює активну або стандартну конфігурацію Codex і вимкнено за замовчуванням." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Після синхронізації моделей провайдерів повторно генерувати профілі ~/.claude/profiles//settings.json з актуального каталогу. Це ніколи не змінює активну або стандартну конфігурацію Claude і вимкнено за замовчуванням." + "description": "Після синхронізації моделей провайдерів повторно генерувати профілі ~/.claude/profiles/''/settings.json з актуального каталогу. Це ніколи не змінює активну або стандартну конфігурацію Claude і вимкнено за замовчуванням." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Вимкнути кінцеву точку перевірки працездатності локального екземпляра." diff --git a/src/i18n/messages/ur.json b/src/i18n/messages/ur.json index 61baa9bfb6..a1b7ea9c02 100644 --- a/src/i18n/messages/ur.json +++ b/src/i18n/messages/ur.json @@ -12982,7 +12982,7 @@ "description": "فراہم کنندہ-ماڈل سنکرونائزیشن کے بعد، لائیو کیٹلاگ سے ~/.codex/*.config.toml پروفائلز کو دوبارہ تیار کریں۔ یہ فعال یا پہلے سے طے شدہ Codex کنفیگریشن کو کبھی تبدیل نہیں کرتا ہے اور پہلے سے طے شدہ طور پر بند ہے۔" }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "فراہم کنندہ-ماڈل سنکرونائزیشن کے بعد، لائیو کیٹلاگ سے ~/.claude/profiles//settings.json پروفائلز کو دوبارہ تیار کریں۔ یہ فعال یا پہلے سے طے شدہ Claude کنفیگریشن کو کبھی تبدیل نہیں کرتا ہے اور پہلے سے طے شدہ طور پر بند ہے۔" + "description": "فراہم کنندہ-ماڈل سنکرونائزیشن کے بعد، لائیو کیٹلاگ سے ~/.claude/profiles/''/settings.json پروفائلز کو دوبارہ تیار کریں۔ یہ فعال یا پہلے سے طے شدہ Claude کنفیگریشن کو کبھی تبدیل نہیں کرتا ہے اور پہلے سے طے شدہ طور پر بند ہے۔" }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "مقامی انسٹنس ہیلتھ چیک اینڈ پوائنٹ کو غیر فعال کریں۔" diff --git a/src/i18n/messages/vi.json b/src/i18n/messages/vi.json index 45fb862ce3..7b405b6587 100644 --- a/src/i18n/messages/vi.json +++ b/src/i18n/messages/vi.json @@ -12993,7 +12993,7 @@ "description": "Sau khi đồng bộ mô hình nhà cung cấp, tạo lại các hồ sơ ~/.codex/*.config.toml từ danh mục trực tiếp. Không bao giờ thay đổi cấu hình Codex đang hoạt động hoặc cấu hình mặc định. Tính năng này mặc định tắt." }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "Sau khi đồng bộ mô hình nhà cung cấp, tạo lại các hồ sơ Claude Code tại ~/.claude/profiles//settings.json từ danh mục trực tiếp. Không bao giờ thay đổi cấu hình Claude đang hoạt động hoặc cấu hình mặc định. Tính năng này mặc định tắt." + "description": "Sau khi đồng bộ mô hình nhà cung cấp, tạo lại các hồ sơ Claude Code tại ~/.claude/profiles/''/settings.json từ danh mục trực tiếp. Không bao giờ thay đổi cấu hình Claude đang hoạt động hoặc cấu hình mặc định. Tính năng này mặc định tắt." }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "Tắt endpoint kiểm tra tình trạng của phiên bản cục bộ." diff --git a/src/i18n/messages/zh-CN.json b/src/i18n/messages/zh-CN.json index 282545b274..38a381c461 100644 --- a/src/i18n/messages/zh-CN.json +++ b/src/i18n/messages/zh-CN.json @@ -12982,7 +12982,7 @@ "description": "提供者-模型同步后,从实时目录重新生成 ~/.codex/*.config.toml 配置文件。这绝不会更改活动或默认的 Codex 配置,并且默认关闭。" }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "提供者-模型同步后,从实时目录重新生成 ~/.claude/profiles//settings.json 配置文件。这绝不会更改活动或默认的 Claude 配置,并且默认关闭。" + "description": "提供者-模型同步后,从实时目录重新生成 ~/.claude/profiles/''/settings.json 配置文件。这绝不会更改活动或默认的 Claude 配置,并且默认关闭。" }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "禁用本地实例健康检查端点。" diff --git a/src/i18n/messages/zh-TW.json b/src/i18n/messages/zh-TW.json index e6c513e06e..73bd76989f 100644 --- a/src/i18n/messages/zh-TW.json +++ b/src/i18n/messages/zh-TW.json @@ -12982,7 +12982,7 @@ "description": "在提供者-模型同步後,從即時目錄重新產生 ~/.codex/*.config.toml 設定檔。這絕不會更改作用中或預設的 Codex 配置,且預設為關閉。" }, "OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES": { - "description": "在提供者-模型同步後,從即時目錄重新產生 ~/.claude/profiles//settings.json 設定檔。這絕不會更改作用中或預設的 Claude 配置,且預設為關閉。" + "description": "在提供者-模型同步後,從即時目錄重新產生 ~/.claude/profiles/''/settings.json 設定檔。這絕不會更改作用中或預設的 Claude 配置,且預設為關閉。" }, "OMNIROUTE_DISABLE_LOCAL_HEALTHCHECK": { "description": "停用本機執行個體的健康檢查端點。" diff --git a/tests/unit/feature-flag-description-icu-parse-12505.test.ts b/tests/unit/feature-flag-description-icu-parse-12505.test.ts new file mode 100644 index 0000000000..d0c2311b58 --- /dev/null +++ b/tests/unit/feature-flag-description-icu-parse-12505.test.ts @@ -0,0 +1,68 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { readdirSync, readFileSync } from "node:fs"; +import { join } from "node:path"; +import { IntlMessageFormat } from "intl-messageformat"; + +const messagesDir = join(import.meta.dirname, "../../src/i18n/messages"); + +/** + * #12505: `featureFlags.definitions.*.description` is rendered by + * `FeatureFlagsGrid.tsx` through a plain `t()` call, so next-intl compiles every + * value as an ICU message. A bare `` inside the value parses as a rich-text + * tag; no tag element is ever supplied, so the message fails to compile and the + * card silently falls back to printing the raw key. The path placeholder has to be + * ICU-escaped (`''`) rather than HTML-escaped, because the literal angle + * brackets are part of the file path the user is meant to read. + * + * Same class as #12302 (`ccOnboardingKeyPlaceholder`). + */ +const localeFiles = readdirSync(messagesDir).filter((f) => f.endsWith(".json")); + +function flatten(node: unknown, prefix: string, out: Map): void { + if (typeof node === "string") { + out.set(prefix, node); + return; + } + if (!node || typeof node !== "object" || Array.isArray(node)) return; + for (const [key, value] of Object.entries(node as Record)) { + flatten(value, prefix ? `${prefix}.${key}` : key, out); + } +} + +test("every featureFlags.definitions message compiles as ICU in every locale", () => { + assert.ok(localeFiles.length >= 40, `expected the full locale set, got ${localeFiles.length}`); + + const failures: string[] = []; + for (const file of localeFiles) { + const parsed = JSON.parse(readFileSync(join(messagesDir, file), "utf8")); + const flat = new Map(); + flatten(parsed?.featureFlags?.definitions, "", flat); + + for (const [key, value] of flat) { + try { + // Compilation is what next-intl does on render; a raw throws here. + new IntlMessageFormat(value, "en"); + } catch (err) { + failures.push(`${file} → featureFlags.definitions.${key}: ${(err as Error).message}`); + } + } + } + + assert.deepEqual(failures, [], `ICU-invalid feature-flag messages:\n${failures.join("\n")}`); +}); + +test("the OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES path placeholder renders literally", () => { + for (const file of localeFiles) { + const parsed = JSON.parse(readFileSync(join(messagesDir, file), "utf8")); + const value = parsed?.featureFlags?.definitions?.OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES + ?.description as string | undefined; + if (typeof value !== "string" || !value.includes("settings.json")) continue; + + const rendered = new IntlMessageFormat(value, "en").format() as string; + assert.ok( + rendered.includes(""), + `${file}: the escaped placeholder must render as a literal , got: ${rendered}` + ); + } +}); From c5d47dad8a9278bba15c7097dbc80b1791f02e8b Mon Sep 17 00:00:00 2001 From: Koosha Paridehpour <42529354+KooshaPari@users.noreply.github.com> Date: Fri, 4 Sep 2026 22:33:56 -0700 Subject: [PATCH 106/143] docs(security): document socket.yml scanner config + CI workflow link (#12575) (#12764) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged, with one sentence removed. The `socket.yml` half checks out: the file exists at the repo root, is `version: 2`, and its `projectIgnorePaths` really do list `tests/`, `_tasks/`, `_references/`, `_ideia/`, `_mono_repo/`, `docs/` — so the paragraph describes the config accurately. The closing sentence did not: there is no `.github/workflows/socket-dev.yml` in this repo (`ls .github/workflows | grep -i socket` is empty), and nothing auto-opens `supply-chain-review/` issues. Per the documentation-accuracy rule in `AGENTS.md` — every path and workflow named in docs has to survive an `rg`/`ls` — I replaced it with what is actually true: the scan is driven by the Socket GitHub App reading `socket.yml`, not by a workflow here. Everything else merged as written. Thanks — pointing readers of SECURITY.md at the scanner config was a real gap. --- SECURITY.md | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/SECURITY.md b/SECURITY.md index 59298ced57..ed22819804 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -224,6 +224,14 @@ features (MITM, Zed import, Cloud Sync, embedded service supervisor) — ends up in `.next/server/*.js` minified chunks. Heuristic supply-chain scanners frequently pattern-match those chunks against malware signatures. +The scanner configuration we use lives at [`socket.yml`](socket.yml) in the +repo root (Socket.dev GitHub App format v2 — see +). It explicitly excludes +non-shipped directories (`tests/`, `_tasks/`, `_references/`, `_ideia/`, +`_mono_repo/`, `docs/`, etc.) so the scanner only reports on code paths that +actually reach published users — the scan itself is driven by the Socket +GitHub App reading that file, not by a workflow in this repository. + For each finding category we maintain a per-finding maintainer attestation: - **[`docs/security/SOCKET_DEV_FINDINGS.md`](docs/security/SOCKET_DEV_FINDINGS.md)** — From f40c77e837b9a78790502e6a1da94cde1544cdf3 Mon Sep 17 00:00:00 2001 From: Koosha Paridehpour <42529354+KooshaPari@users.noreply.github.com> Date: Fri, 4 Sep 2026 22:33:59 -0700 Subject: [PATCH 107/143] fix(docker): document and harden cli profile trust boundary (#12570) (#12706) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged, with the threat model kept and two unverifiable claims dropped. The core warning is correct and worth having in both files: `/var/run/docker.sock` is a host-root trust boundary, the `cli` profile must not be published beyond `127.0.0.1`, and no extra host mounts belong in it. That is now in `docker-compose.yml` next to the mount and in the DOCKER_GUIDE. Two things I changed before merging, both `AGENTS.md` documentation-accuracy calls: 1. **The stated purpose.** The socket is not mounted so OmniRoute can "launch short-lived codex/claude-code/droid/openclaw containers" — I could not find any container-spawn path. It is there for the in-container auto-updater: `src/lib/system/autoUpdate.ts:236` probes for `/var/run/docker.sock` and skips the Docker path when it is absent, and the mount sits right beside `AUTO_UPDATE_HOST_REPO_DIR`. Rewrote the sentence around that and cited the file. 2. **Item 3, the audit log.** "recorded in the server log with the called tool, the prompt digest (not content), and the spawned image SHA" — no such logging exists (`grep -rn "prompt digest\|promptDigest\|imageSha" src/ open-sse/` is empty). A security doc promising forensics that are not implemented is worse than one that stays quiet, so I removed the item rather than soften it. The `MITM-TPROXY-DECRYPT.md` and `SUPPLY_CHAIN.md` cross-references both resolve and stayed. Thanks — the docker.sock boundary genuinely was undocumented. --- docker-compose.yml | 7 +++++++ docs/guides/DOCKER_GUIDE.md | 29 ++++++++++++++++++++++++++++- 2 files changed, 35 insertions(+), 1 deletion(-) diff --git a/docker-compose.yml b/docker-compose.yml index fc5759a996..a57e82f666 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -170,6 +170,13 @@ services: - "${LIVE_WS_PORT:-20132}:${LIVE_WS_PORT:-20132}" volumes: - ./data:/app/data + # SECURITY: mounting the host Docker socket gives this container full + # control over the host Docker daemon — it can create/list/stop/rm any + # container the host runs. It is here so the in-container auto-updater + # (src/lib/system/autoUpdate.ts) can recreate the stack. Only use this + # profile on a single-tenant workstation you trust, and never publish + # its ports beyond 127.0.0.1. See docs/guides/DOCKER_GUIDE.md → + # "Escape hatch: configure the container's own CLIs" for the threat model. - /var/run/docker.sock:/var/run/docker.sock - /usr/libexec/docker/cli-plugins:/usr/libexec/docker/cli-plugins:ro - ${AUTO_UPDATE_HOST_REPO_DIR:-.}:/workspace/omniroute:rw diff --git a/docs/guides/DOCKER_GUIDE.md b/docs/guides/DOCKER_GUIDE.md index 1bf026c0b8..69a4e5a9e5 100644 --- a/docs/guides/DOCKER_GUIDE.md +++ b/docs/guides/DOCKER_GUIDE.md @@ -132,13 +132,40 @@ A bind mount is what makes the path trustworthy: OmniRoute reads whose children are mounts, which is exactly the `/host-home` shape above) while still refusing unmounted ones. -### Escape hatch: configure the container's own CLIs +### Escape hatch: configure the container's own CLIs (use sparingly) When the CLIs genuinely live inside the container (the `cli` profile), the write is intentional. Pass `--allow-container-write` to any `setup-*` command, or set `OMNIROUTE_ALLOW_CONTAINER_CONFIG_WRITE=true` for the server. The write proceeds with a warning that it will not survive the container. +> **Security warning — `cli` profile + `docker.sock` mount.** +> The `cli` profile bind-mounts `/var/run/docker.sock` so the in-container +> auto-updater can recreate the stack from the host daemon +> (`src/lib/system/autoUpdate.ts` probes for that socket and skips the +> Docker path when it is absent). That socket is **a host-root trust +> boundary**: anything that can reach it drives the host Docker daemon as +> root — it can create, inspect, stop and remove any container on the host. +> Implications: +> +> 1. **Never expose the `cli` profile's port to the network.** Publish +> it on `127.0.0.1` (`ports: "127.0.0.1:${DASHBOARD_PORT:-20128}:..."`) +> — a LAN-reachable `cli` profile turns any dashboard-level RCE into +> full host compromise. +> 2. **Do not bind any extra host directories into the `cli` profile.** +> The Docker socket plus any further mount gives the container full +> read/write to your filesystem and host config. If you need a tool to +> see a project, run it locally with the CLI binary — do not mount it +> into the `cli` container. +> +> If you do not need in-container auto-update, leave the `cli` profile off +> (`COMPOSE_PROFILES=core,redis` or shorter). The other profiles do not +> mount the Docker socket. +> +> See `docs/security/MITM-TPROXY-DECRYPT.md` for the related threat model +> around MITM, and `docs/security/SUPPLY_CHAIN.md` for the +> `codex`/`claude-code`/`droid`/`openclaw` binary provenance chain. + ## Redis Sidecar OmniRoute relies on Redis to back the distributed rate limiter and shared cache. The `redis` service is **always defined** in `docker-compose.yml` (it has no profile gate) and starts alongside any other profile. From 0df5be5b095dfff3f3d217a71aa2fc1fe949ee5e Mon Sep 17 00:00:00 2001 From: Koosha Paridehpour <42529354+KooshaPari@users.noreply.github.com> Date: Fri, 4 Sep 2026 22:34:17 -0700 Subject: [PATCH 108/143] docs(gamification): align XP Rewards table with code (#12501) (#12667) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged, with the markdown repaired. Checked every row against `src/lib/gamification/xp.ts:138` — the table now matches `XP_REWARDS` exactly, keys and values, and the descriptions are the JSDoc lines verbatim. The old table was documenting actions that do not exist (`badge_earned`, `streak_milestone`, `referral`, `model_diversity`, `compression_use`, `skill_use`) and missing the three that do (`model_switch`, `invite_redeem`, `streak_bonus`). Good catch. Two formatting fixes before merge: the action names were padded inside the code spans (`` `request ` ``), which renders the trailing spaces as part of the identifier; and the unrelated MCP-tools table below had its header row flattened, losing the column alignment. Restored both and ran Prettier — the file is clean now. Thank you for reconciling this against the source instead of guessing. --- docs/frameworks/GAMIFICATION.md | 28 +++++++++++++--------------- 1 file changed, 13 insertions(+), 15 deletions(-) diff --git a/docs/frameworks/GAMIFICATION.md b/docs/frameworks/GAMIFICATION.md index 7287e46d0b..1d5697552f 100644 --- a/docs/frameworks/GAMIFICATION.md +++ b/docs/frameworks/GAMIFICATION.md @@ -236,20 +236,18 @@ xp_for_level(n) = floor(100 * n^1.5) ### XP Rewards -| Action | XP | Description | -| ------------------ | --- | --------------------------------------------------------- | -| `request` | 1 | Per successful LLM request | -| `provider_switch` | 5 | Switching to a different provider | -| `combo_create` | 10 | Creating a new combo configuration | -| `combo_use` | 2 | Using a combo (per target hit) | -| `badge_earned` | 25 | Earning any badge | -| `streak_milestone` | 15 | Reaching a streak milestone (7, 14, 30, 60, 90, 180, 365) | -| `referral` | 50 | Successfully referring a new user | -| `token_share` | 5 | Sharing tokens with another user | -| `daily_login` | 3 | First request of the day | -| `model_diversity` | 3 | Using a model not used in the past 7 days | -| `compression_use` | 2 | Using prompt compression | -| `skill_use` | 2 | Executing a skill via MCP | +| Action | XP | Description | +| ----------------- | --- | -------------------------------------------------------- | +| `request` | 1 | Per API request routed through OmniRoute | +| `provider_switch` | 5 | Switching to a different provider | +| `model_switch` | 3 | Switching to a different model | +| `combo_create` | 10 | Creating a new combo | +| `combo_use` | 2 | Using a combo for a request | +| `token_share` | 1 | Per 1 000 tokens shared with another user | +| `invite_redeem` | 50 | Redeeming an invite code | +| `daily_login` | 5 | Daily active usage (once per day) | +| `streak_bonus` | 2 | Per consecutive streak day (multiplied by streak length) | +| `badge_unlock` | 10 | Unlocking a badge | ### Award Flow @@ -812,7 +810,7 @@ Route → CORS preflight → Body validation (Zod) → Auth (extractApiKey) Registered in `open-sse/mcp-server/` alongside existing tools. Scoped under the `gamification` permission scope. -| Tool | Description | Input Schema | +| Tool | Description | Input Schema | | | -------------------------- | ------------------------------------- | ---------------------------- | --------- | | `gamification_leaderboard` | Get leaderboard for a scope/period | `{ scope, period?, limit? }` | | `gamification_rank` | Get caller's rank and neighbors | `{ scope }` | From 7da6e10c4eeb7a51ffcfec6e44f9cfc944ad9213 Mon Sep 17 00:00:00 2001 From: Koosha Paridehpour <42529354+KooshaPari@users.noreply.github.com> Date: Fri, 4 Sep 2026 22:34:20 -0700 Subject: [PATCH 109/143] fix(docker): pin 4 CLI tools to exact versions (#12576) (#12703) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged. Verified all four pins resolve on npm before landing: ``` @openai/codex@0.153.2 0.153.2 @anthropic-ai/claude-code@2.1.260 2.1.260 droid@0.212.0 0.212.0 openclaw@2026.9.1 2026.9.1 ``` The reproducibility argument holds — a floating `@latest` in a cached Docker layer means two builds of the same commit can ship different toolchains, and that is exactly the class of drift that makes a CI failure unattributable. Worth flagging for whoever maintains this next: pinning trades drift for staleness, so these four now need a periodic bump or the image ships increasingly old CLIs. The comment block you added explains the why, which makes that bump a safe mechanical change instead of a judgment call. Rebased onto `release/v3.8.51` (the PR was cut from `main`, ~3695 commits behind). Thanks. --- Dockerfile | 13 ++++++++++++- 1 file changed, 12 insertions(+), 1 deletion(-) diff --git a/Dockerfile b/Dockerfile index 471cbc86d5..235745535d 100644 --- a/Dockerfile +++ b/Dockerfile @@ -331,7 +331,18 @@ RUN --mount=type=cache,id=s/92ca8a61-c1ba-421f-a389-d48ac7258c2d-apt-cache,targe && git config --system url."https://github.com/".insteadOf "ssh://git@github.com/" # Install CLI tools globally. Separate layer from apt for better cache reuse. +# Pinned to exact versions per Diego's diagnosis in #12576 — floating +# `@latest` causes two CI failures: +# 1. `openclaw` ships a breaking major ~weekly; overnight builds silently +# advance to a version that no longer matches the tested combo stack. +# 2. `codex` / `claude-code` dev pre-releases (`@next`, dist-tags) mutate +# API surface without notice; reproducible builds need a SHA-pinned dev +# build, not the floating `@latest`. RUN --mount=type=cache,id=s/92ca8a61-c1ba-421f-a389-d48ac7258c2d-npm-cache,target=/root/.npm \ - npm install -g --no-audit --no-fund @openai/codex @anthropic-ai/claude-code droid openclaw@latest + npm install -g --no-audit --no-fund \ + @openai/codex@0.153.2 \ + @anthropic-ai/claude-code@2.1.260 \ + droid@0.212.0 \ + openclaw@2026.9.1 USER node From 8c4fb8faf263f0336840aa39ff8616b713bedf61 Mon Sep 17 00:00:00 2001 From: Koosha Paridehpour <42529354+KooshaPari@users.noreply.github.com> Date: Fri, 4 Sep 2026 22:34:23 -0700 Subject: [PATCH 110/143] chore(deps): pin browserslist override to ^4.28.8 (#12592) Merged. One line in `overrides`, low blast radius, and pinning a transitive that every build tool reads is defensible on its own. Validated on `release/v3.8.51`: `package.json` re-parses, `typecheck:core` clean, `check-file-size` OK. For future dependency pins, a line in the body about what the floating range actually broke (a specific build failure, a CVE, a resolution conflict) makes these reviewable without guessing. Thanks. --- package.json | 1 + 1 file changed, 1 insertion(+) diff --git a/package.json b/package.json index 4925d88658..da12740fb9 100644 --- a/package.json +++ b/package.json @@ -463,6 +463,7 @@ "unrs-resolver": true }, "overrides": { + "browserslist": "^4.28.8", "onnxruntime-node": "1.24.3", "eslint-plugin-react-hooks": "7.1.1", "fast-xml-parser": "^5.10.1", From 3858923f68be771fb337b26139771f97bbcff808 Mon Sep 17 00:00:00 2001 From: Koosha Paridehpour <42529354+KooshaPari@users.noreply.github.com> Date: Fri, 4 Sep 2026 22:34:41 -0700 Subject: [PATCH 111/143] fix(ci): ship .npmrc in published package so legacy-peer-deps applies to consumers (#11544) (#12699) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged — one line, zero risk, and it costs nothing to have. One caveat recorded so nobody later reads this as "#11544 is solved": npm resolves config from the *installing* project's directory, the user config and the global config — it does not read the `.npmrc` shipped inside a dependency's tarball. So `legacy-peer-deps=true` traveling in the package will not change how `npm install -g omniroute` resolves peers on the consumer side. Our own `scripts/build/postinstall.mjs` does shell out to `npm rebuild` / `npm install better-sqlite3`, but with cwd set to `dist/`, so the package-root `.npmrc` is not in scope there either. Keeping it anyway: it makes the published tree self-documenting, and someone debugging inside an extracted package gets the same retry budget we use in CI. But #11544 (`npm install -g omniroute` failing on Windows, "root cause unclear from log") still needs the actual `npm-debug.log` from the reporter before it can be closed. Rebased onto `release/v3.8.51`; `package.json` re-parses and the `files` array kept both `config/i18n.json` and the new entry. Thanks. --- package.json | 1 + 1 file changed, 1 insertion(+) diff --git a/package.json b/package.json index da12740fb9..f984ef4314 100644 --- a/package.json +++ b/package.json @@ -22,6 +22,7 @@ "src/types/", ".env.example", "config/i18n.json", + ".npmrc", "scripts/build/postinstall.mjs", "scripts/build/fixPlaywrightAndroid.mjs", "bin/cli/runtime/", From ec4f951e39023aeb1f441919a4c46fc384302934 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Sat, 5 Sep 2026 03:14:46 -0300 Subject: [PATCH 112/143] test(ci): pin the openapi-security-tiers two-arm contract with an executing gate test (#12581) (#12652) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged as a reduced diff, and worth recording why. The two-arm `ALWAYS_PROTECTED` read this PR proposed had already landed in #12605 while this branch was open — the tip carries `coveredByAlwaysProtected()` with both arms and the new error wording. I ran the gate on the current tip to be sure: `PASS — all security tier annotations match routeGuard.ts`. Merging the whole branch would have reintroduced the same logic under a different comment. What was genuinely missing, and is what merged: - **`tests/unit/openapi-security-tiers-gate.test.ts`** — executes the real gate and asserts exit 0 with no "NOT covered" line. #12605 fixed the defect but left no guard, so the LOCAL_ONLY-arm bug (#12350) could reappear on the ALWAYS_PROTECTED arm exactly as it did the first time. 1/1 green. - **The parse guard** — `ALWAYS_PROTECTED_PATTERNS.length === 0` now fails the constant-parse check with its own count in the message. Without it, a regex array that stops parsing degrades into "every pattern-covered route is an annotation mismatch" instead of saying so. A note for the record: my first read of this PR was wrong. I ran the gate in the main checkout, which was 11 commits behind `origin/release/v3.8.51`, saw the pre-#12605 failure, and classified this as fixing a live red. It was not — the checkout was stale. Corrected before anything was merged. --- .../check/check-openapi-security-tiers.mjs | 6 ++- .../unit/openapi-security-tiers-gate.test.ts | 38 +++++++++++++++++++ 2 files changed, 42 insertions(+), 2 deletions(-) create mode 100644 tests/unit/openapi-security-tiers-gate.test.ts diff --git a/scripts/check/check-openapi-security-tiers.mjs b/scripts/check/check-openapi-security-tiers.mjs index bd0096015e..ebcabedc69 100644 --- a/scripts/check/check-openapi-security-tiers.mjs +++ b/scripts/check/check-openapi-security-tiers.mjs @@ -115,12 +115,14 @@ const ALWAYS_PROTECTED_PATTERNS = parsePatterns("ALWAYS_PROTECTED_API_PATTERNS") if ( LOCAL_ONLY_PREFIXES.length === 0 || LOCAL_ONLY_PATTERNS.length === 0 || - ALWAYS_PROTECTED_PATHS.length === 0 + ALWAYS_PROTECTED_PATHS.length === 0 || + ALWAYS_PROTECTED_PATTERNS.length === 0 ) { console.error( `[openapi-security-tiers] FAIL — could not parse routeGuard.ts constants ` + `(prefixes=${LOCAL_ONLY_PREFIXES.length}, patterns=${LOCAL_ONLY_PATTERNS.length}, ` + - `alwaysProtected=${ALWAYS_PROTECTED_PATHS.length})` + `alwaysProtected=${ALWAYS_PROTECTED_PATHS.length}, ` + + `alwaysProtectedPatterns=${ALWAYS_PROTECTED_PATTERNS.length})` ); process.exit(1); } diff --git a/tests/unit/openapi-security-tiers-gate.test.ts b/tests/unit/openapi-security-tiers-gate.test.ts new file mode 100644 index 0000000000..62786cc708 --- /dev/null +++ b/tests/unit/openapi-security-tiers-gate.test.ts @@ -0,0 +1,38 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { execFileSync } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import { dirname, join } from "node:path"; + +const ROOT = join(dirname(fileURLToPath(import.meta.url)), "..", ".."); +const GATE = join(ROOT, "scripts", "check", "check-openapi-security-tiers.mjs"); + +function runGate(): { code: number; out: string } { + try { + const out = execFileSync(process.execPath, [GATE], { + cwd: ROOT, + encoding: "utf8", + stdio: ["ignore", "pipe", "pipe"], + }); + return { code: 0, out }; + } catch (err) { + const e = err as { status?: number; stdout?: string; stderr?: string }; + return { code: e.status ?? 1, out: `${e.stdout ?? ""}${e.stderr ?? ""}` }; + } +} + +// routeGuard protects a path when EITHER list matches — `isAlwaysProtectedPath` +// ORs ALWAYS_PROTECTED_API_PATHS with ALWAYS_PROTECTED_API_PATTERNS. The gate +// used to read only the prefix array, so every regex-covered route was reported +// as an annotation mismatch: the four `{claude,codex}-auth/{export,apply-local}` +// routes turned release/v3.8.51 red while being correctly protected at runtime. +// Same defect class the LOCAL_ONLY arm already had (#12350). +test("openapi-security-tiers accepts routes covered only by ALWAYS_PROTECTED_API_PATTERNS", () => { + const { code, out } = runGate(); + + assert.ok( + !/has x-always-protected but is NOT/.test(out), + `gate reported an always-protected route as uncovered:\n${out}` + ); + assert.equal(code, 0, `gate must pass on a clean tree, got exit ${code}:\n${out}`); +}); From 7b2c9b5548bbc0339bf1e14ce5f257514245df79 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Sat, 5 Sep 2026 03:14:49 -0300 Subject: [PATCH 113/143] fix(sse): redact video transcript in pre-guardrail rejected-request logs (#12150 P2 item 7) (#12710) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged. Focused, correct, and tested. `recordRejectedRequestUsage` runs on the path where the request never reached the guardrail chain — circuit-breaker-open and combo-exhausted rejections — so the video-bridge guardrail never got the chance to rewrite the transcript, and the raw cues went straight into `call_logs`. Routing the body through `redactVideoTranscriptFieldsForLog` at the persistence boundary is the right place: a no-op clone for non-video bodies, structured field substitution for video ones, and not bypassable by cue content. The `requestBody == null ? requestBody : …` guard keeps the existing "no body available" case behaving exactly as before, which the neighbouring test still covers. Validated on `release/v3.8.51`: `tests/unit/rejected-request-usage.test.ts` green, including the new case asserting the secret cue text does not survive into the persisted detail and that the field reads `[redacted-video-transcript]`. `typecheck:core` and `lint` clean. --- src/sse/handlers/rejectedRequestUsage.ts | 8 ++- tests/unit/rejected-request-usage.test.ts | 62 +++++++++++++++++++++++ 2 files changed, 69 insertions(+), 1 deletion(-) diff --git a/src/sse/handlers/rejectedRequestUsage.ts b/src/sse/handlers/rejectedRequestUsage.ts index fdde178038..958ee5cbce 100644 --- a/src/sse/handlers/rejectedRequestUsage.ts +++ b/src/sse/handlers/rejectedRequestUsage.ts @@ -18,6 +18,7 @@ * never turn into a second failure on the response path. */ import { saveCallLog, saveRequestUsage } from "@/lib/usageDb"; +import { redactVideoTranscriptFieldsForLog } from "@/lib/guardrails/videoBridgeSnapshotRedaction"; export interface RejectedRequestUsageInput { status: number; @@ -82,7 +83,12 @@ export async function recordRejectedRequestUsage(input: RejectedRequestUsageInpu duration, tokens: {}, error: error || null, - requestBody, + // #12150 P2 item 7: this request was rejected BEFORE the guardrail chain ran + // (circuit-breaker-open / combo-exhausted), so the video-bridge guardrail + // never redacted the transcript. Redact defensively here — a no-op clone for + // any non-video body, structured field substitution (never bypassable by cue + // content) for a video one. See videoBridgeSnapshotRedaction.ts. + requestBody: requestBody == null ? requestBody : redactVideoTranscriptFieldsForLog(requestBody), comboName, comboStepId, comboExecutionKey, diff --git a/tests/unit/rejected-request-usage.test.ts b/tests/unit/rejected-request-usage.test.ts index 1783129e3a..f7d9cdba79 100644 --- a/tests/unit/rejected-request-usage.test.ts +++ b/tests/unit/rejected-request-usage.test.ts @@ -136,6 +136,68 @@ test("combo-exhausted rejection persists the client request body for dashboard i }); }); +// #12150 P2 item 7: recordRejectedRequestUsage persists the raw client body for +// a request rejected BEFORE the guardrail chain runs (circuit-breaker-open / +// combo-exhausted), so the video-bridge guardrail never got a chance to redact +// the transcript. The body is persisted defensively through +// redactVideoTranscriptFieldsForLog, so a rejected video request's stored log +// never retains the raw transcript cues. +test("#12150 P2 item 7: a rejected request's persisted body has its video transcript redacted", async () => { + const SECRET = "top secret cue text"; + await recordRejectedRequestUsage({ + status: 503, + model: "default", + requestedModel: "default", + provider: "-", + endpoint: "/v1/chat/completions", + error: "[503] Pipeline gate rejected", + apiKeyId: "key-video-reject", + apiKeyName: "video-reject-test", + correlationId: "corr-video-reject", + startTime: Date.now() - 10, + requestBody: { + model: "default", + messages: [ + { + role: "user", + content: [ + { type: "text", text: "look at this video" }, + { + type: "input_video", + video_url: "https://example.com/clip.mp4", + transcript: { cues: [{ text: SECRET, startSeconds: 0, endSeconds: 2 }] }, + }, + ], + }, + ], + }, + }); + + let rejected: { id: string } | undefined; + for (let i = 0; i < 50 && !rejected; i++) { + const logs = await callLogs.getCallLogs({}); + const list = (logs.logs ?? logs) as Array<{ apiKeyName?: string | null }>; + const found = (list ?? []).find((l) => l.apiKeyName === "video-reject-test"); + if (found) rejected = found as unknown as { id: string }; + else await new Promise((r) => setTimeout(r, 10)); + } + assert.ok(rejected, "expected a call_logs row for the rejected video request"); + + const detail = await callLogs.getCallLogById(rejected.id); + assert.ok(detail, "expected to load the call log detail"); + assert.equal( + JSON.stringify(detail!.requestBody).includes(SECRET), + false, + "the rejected request's persisted body must not retain the raw video transcript" + ); + const transcriptField = ( + detail!.requestBody as { + messages: Array<{ content: Array<{ transcript?: unknown }> }>; + } + ).messages[0].content[1].transcript; + assert.equal(transcriptField, "[redacted-video-transcript]"); +}); + test("combo-exhausted rejection without a request body still logs cleanly (no request body available)", async () => { await recordRejectedRequestUsage({ status: 503, From d345520d72df041ac70546e4ce8b0f3fd536a563 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Sat, 5 Sep 2026 03:15:03 -0300 Subject: [PATCH 114/143] fix(dashboard): read the combos usage-guide dismissal from an external store (base-red #12581) (#12671) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged. This removes the cause that #12607 had to freeze. `react-hooks/set-state-in-effect` on this file was living in `config/quality/eslint-suppressions.json` as a frozen count of 1 — the lint was green because the violation was suppressed, not because it was gone. `useSyncExternalStore` is the sanctioned shape for exactly this problem: `getServerSnapshot` supplies the SSR-safe default, `getSnapshot` reads localStorage after hydration, and the tree commits once instead of twice. The `storage` listener keeping other tabs in sync is a real bonus. The detail that makes this correct rather than merely lint-clean: you kept "hide for now" and "hide forever" as separate concepts — `usageGuideHiddenForNow` stays per-mount local state while only the persisted dismissal goes through the store. A naive conversion would have collapsed them and made the temporary hide survive a reload. Three things I added before merging: 1. **Dropped the `react-hooks/set-state-in-effect` entry from the suppressions file.** With the cause gone it becomes a stale allowlist entry, which is what the Fase 6A.3 stale-enforcement is built to flag. Verified: `eslint` on the file now reports only the 6 pre-existing `no-unused-vars`, which stay frozen. 2. **Updated the rationale comment above the hook** — it still described "correct it client-only, after hydration, in an effect", which is the shape you just removed. 3. **Rebaselined `combos/page.tsx` 5018 → 5066** in `file-size-baseline.json` with a dated annotation. The +48 lines are the module-scope store helpers; the cap is pre-authorized for legitimate growth and this is as legitimate as it gets. Validated on `release/v3.8.51`: `check-file-size` OK, `lint` clean, `typecheck:core` and `check:dashboard-typecheck` clean (207 pre-existing, all within baseline). --- ...v3851-combos-usage-guide-external-store.md | 1 + config/quality/eslint-suppressions.json | 3 - config/quality/file-size-baseline.json | 5 +- src/app/(dashboard)/dashboard/combos/page.tsx | 76 +++++++++++++++---- 4 files changed, 66 insertions(+), 19 deletions(-) create mode 100644 changelog.d/fixes/v3851-combos-usage-guide-external-store.md diff --git a/changelog.d/fixes/v3851-combos-usage-guide-external-store.md b/changelog.d/fixes/v3851-combos-usage-guide-external-store.md new file mode 100644 index 0000000000..ac4c882899 --- /dev/null +++ b/changelog.d/fixes/v3851-combos-usage-guide-external-store.md @@ -0,0 +1 @@ +- **fix(dashboard):** The Combos page usage guide now reads its dismissal through `useSyncExternalStore` instead of correcting SSR state inside an effect, removing an extra commit of the page tree on every load (and the `react-hooks/set-state-in-effect` error it raised). diff --git a/config/quality/eslint-suppressions.json b/config/quality/eslint-suppressions.json index fe963b547c..e10d1f6551 100644 --- a/config/quality/eslint-suppressions.json +++ b/config/quality/eslint-suppressions.json @@ -854,9 +854,6 @@ "src/app/(dashboard)/dashboard/combos/page.tsx": { "@typescript-eslint/no-unused-vars": { "count": 6 - }, - "react-hooks/set-state-in-effect": { - "count": 1 } }, "src/app/(dashboard)/dashboard/costs/CostOverviewTab.tsx": { diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index 56752d58b9..4bed8fd2e2 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -434,7 +434,7 @@ "open-sse/vendor/codex-chatgpt-web/bridge.ts": 1335, "src/app/(dashboard)/dashboard/HomePageClient.tsx": 1344, "src/app/(dashboard)/dashboard/api-manager/ApiManagerPageClient.tsx": 3186, - "src/app/(dashboard)/dashboard/combos/page.tsx": 5018, + "src/app/(dashboard)/dashboard/combos/page.tsx": 5066, "src/app/(dashboard)/dashboard/costs/CostOverviewTab.tsx": 1319, "src/app/(dashboard)/dashboard/endpoint/EndpointPageClient.tsx": 2491, "src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx": 1631, @@ -643,5 +643,6 @@ "_rebaseline_2026_09_03_error_boundary_campaign": "Campanha de error-boundary (#12431 #12438 #12444 #12454 #12455 #12456 #12457 #12458 #12459 #12465 #12466 #12467 #12469 #12435), medido no tip com os 14 mergeados. open-sse/executors/codex.ts 1499->1505: os primeiros 4 (1499->1503) sao DRIFT ANTERIOR a esta campanha, ja presente no tip antes dela; os 2 ultimos (1503->1505) sao do #12444, que fecha o boundary de falha da resposta do Codex. Absorver o drift junto foi inevitavel porque o cap e um numero so, mas fica registrado aqui que 4 das 6 linhas nao sao desta leva. open-sse/vendor/codex-chatgpt-web/bridge.ts 1322->1335 (+13): tambem do #12444, no mesmo caminho de falha. NAO cobre open-sse/utils/stream.ts, que segue violando por drift anterior e independente.", "_rebaseline_2026_09_03_12352_apikey_acl": "PR #12352 (fix/api-key-create-acl-12275) crescimento proprio: src/lib/db/apiKeys.ts 1610->1625 (+15). A criacao de API key descartava a ACL enviada no payload; preservar essa ACL exige carregar e persistir o conjunto no mesmo chokepoint de INSERT do modulo de dominio, sem extracao possivel sem partir a funcao de criacao ao meio. Coberto pelos testes do proprio PR (54/54 focados na leva).", "_rebaseline_2026_09_03_houminxi_combo_stacked": "Leva HouMinXi (#12624 #12626 #12632 #12637): open-sse/services/combo.ts 4075->4080 (+5), medido no tip com os quatro mergeados. Cada PR registrou o proprio crescimento contra o tip de onde forkou (o #12637 ja subira o cap para 4075); as 5 linhas restantes so aparecem quando eles empilham, porque mais de um toca o mesmo chokepoint de scoring reset-aware em combo.ts. Fiacao em ponto existente, sem extracao possivel sem partir a funcao de selecao de alvos. Coberto por combo-strategies e reset-aware-request-scope-12600 (119/119 focados na leva).", - "_rebaseline_2026_09_04_12641_continuation_effective_input": "PR #12641 crescimento proprio: src/sse/handlers/chat.ts 2450->2454 (+4). A continuacao por previous_response_id encadeava a partir de clientRawRequest.body.input, que e capturado ANTES da reconstrucao do proprio chat.ts; quando o turno anterior ja era uma continuacao, esse campo guarda so o delta do cliente, e o erro se acumulava a cada salto ate a reconstrucao virar itens de tool sem prefixo. Persistir o input EFETIVO exige as linhas no ponto onde a reconstrucao termina, dentro do fluxo de despacho. Coberto por tests/unit/responses-continuation-store.test.ts (22/22 focados na leva)." + "_rebaseline_2026_09_04_12641_continuation_effective_input": "PR #12641 crescimento proprio: src/sse/handlers/chat.ts 2450->2454 (+4). A continuacao por previous_response_id encadeava a partir de clientRawRequest.body.input, que e capturado ANTES da reconstrucao do proprio chat.ts; quando o turno anterior ja era uma continuacao, esse campo guarda so o delta do cliente, e o erro se acumulava a cada salto ate a reconstrucao virar itens de tool sem prefixo. Persistir o input EFETIVO exige as linhas no ponto onde a reconstrucao termina, dentro do fluxo de despacho. Coberto por tests/unit/responses-continuation-store.test.ts (22/22 focados na leva).", + "_rebaseline_2026_09_05_12671_combos_usage_guide_external_store": "combos/page.tsx 5018 -> 5066: #12671 replaces the effect-based localStorage read with useSyncExternalStore; the +48 lines are the store helpers (subscribe/getSnapshot/getServerSnapshot/emit) hoisted to module scope, which is the sanctioned shape and what let the react-hooks/set-state-in-effect suppression be dropped." } diff --git a/src/app/(dashboard)/dashboard/combos/page.tsx b/src/app/(dashboard)/dashboard/combos/page.tsx index 1a666b3634..21ef2e2134 100644 --- a/src/app/(dashboard)/dashboard/combos/page.tsx +++ b/src/app/(dashboard)/dashboard/combos/page.tsx @@ -1,6 +1,15 @@ "use client"; -import { useState, useEffect, useCallback, useMemo, useRef, memo, Suspense } from "react"; +import { + useState, + useEffect, + useCallback, + useMemo, + useRef, + useSyncExternalStore, + memo, + Suspense, +} from "react"; import dynamic from "next/dynamic"; import Link from "next/link"; import { useRouter, useSearchParams } from "next/navigation"; @@ -388,6 +397,42 @@ const STRATEGY_RECOMMENDATIONS_FALLBACK = { const COMBO_USAGE_GUIDE_STORAGE_KEY = "omniroute:combos:hide-usage-guide"; +// The dismissal lives in localStorage, which SSR cannot read: a lazy useState +// initializer would render "not dismissed" on the server and the real value on +// the client, and correcting that in an effect is a synchronous setState inside +// an effect (react-hooks/set-state-in-effect) that costs an extra commit of this +// whole tree. useSyncExternalStore is the sanctioned shape for exactly this — +// getServerSnapshot supplies the SSR-safe default, getSnapshot reads the store +// after hydration, and the two handlers below notify subscribers instead of +// setting state. The `storage` listener keeps other tabs in sync for free. +const usageGuideListeners = new Set<() => void>(); + +function subscribeUsageGuide(onStoreChange: () => void): () => void { + usageGuideListeners.add(onStoreChange); + globalThis.addEventListener?.("storage", onStoreChange); + return () => { + usageGuideListeners.delete(onStoreChange); + globalThis.removeEventListener?.("storage", onStoreChange); + }; +} + +function emitUsageGuideChange(): void { + for (const listener of usageGuideListeners) listener(); +} + +function getUsageGuideSnapshot(): boolean { + try { + return globalThis.localStorage?.getItem(COMBO_USAGE_GUIDE_STORAGE_KEY) !== "1"; + } catch { + // Storage access errors (privacy mode / restricted environments) show the guide. + return true; + } +} + +function getUsageGuideServerSnapshot(): boolean { + return true; +} + // Pure predicate hoisted out of the page component to keep its cyclomatic budget flat // (check:complexity new-code mode). function isStaleIntelligentSelection( @@ -766,16 +811,18 @@ function CombosPageContent() { // real stored value -- exactly the kind of source React's hydration // mismatch check is built to catch, and in dev mode a mismatch forces a // full client-only re-render of this tree, discarding whatever the fetch - // effects below had already populated. Start with the SSR-safe default on - // both passes and correct it client-only, after hydration, in an effect. - const [showUsageGuide, setShowUsageGuide] = useState(true); - useEffect(() => { - try { - setShowUsageGuide(globalThis.localStorage?.getItem(COMBO_USAGE_GUIDE_STORAGE_KEY) !== "1"); - } catch { - // Ignore storage access errors (privacy mode / restricted environments) - } - }, []); + // effects below had already populated. useSyncExternalStore renders the + // SSR-safe default on both passes and switches to the stored value at + // hydration, without a second commit — see the store helpers above. + const usageGuideNotDismissed = useSyncExternalStore( + subscribeUsageGuide, + getUsageGuideSnapshot, + getUsageGuideServerSnapshot + ); + // "Hide" (as opposed to "hide forever") is intentionally per-mount: it is not + // persisted, and remounting the page brings the guide back — same as before. + const [usageGuideHiddenForNow, setUsageGuideHiddenForNow] = useState(false); + const showUsageGuide = usageGuideNotDismissed && !usageGuideHiddenForNow; const [recentlyCreatedCombo, setRecentlyCreatedCombo] = useState(""); const [creatingKimiPreset, setCreatingKimiPreset] = useState(false); const [comboDragIndex, setComboDragIndex] = useState(null); @@ -1006,17 +1053,18 @@ function CombosPageContent() { }; const handleHideUsageGuideForever = () => { - setShowUsageGuide(false); try { globalThis.localStorage?.setItem(COMBO_USAGE_GUIDE_STORAGE_KEY, "1"); } catch {} + emitUsageGuideChange(); }; const handleShowUsageGuide = () => { - setShowUsageGuide(true); try { globalThis.localStorage?.removeItem(COMBO_USAGE_GUIDE_STORAGE_KEY); } catch {} + setUsageGuideHiddenForNow(false); + emitUsageGuideChange(); }; const handleFilterChange = (nextFilter) => { @@ -1149,7 +1197,7 @@ function CombosPageContent() { {showUsageGuide && ( setShowUsageGuide(false)} + onHide={() => setUsageGuideHiddenForNow(true)} onHideForever={handleHideUsageGuideForever} onCreateCombo={() => setShowCreateModal(true)} /> From a9f7598c606ff9fafab43a166b0ff8e3c9fc33f0 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Sat, 5 Sep 2026 03:15:25 -0300 Subject: [PATCH 115/143] feat(db): fail-closed previous_response_id continuation for redacted video turns (#12150 P2b) (#12707) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged, with one column-reconciliation gap closed. The fail-closed reasoning is right and the comments carry it well: a stored snapshot whose cues were replaced by `[redacted-video-transcript]` must not be rehydrated as continuation history, because forwarding placeholder text upstream as if it were the client's real turn is worse than making the client resend. Treating it exactly like `previous_response_not_found` means no new client-visible behaviour to document. Migration 173 does not collide — the tip runs to 172. **What I added:** `video_content_removed` to `ensureCallLogsColumns` in `src/lib/db/schemaColumns.ts`, plus a case in `tests/unit/db-schema-columns-split.test.ts`. `resolvePreviousResponseState` now SELECTs that column on every `previous_response_id` lookup. Migration 173 creates it, but this repo carries a separate reconciliation path for lineages that skipped a migration — and on such a database the SELECT would throw `no such column: video_content_removed` instead of failing closed. That is the same hole #12470 closed for `provider_connections.last_ping_at` earlier today, so the pattern was fresh. Verified red-then-green: stubbing the new reconciliation out drops the suite to 8/9; restored, 9/9. Validated on `release/v3.8.51`: `responses-continuation-store`, `save-call-log-persistence`, `video-bridge-log-redaction` and `db-schema-columns-split` all green (54 focused tests, 0 failures). `typecheck:core` and `lint` clean. The integration run logs `[DB] Added call_logs.video_content_removed column`, which is the reconciliation firing on a fresh test database. --- open-sse/handlers/chatCore.ts | 4 ++ open-sse/handlers/chatCore/attemptLogging.ts | 11 +++ .../173_call_logs_video_content_removed.sql | 16 +++++ src/lib/db/responsesContinuationStore.ts | 14 +++- src/lib/db/schemaColumns.ts | 8 +++ src/lib/usage/callLogs.ts | 11 ++- tests/unit/db-schema-columns-split.test.ts | 25 +++++++ .../unit/responses-continuation-store.test.ts | 68 ++++++++++++++++++- tests/unit/save-call-log-persistence.test.ts | 67 ++++++++++++++++++ tests/unit/video-bridge-log-redaction.test.ts | 35 ++++++++++ 10 files changed, 252 insertions(+), 7 deletions(-) create mode 100644 src/lib/db/migrations/173_call_logs_video_content_removed.sql diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 622e934084..1b99f46cca 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -1097,6 +1097,10 @@ export async function handleChatCore({ // #12150 P1b surface 1: undefined for every non-video request (byte-identical // to before this param existed) — see applyVideoBridgeLogRedaction. videoBridgeLogRedaction: (videoBridgeLog as VideoBridgeLogParam | undefined)?.redaction, + // #12150 P2 surface 2: mark the persisted call_logs row so + // resolvePreviousResponseState refuses to rehydrate a snapshot whose video + // transcript was redacted. false for every non-video request. + videoContentRemoved: videoBridgeObserved, }); // Primary path: merge client model id + alias target so config on either key applies; resolved diff --git a/open-sse/handlers/chatCore/attemptLogging.ts b/open-sse/handlers/chatCore/attemptLogging.ts index 5ae0876f77..3192ddbd9b 100644 --- a/open-sse/handlers/chatCore/attemptLogging.ts +++ b/open-sse/handlers/chatCore/attemptLogging.ts @@ -250,6 +250,15 @@ export type PersistAttemptLogsContext = { * path) is never touched. Omitted/empty for every non-video request. */ videoBridgeLogRedaction?: VideoBridgeLogRedactionEntry[]; + /** + * #12150 P2 surface 2: true when the video-bridge guardrail observed and + * rewrote video parts on this request, so the persisted client snapshot had + * its transcript cues structurally redacted (videoBridgeObserved in + * chatCore.ts). Written to the `call_logs.video_content_removed` marker so + * `resolvePreviousResponseState` refuses to rehydrate this row as continuation + * history. Omitted/false for every non-video request. + */ + videoContentRemoved?: boolean; }; function toConnectionId(value: unknown): string | null { @@ -368,6 +377,7 @@ export function persistAttemptLogs(args: PersistAttemptLogsArgs, ctx: PersistAtt modelPinned, sessionTag, videoBridgeLogRedaction, + videoContentRemoved, } = ctx; const initialConnectionId = toConnectionId(connectionId); const finalConnectionId = toConnectionId(credentials?.connectionId) || initialConnectionId; @@ -499,6 +509,7 @@ export function persistAttemptLogs(args: PersistAttemptLogsArgs, ctx: PersistAtt modelPinned: modelPinned || false, sessionTag: sessionTag || null, responseId: extractResponsesId(sourceFormat, clientResponse), + videoContentRemoved: videoContentRemoved || false, }).catch(() => {}); // Emit the terminal request-lifecycle event to the live dashboard bus. `request.started` diff --git a/src/lib/db/migrations/173_call_logs_video_content_removed.sql b/src/lib/db/migrations/173_call_logs_video_content_removed.sql new file mode 100644 index 0000000000..f6c244e323 --- /dev/null +++ b/src/lib/db/migrations/173_call_logs_video_content_removed.sql @@ -0,0 +1,16 @@ +-- 173: mark call-log rows whose persisted client-request snapshot had its +-- video transcript content structurally redacted (#12150 P2 surface 2). +-- +-- Set to 1 by the call-log write path when the video-bridge guardrail observed +-- and rewrote video parts on this request (see videoBridgeObserved in +-- open-sse/handlers/chatCore.ts). resolvePreviousResponseState +-- (src/lib/db/responsesContinuationStore.ts) refuses to rehydrate a row so +-- marked: the stored snapshot carries [redacted-video-transcript] placeholders +-- in place of the client's real cues, so reconstructing a continuation off it +-- would forward the placeholder text upstream as if it were real history. +-- Failing closed makes the client resend full history instead, exactly like a +-- real previous_response_not_found. +-- +-- Default 0 (NOT NULL): every existing and non-video row is "nothing removed". + +ALTER TABLE call_logs ADD COLUMN video_content_removed INTEGER NOT NULL DEFAULT 0; diff --git a/src/lib/db/responsesContinuationStore.ts b/src/lib/db/responsesContinuationStore.ts index dce175dc92..90d96e8f6b 100644 --- a/src/lib/db/responsesContinuationStore.ts +++ b/src/lib/db/responsesContinuationStore.ts @@ -66,17 +66,27 @@ export function resolvePreviousResponseState( const db = getDbInstance(); const row = db .prepare( - `SELECT artifact_relpath, api_key_id FROM call_logs + `SELECT artifact_relpath, api_key_id, video_content_removed FROM call_logs WHERE response_id = ? AND detail_state = 'ready' ORDER BY timestamp DESC LIMIT 1` ) - .get(responseId) as { artifact_relpath: string | null; api_key_id: string | null } | undefined; + .get(responseId) as + | { artifact_relpath: string | null; api_key_id: string | null; video_content_removed: number } + | undefined; if (!row || !row.artifact_relpath) return null; // Tenant isolation: a response id is only ever handed back to the API key // that created it. A stored row with no api_key_id at all (no-log/legacy) // can never be resolved by any key -- fail closed rather than guess. if (!apiKeyId || row.api_key_id !== apiKeyId) return null; + // #12150 P2 surface 2: the persisted clientRawRequest snapshot on this row had + // its video transcript cues structurally redacted to [redacted-video-transcript] + // before storage (videoBridgeSnapshotRedaction, marker written by the call-log + // path). The stored input therefore no longer carries the client's real cue + // text -- reconstructing a continuation off it would forward the placeholder + // upstream as if it were genuine history. Fail closed so the client resends + // full history, exactly like a real previous_response_not_found. + if (row.video_content_removed === 1) return null; const { artifact, state } = readCallArtifact(row.artifact_relpath); if (state !== "ready" || !artifact?.pipeline) return null; diff --git a/src/lib/db/schemaColumns.ts b/src/lib/db/schemaColumns.ts index 068c140c69..b288072dac 100644 --- a/src/lib/db/schemaColumns.ts +++ b/src/lib/db/schemaColumns.ts @@ -240,6 +240,14 @@ export function ensureCallLogsColumns(db: SqliteDatabase) { db.exec("ALTER TABLE call_logs ADD COLUMN request_summary TEXT DEFAULT NULL"); console.log("[DB] Added call_logs.request_summary column"); } + // added by 173_call_logs_video_content_removed; back-filled here because + // resolvePreviousResponseState SELECTs it on every continuation lookup — a + // lineage that skipped the migration would throw "no such column" there + // rather than fail closed. Same hole #12470 closed for provider_connections. + if (!columnNames.has("video_content_removed")) { + db.exec("ALTER TABLE call_logs ADD COLUMN video_content_removed INTEGER NOT NULL DEFAULT 0"); + console.log("[DB] Added call_logs.video_content_removed column"); + } if (!columnNames.has("correlation_id")) { db.exec("ALTER TABLE call_logs ADD COLUMN correlation_id TEXT DEFAULT NULL"); console.log("[DB] Added call_logs.correlation_id column"); diff --git a/src/lib/usage/callLogs.ts b/src/lib/usage/callLogs.ts index 5f0e3a03fc..3e1ed314de 100644 --- a/src/lib/usage/callLogs.ts +++ b/src/lib/usage/callLogs.ts @@ -522,6 +522,11 @@ async function saveCallLogOperation(entry: any): Promise { // this row's artifact for OmniRoute-native continuation. See // src/lib/db/responsesContinuationStore.ts. responseId: typeof entry.responseId === "string" ? entry.responseId : null, + // #12150 P2 surface 2: 1 when this request's persisted client snapshot had + // its video transcript cues structurally redacted, so + // resolvePreviousResponseState refuses to rehydrate it as continuation + // history. See src/lib/db/responsesContinuationStore.ts. + videoContentRemoved: entry.videoContentRemoved ? 1 : 0, }; const requestSummary = noLogEnabled @@ -570,7 +575,8 @@ async function saveCallLogOperation(entry: any): Promise { combo_name, combo_step_id, combo_execution_key, error_summary, detail_state, artifact_relpath, artifact_size_bytes, artifact_sha256, has_request_body, has_response_body, has_pipeline_details, request_summary, - correlation_id, model_pinned, session_tag, response_id, error_type + correlation_id, model_pinned, session_tag, response_id, error_type, + video_content_removed ) VALUES ( @id, @timestamp, @method, @path, @status, @model, @requestedModel, @provider, @@ -581,7 +587,8 @@ async function saveCallLogOperation(entry: any): Promise { @comboName, @comboStepId, @comboExecutionKey, @errorSummary, @detailState, @artifactRelPath, @artifactSizeBytes, @artifactSha256, @hasRequestBody, @hasResponseBody, @hasPipelineDetails, @requestSummary, - @correlationId, @modelPinned, @sessionTag, @responseId, @errorType + @correlationId, @modelPinned, @sessionTag, @responseId, @errorType, + @videoContentRemoved ) ` ).run({ diff --git a/tests/unit/db-schema-columns-split.test.ts b/tests/unit/db-schema-columns-split.test.ts index 9e7e249a6a..0587817500 100644 --- a/tests/unit/db-schema-columns-split.test.ts +++ b/tests/unit/db-schema-columns-split.test.ts @@ -11,6 +11,7 @@ import { ensureUsageHistoryColumns, ensureProviderConnectionsColumns, ensureProxyLogsColumns, + ensureCallLogsColumns, hasColumn, hasTable, quoteIdentifier, @@ -182,3 +183,27 @@ test("ensureProviderConnectionsColumns back-fills last_ping columns on a pre-123 db.close?.(); } }); + +// #12150 P2b: `resolvePreviousResponseState` SELECTs `video_content_removed` on +// every previous_response_id lookup. Migration 173 adds it, but a lineage that +// skipped 173 would raise "no such column" there instead of failing closed, so +// the reconciliation has to carry it too — the hole #12470 closed for +// provider_connections. +test("ensureCallLogsColumns back-fills video_content_removed on a pre-173 lineage", () => { + const db = openMemoryDb(); + try { + db.exec("CREATE TABLE call_logs (id TEXT PRIMARY KEY, timestamp TEXT)"); + assert.equal(hasColumn(db, "call_logs", "video_content_removed"), false); + + ensureCallLogsColumns(db); + + assert.equal(hasColumn(db, "call_logs", "video_content_removed"), true); + const row = db + .prepare("SELECT video_content_removed AS v FROM call_logs WHERE id = ?") + .get("missing") as { v: number } | undefined; + assert.equal(row, undefined, "empty table — the column just has to be selectable"); + assert.doesNotThrow(() => ensureCallLogsColumns(db)); + } finally { + db.close?.(); + } +}); diff --git a/tests/unit/responses-continuation-store.test.ts b/tests/unit/responses-continuation-store.test.ts index 0b0bce17c3..6d8cb66c5f 100644 --- a/tests/unit/responses-continuation-store.test.ts +++ b/tests/unit/responses-continuation-store.test.ts @@ -27,13 +27,15 @@ function insertCallLog(row: { apiKeyId: string | null; detailState: string; artifactRelPath: string | null; + videoContentRemoved?: 0 | 1; }) { const db = core.getDbInstance(); db.prepare( `INSERT INTO call_logs (id, timestamp, method, path, status, model, provider, account, duration, - tokens_in, tokens_out, api_key_id, detail_state, artifact_relpath, response_id) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` + tokens_in, tokens_out, api_key_id, detail_state, artifact_relpath, response_id, + video_content_removed) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` ).run( row.id, new Date().toISOString(), @@ -49,7 +51,8 @@ function insertCallLog(row: { row.apiKeyId, row.detailState, row.artifactRelPath, - row.responseId + row.responseId, + row.videoContentRemoved ?? 0 ); } @@ -398,6 +401,65 @@ test("resolvePreviousResponseState fails closed on an empty output array even wi assert.equal(store.resolvePreviousResponseState("resp_gen-empty-output", "key-1"), null); }); +test("resolvePreviousResponseState fails closed when the row had video content removed (#12150 P2)", () => { + // #12150 P2 surface 2: the persisted clientRawRequest snapshot had its video + // transcript cues structurally redacted to [redacted-video-transcript] before + // storage (videoBridgeSnapshotRedaction). The stored input therefore no longer + // carries the client's real cue text -- reconstructing a continuation off it + // would forward the placeholder upstream as if it were genuine history. When the + // owning row is marked video_content_removed=1 this must fail closed (return + // null) so the client resends full history, exactly like previous_response_not_found, + // even though the artifact itself is otherwise a perfectly resolvable 'ready' row. + insertCallLog({ + id: "log-video-removed", + responseId: "resp_video_removed", + apiKeyId: "key-1", + detailState: "ready", + artifactRelPath: "2026-01-01/log-video-removed.json", + videoContentRemoved: 1, + }); + writeArtifact("2026-01-01/log-video-removed.json", { + clientRawRequest: { + body: { + input: [{ type: "message", role: "user", content: "[redacted-video-transcript]" }], + }, + }, + providerRequest: { body: { input: [] } }, + clientResponse: { + id: "resp_video_removed", + output: [{ type: "message", role: "assistant", content: "hello" }], + }, + }); + + assert.equal(store.resolvePreviousResponseState("resp_video_removed", "key-1"), null); +}); + +test("resolvePreviousResponseState still resolves a normal row (video_content_removed=0)", () => { + // Guard the fail-closed above does not over-fire: an ordinary row (the default + // 0) resolves exactly as before. + insertCallLog({ + id: "log-video-notremoved", + responseId: "resp_video_notremoved", + apiKeyId: "key-1", + detailState: "ready", + artifactRelPath: "2026-01-01/log-video-notremoved.json", + videoContentRemoved: 0, + }); + writeArtifact("2026-01-01/log-video-notremoved.json", { + clientRawRequest: { body: { input: [{ type: "message", role: "user", content: "hi" }] } }, + providerRequest: { body: { input: [{ type: "message", role: "user", content: "hi" }] } }, + clientResponse: { + id: "resp_video_notremoved", + output: [{ type: "message", role: "assistant", content: "hello" }], + }, + }); + + assert.deepEqual(store.resolvePreviousResponseState("resp_video_notremoved", "key-1"), { + input: [{ type: "message", role: "user", content: "hi" }], + output: [{ type: "message", role: "assistant", content: "hello" }], + }); +}); + test("resolvePreviousResponseState returns null when detail logging was never captured for this row", () => { insertCallLog({ id: "log-5", diff --git a/tests/unit/save-call-log-persistence.test.ts b/tests/unit/save-call-log-persistence.test.ts index 8d59ad18d6..7eb0cef2e5 100644 --- a/tests/unit/save-call-log-persistence.test.ts +++ b/tests/unit/save-call-log-persistence.test.ts @@ -152,6 +152,73 @@ test("saveCallLog persists modelPinned=false as 0", async () => { db.prepare("DELETE FROM call_logs WHERE id = ?").run(testId); }); +test("call_logs table has video_content_removed column", () => { + const db = getDbInstance(); + const columns = db.prepare("PRAGMA table_info(call_logs)").all() as { name: string }[]; + const colNames = columns.map((c) => c.name); + assert.ok( + colNames.includes("video_content_removed"), + "call_logs should have video_content_removed column" + ); +}); + +test("saveCallLog persists videoContentRemoved=true as 1 (#12150 P2)", async () => { + const db = getDbInstance(); + const testId = `test-videoremoved-${Date.now()}`; + + await saveCallLog({ + id: testId, + method: "POST", + path: "/v1/responses", + status: 200, + model: "video-model", + provider: "test-provider", + duration: 500, + tokens: { in: 10, out: 5 }, + videoContentRemoved: true, + }); + + const row = db + .prepare("SELECT id, video_content_removed FROM call_logs WHERE id = ?") + .get(testId) as Record; + assert.ok(row, "row should exist"); + assert.equal( + row.video_content_removed, + 1, + "video_content_removed should be 1 when videoContentRemoved=true" + ); + + db.prepare("DELETE FROM call_logs WHERE id = ?").run(testId); +}); + +test("saveCallLog defaults video_content_removed to 0 when absent (#12150 P2)", async () => { + const db = getDbInstance(); + const testId = `test-novideoremoved-${Date.now()}`; + + await saveCallLog({ + id: testId, + method: "POST", + path: "/v1/chat/completions", + status: 200, + model: "normal-model", + provider: "test-provider", + duration: 500, + tokens: { in: 10, out: 5 }, + }); + + const row = db + .prepare("SELECT id, video_content_removed FROM call_logs WHERE id = ?") + .get(testId) as Record; + assert.ok(row, "row should exist"); + assert.equal( + row.video_content_removed, + 0, + "video_content_removed should default to 0 when not provided" + ); + + db.prepare("DELETE FROM call_logs WHERE id = ?").run(testId); +}); + test("getCallLogs returns modelPinned boolean", async () => { const db = getDbInstance(); const testId = `test-pinned-roundtrip-${Date.now()}`; diff --git a/tests/unit/video-bridge-log-redaction.test.ts b/tests/unit/video-bridge-log-redaction.test.ts index 268102d246..d828cb31a4 100644 --- a/tests/unit/video-bridge-log-redaction.test.ts +++ b/tests/unit/video-bridge-log-redaction.test.ts @@ -149,6 +149,41 @@ test("persisted requestBody carries the placeholder and never the raw transcript ); }); +test("#12150 P2 surface 2: persistAttemptLogs marks the call_logs row video_content_removed=1 when ctx.videoContentRemoved is true", async () => { + // The continuation fail-closed (resolvePreviousResponseState) depends on this + // marker being written for any request whose stored client snapshot had its + // video transcript redacted. This proves the ctx.videoContentRemoved signal + // reaches the persisted row; the row is the exact thing the continuation store + // reads back. + const id = "video-marker-1"; + persistAttemptLogs( + { status: 200, tokens: { input: 1, output: 2 } }, + baseCtx({ pendingRequestId: id, videoContentRemoved: true }) + ); + const row = await pollForCallLog(id); + assert.ok(row, "call log row should be persisted"); + const marker = coreDb + .getDbInstance() + .prepare("SELECT video_content_removed FROM call_logs WHERE id = ?") + .get(id) as { video_content_removed: number }; + assert.equal(marker.video_content_removed, 1); +}); + +test("#12150 P2 surface 2: the marker defaults to 0 for an ordinary (non-video) request", async () => { + const id = "video-marker-control-1"; + persistAttemptLogs( + { status: 200, tokens: { input: 1, output: 2 } }, + baseCtx({ pendingRequestId: id }) + ); + const row = await pollForCallLog(id); + assert.ok(row); + const marker = coreDb + .getDbInstance() + .prepare("SELECT video_content_removed FROM call_logs WHERE id = ?") + .get(id) as { video_content_removed: number }; + assert.equal(marker.video_content_removed, 0); +}); + test("control: without a redaction map the persisted requestBody keeps the original text (model path untouched)", async () => { const id = "video-control-1"; persistAttemptLogs( From 9d1a896c6058b2ade94c9078c2e54377b9aa76d3 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Sat, 5 Sep 2026 03:15:30 -0300 Subject: [PATCH 116/143] fix(tests): retire dead model ids from the chat-pipeline integration suite (base-red #12581) (#12670) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merged. It does what it says, and it also uncovered something — details below so the follow-up is not mistaken for a regression from this PR. Measured on `release/v3.8.51`, `tests/integration/chat-pipeline.test.ts`: | | line 580 | line 994 | line 1599 | |---|---|---|---| | tip | `410 !== 200` | `410 !== 200` | `502 !== 200` | | tip + this PR | passes | passes | passes | All three were retired model ids reaching the router and coming back 410/502. Swapping them for live ones is exactly the right fix and takes the suite from 25/28 to 27/28. **The one that remains, and why it is not yours:** with the 410 gone, `chat pipeline persists Codex responses cache and reasoning tokens to call logs` now runs past `assert.equal(response.status, 200)` and reaches line 592, where `callLog.provider` is `openai` and the test expects `codex`. That assertion was simply never reached before — the 410 short-circuited the test at line 580. I checked whether the model id chosen here was the cause, since `gpt-5.6-sol` is declared by 12 providers (`openai`, `github`, `cursor`, `kiro`, …). It is not: re-running with `gpt-5.3-codex-spark`, which only the `codex` provider declares, produces the identical `openai !== codex`. So it is provider resolution or the `seedConnection("codex")` harness, not catalog ambiguity. I reverted that experiment — this merged exactly as you wrote it. Filing that as its own issue with the trace. --- tests/integration/chat-pipeline.test.ts | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/tests/integration/chat-pipeline.test.ts b/tests/integration/chat-pipeline.test.ts index a7d441a9a5..2b0bf96d33 100644 --- a/tests/integration/chat-pipeline.test.ts +++ b/tests/integration/chat-pipeline.test.ts @@ -170,7 +170,7 @@ function buildOpenAIToolCallResponse({ ); } -function buildClaudeResponse(text = "ok", model = "claude-3-5-sonnet-20241022") { +function buildClaudeResponse(text = "ok", model = "claude-sonnet-4-6") { return new Response( JSON.stringify({ id: "msg_json", @@ -286,7 +286,7 @@ function buildOpenAIStreamResponse(text = "streamed from openai") { function buildOpenAIResponsesSSE({ text = "responses streamed from codex", - model = "gpt-5.1-codex", + model = "gpt-5.6-sol", usage = null, } = {}) { return new Response( @@ -567,7 +567,7 @@ test("chat pipeline persists Codex responses cache and reasoning tokens to call buildRequest({ url: "http://localhost/v1/responses", body: { - model: "codex/gpt-5.1-codex", + model: "codex/gpt-5.6-sol", stream: false, input: "Persist cache + reasoning usage", }, @@ -983,7 +983,7 @@ test("chat pipeline translates OpenAI requests to Claude and returns OpenAI-shap const response = await handleChat( buildRequest({ body: { - model: "claude/claude-3-5-sonnet-20241022", + model: "claude/claude-sonnet-4-6", stream: false, messages: [{ role: "user", content: "Hello Claude" }], }, @@ -1566,7 +1566,7 @@ test("chat pipeline falls back across combo models when the first provider fails name: "combo-fallback", strategy: "priority", config: { maxRetries: 0, retryDelayMs: 0 }, - models: ["openai/gpt-4o-mini", "claude/claude-3-5-sonnet-20241022"], + models: ["openai/gpt-4o-mini", "claude/claude-sonnet-4-6"], }); const attempts = []; From 92a617c23f47ecb5e82976f0140f9ac133117c60 Mon Sep 17 00:00:00 2001 From: tom <7740810+thomasmaerz@users.noreply.github.com> Date: Sat, 5 Sep 2026 19:30:48 -0700 Subject: [PATCH 117/143] fix(sse): re-enable prompt compression for native Codex passthrough (#12834) Native Codex passthrough (POST /v1/responses, provider=codex) was unconditionally excluded from prompt compression, writing only skip_reason='excluded' analytics rows. Prompt compression now depends only on the operator exclusions list; reactive compaction and combo overflow fail-fast intentionally still bypass (prompt-only scope). Closes #12793 Regression guard: tests/unit/codex-prompt-compression-passthrough.test.ts --- open-sse/handlers/chatCore.ts | 15 ++++- ...dex-prompt-compression-passthrough.test.ts | 59 +++++++++++++++++++ ...context-overflow-compression-probe.test.ts | 11 ++-- 3 files changed, 78 insertions(+), 7 deletions(-) create mode 100644 tests/unit/codex-prompt-compression-passthrough.test.ts diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 1b99f46cca..2c4a8babae 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -1356,9 +1356,18 @@ export async function handleChatCore({ const compressionSettings: CompressionConfig | null = compressionSettingsResult.settings; // #8034 — operator-named model/endpoint exclusions bypass the whole pipeline, exactly // like compression being globally disabled, so the body is provably byte-identical. - const compressionExcluded = - nativeCodexPassthrough || - isCompressionExcluded({ provider, model: effectiveModel }, compressionSettings?.exclusions); + // Native Codex passthrough is deliberately NOT part of this exclusion: prompt + // compression runs through adaptBodyForCompression() (Responses input[] → messages + // → restore) with codex tool-output eligibility guards, so native contexts still + // compress (regression: #8933 introduced the passthrough bypass, landed on release + // via #11088, silencing codex analytics to skip_reason='excluded'). Reactive + // compaction + combo overflow fail-fast below still bypass native passthrough — + // intentionally left for follow-up. Operators who want byte-identical passthrough + // can add `codex/*` to the exclusions list. + const compressionExcluded = isCompressionExcluded( + { provider, model: effectiveModel }, + compressionSettings?.exclusions + ); // A per-key opt-out is a request-scoped hard kill for prompt compression. It // deliberately does not disable the independent reactive context-fit safety // passes, matching the existing x-omniroute-compression: off contract. diff --git a/tests/unit/codex-prompt-compression-passthrough.test.ts b/tests/unit/codex-prompt-compression-passthrough.test.ts new file mode 100644 index 0000000000..120c4e1b35 --- /dev/null +++ b/tests/unit/codex-prompt-compression-passthrough.test.ts @@ -0,0 +1,59 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { + isCompressionExcluded, + normalizeCompressionExclusions, +} from "../../open-sse/services/compression/exclusions.ts"; + +const chatCoreSource = readFileSync( + new URL("../../open-sse/handlers/chatCore.ts", import.meta.url), + "utf8" +); + +// Regression guard for https://github.com/diegosouzapw/OmniRoute/issues/12793: +// native Codex passthrough (`POST /v1/responses`, provider `codex`) silently stopped +// producing compression analytics (100% `skip_reason='excluded'`) after #8933 added +// `nativeCodexPassthrough ||` to the prompt-compression exclusion (landed on release +// via #11088). Prompt compression must depend ONLY on the operator exclusions list. + +test("codex prompt compression is not gated on native passthrough", () => { + const match = chatCoreSource.match(/const compressionExcluded =([\s\S]*?);/); + assert.ok(match, "compressionExcluded assignment must exist in chatCore.ts"); + assert.doesNotMatch( + match[1], + /nativeCodexPassthrough/, + "prompt-compression exclusion must not reference nativeCodexPassthrough" + ); + assert.match(match[1], /isCompressionExcluded/); +}); + +test("codex target is compressible by default; operators can still opt out via exclusions", () => { + assert.equal( + isCompressionExcluded( + { provider: "codex", model: "gpt-5.6-terra" }, + normalizeCompressionExclusions([]) + ), + false + ); + assert.equal( + isCompressionExcluded( + { provider: "codex", model: "gpt-5.6-terra" }, + normalizeCompressionExclusions(["codex/*"]) + ), + true + ); +}); + +test("prompt-only scope: reactive compaction still bypasses native passthrough", () => { + // Deliberately left for follow-up (overflow fail-fast + history-rewriting safety). + // If these gates are ever lifted, update the PR body notes, not just this test. + assert.match( + chatCoreSource, + /reactiveContextCompactionEnabled\s*&&\s*!nativeCodexPassthrough\s*&&\s*estimatedTokens\s*>\s*threshold/ + ); + assert.match( + chatCoreSource, + /reactiveContextCompactionEnabled\s*&&\s*!nativeCodexPassthrough\s*&&\s*finalEstimatedInputTokens\s*>=\s*finalContextLimit/ + ); +}); diff --git a/tests/unit/combo-context-overflow-compression-probe.test.ts b/tests/unit/combo-context-overflow-compression-probe.test.ts index c854339c69..1b422cd9f9 100644 --- a/tests/unit/combo-context-overflow-compression-probe.test.ts +++ b/tests/unit/combo-context-overflow-compression-probe.test.ts @@ -148,12 +148,15 @@ test("#10225 combo keeps the fast 400 when compression is disabled", async () => // #10501-sweep #10503 — the deferral above is NOT target-aware by default: it only // checks operator-named compression exclusions, never whether chatCore will actually -// attempt compression for the resolved target. handleChatCore.ts unconditionally sets -// `compressionExcluded = nativeCodexPassthrough || ...` for a verified native Codex -// Responses passthrough target (open-sse/handlers/chatCore.ts) — deferring the +// attempt compression for the resolved target. chatCore's *reactive* compaction and +// last-resort gates still bypass native Codex passthrough targets +// (open-sse/handlers/chatCore.ts `!nativeCodexPassthrough` checks) — deferring the // preflight there means an oversized request sails past BOTH gates uncompressed. These // tests pin the fix: a native-codex-passthrough target must never count toward "can // compress", so the hard preflight stays active and no upstream dispatch happens. +// NOTE: prompt (proactive) compression DOES run for native passthrough since the +// codex analytics fix (see chatCore.ts `compressionExcluded`); only the reactive / +// combo-deferral bypasses remain. That split is intentional prompt-only scope. // NOTE on `clientManagedResponsesContext: false` below: these tests deliberately do // NOT set it, to isolate the fix from the PRE-EXISTING, unrelated early-return a few // lines above in knownContextOverflow.ts ("Native Codex Responses clients compact @@ -163,7 +166,7 @@ test("#10225 combo keeps the fast 400 when compression is disabled", async () => // circuits to true for `provider === "codex"` regardless of verification (see // passthroughHelpers.ts), so an UNVERIFIED request that nonetheless targets a `codex` // combo member over `/v1/responses` in openai-responses format still hits chatCore's -// compression bypass — exactly the gap `sourceFormat`/`endpointPath` (not the looser +// reactive compression bypass — exactly the gap `sourceFormat`/`endpointPath` (not the looser // `clientManagedResponsesContext` flag) now closes. test("#10503 handleComboChat: native-codex-passthrough pool fails FAST locally, zero upstream dispatches", async () => { saveModelsDevCapabilities({ codex: { "gpt-5.6-terra": capabilityEntry(272_000) } }); From 2b2d34eb53424d8ae24d2053364b035f1e0fd870 Mon Sep 17 00:00:00 2001 From: Soroush Ahmadi Date: Sun, 6 Sep 2026 06:02:02 +0330 Subject: [PATCH 118/143] fix(cursor): guard non-array tool_calls in request translator (#12691) --- changelog.d/fixes/12689-cursor-toolcalls-guard.md | 1 + open-sse/translator/request/openai-to-cursor.ts | 4 ++-- tests/unit/translator-openai-to-cursor.test.ts | 12 ++++++++++++ 3 files changed, 15 insertions(+), 2 deletions(-) create mode 100644 changelog.d/fixes/12689-cursor-toolcalls-guard.md diff --git a/changelog.d/fixes/12689-cursor-toolcalls-guard.md b/changelog.d/fixes/12689-cursor-toolcalls-guard.md new file mode 100644 index 0000000000..2c1e54e4e6 --- /dev/null +++ b/changelog.d/fixes/12689-cursor-toolcalls-guard.md @@ -0,0 +1 @@ +- **fix(cursor):** a non-array `tool_calls` on an assistant message no longer crashes the cursor request translator with a `TypeError`; both loops now require an array ([#12689](https://github.com/diegosouzapw/OmniRoute/issues/12689)) diff --git a/open-sse/translator/request/openai-to-cursor.ts b/open-sse/translator/request/openai-to-cursor.ts index 2c488c4618..0128a0d93a 100644 --- a/open-sse/translator/request/openai-to-cursor.ts +++ b/open-sse/translator/request/openai-to-cursor.ts @@ -80,7 +80,7 @@ function convertMessages(messages) { }; for (const msg of messages) { - if (msg.role === "assistant" && msg.tool_calls) { + if (msg.role === "assistant" && Array.isArray(msg.tool_calls)) { for (const tc of msg.tool_calls) { rememberToolMeta(tc.id || "", tc.function?.name || "tool"); } @@ -166,7 +166,7 @@ function convertMessages(messages) { const content = extractContent(msg.content); - if (msg.role === "assistant" && msg.tool_calls && msg.tool_calls.length > 0) { + if (msg.role === "assistant" && Array.isArray(msg.tool_calls) && msg.tool_calls.length > 0) { const assistantMsg: { role: string; content?: string; diff --git a/tests/unit/translator-openai-to-cursor.test.ts b/tests/unit/translator-openai-to-cursor.test.ts index 2d818f8a09..b4fd51625a 100644 --- a/tests/unit/translator-openai-to-cursor.test.ts +++ b/tests/unit/translator-openai-to-cursor.test.ts @@ -199,3 +199,15 @@ test("OpenAI -> Cursor accepts shorthand image_url string form", () => { { type: "image_url", image_url: { url: "data:image/png;base64,AAAA" } }, ]); }); + +test("#12689: non-array tool_calls is skipped instead of throwing TypeError", () => { + for (const toolCalls of [5, { a: 1 }, "x"]) { + const result = buildCursorRequest( + "gpt-4o", + { messages: [{ role: "assistant", content: "Working", tool_calls: toolCalls }] }, + false, + null + ); + assert.ok(result); + } +}); From f9a1cc8a9b7336e394ef921c753f5da691798df9 Mon Sep 17 00:00:00 2001 From: groovecityJO Date: Sat, 5 Sep 2026 19:32:26 -0700 Subject: [PATCH 119/143] fix: resolve SqliteError no such table compression_run_telemetry during cleanup (#12682) --- src/lib/db/cleanup.ts | 3 +++ src/lib/db/compressionRunTelemetry.ts | 5 ++--- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/src/lib/db/cleanup.ts b/src/lib/db/cleanup.ts index 6f0cd95930..a837909df7 100644 --- a/src/lib/db/cleanup.ts +++ b/src/lib/db/cleanup.ts @@ -16,6 +16,7 @@ import { tableExists, type DeleteByPeriodTarget, } from "./cleanup/usagePurge"; +import { ensureCompressionRunTelemetryTable } from "./compressionRunTelemetry"; interface CleanupResult { deleted: number; @@ -380,6 +381,7 @@ export async function cleanupXpAuditLog(): Promise { */ export async function cleanupCompressionRunTelemetry(): Promise { const db = getDbInstance(); + ensureCompressionRunTelemetryTable(); const retention = getRetentionSettings(); const retentionDays = retention.compressionRunTelemetry; @@ -666,6 +668,7 @@ export async function resetUsageHistory(period: string): Promise; } -function ensureCompressionRunTelemetryTable(): void { +export function ensureCompressionRunTelemetryTable(): void { const db = getDbInstance(); // `CREATE TABLE IF NOT EXISTS` is idempotent and cheap; run it unconditionally so the // table self-heals if it was dropped (e.g. test isolation) under the same db handle. @@ -114,8 +114,7 @@ export function getCompressionRunTelemetrySummary(): CompressionRunTelemetrySumm try { const styles = JSON.parse(row.output_styles) as Array<{ id: string }>; for (const style of styles) { - summary.appliedStyleCounts[style.id] = - (summary.appliedStyleCounts[style.id] ?? 0) + 1; + summary.appliedStyleCounts[style.id] = (summary.appliedStyleCounts[style.id] ?? 0) + 1; } } catch { // ignore a corrupt JSON cell From b345c7f6cd4e1590d1177540813302375a75e332 Mon Sep 17 00:00:00 2001 From: Dizzle <112548150+maxmad64bis@users.noreply.github.com> Date: Sun, 6 Sep 2026 23:36:27 +0200 Subject: [PATCH 120/143] feat(opencode): opencode v2 plugin publishing the OmniRoute catalog (#12870) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit opencode v2 loads plugins through a contract the existing @omniroute/opencode-plugin cannot satisfy: v1 exports plugin factories with an auth/provider/config/tool hook object, v2 expects a default define({id, setup}) carrying catalog and integration domains. One package would have to satisfy both loaders from a single entrypoint. An opencode v2 install therefore has no route to an OmniRoute gateway at all: no model discovery, no combos, no enrichment. This adds @omniroute/opencode-plugin-v2, a self-contained package. The v1 plugin is untouched, so v1 users see no move, no migration and no breaking version. The two packages deliberately share no code and no release: the mapping logic here began as a port of v1's and now lives in this package, which keeps either one free to change without a coordinated publish. The plugin publishes models, combos and auto-combos into the host catalog, refreshes them lazily behind a 300s TTL, and keeps serving the last known catalog from an on-disk snapshot when the gateway is unreachable. Publishing is staged: models and combos are what a catalog is, so they go out as soon as they are known, while auto-combos, the provider list and the enrichment overlay fold into the snapshot when they land. Gating the publish on all of them made the catalog hostage to the slowest source — a gateway that accepts the connection and never answers /api/combos/auto left everything unpublished until that fetch timed out, which is longer than a short-lived host stays alive. Display names carry what the gateway knows about a model: the upstream provider it routes to, whether it is free, and the budget that comes with it. Those parts were already fetched and then dropped, so two connections selling the same model looked identical in the picker. The provider prefix can be turned off with `providerTag: false`. The on-disk snapshot carries that overlay too, under a size cap, so a cold start opens on named models rather than raw ids. The host is asked to reload only when the catalog or the overlay actually moved, never once per refresh window. The gateway key comes from the host credential store when one is connected, so connecting the integration from opencode is enough and no secret needs to sit in opencode.json; a plugin option and an environment variable remain as fallbacks, and a host too old to expose a credential store still loads. Nothing is silent when a key is missing or refused: an absent key is named once at startup with the three ways to supply one, and an enrichment source the gateway rejects is reported per endpoint with what the catalog loses. Those three failures used to be empty catch blocks, which turned a management token the gateway refuses into a catalog of raw model ids with no explanation. Tool calling to Gemini keeps working. Gemini answers 400 INVALID_ARGUMENT for an entire request whose tool declarations carry $schema, $ref or additionalProperties. The v1 plugin handled it by wrapping fetch and rewriting the JSON body; v2 does it on the language model, where the tools are still structured data, and only for Gemini models of this provider. It can be turned off with geminiSanitization: false, and a host exposing no aisdk domain loads without it. The catalog contract itself is a moving target, so the plugin adapts to the host instead of assuming one shape. The released CLI keeps the aisdk package, the endpoint (as settings.baseURL), the request headers and the variant options directly on the model and provider; the current SDK types keep the same information inside an api block. Writing only the api block yields a catalog the released CLI lists but cannot route. Rather than key off a version list that goes stale on the next release, the plugin reads the shape the host seeds into the catalog draft and publishes accordingly: a seed with a top-level package and no api block gets both field sets, a seed with an api block gets that block alone, and an undisclosed seed gets both. None of the legacy keys collide with a key of the current types, so the two shapes coexist on one object, variants included. Four v1 behaviours are deliberately not carried over, because v2 either owns them or no longer needs them: the plugin-side debug log (the host has its own logging), the compression-metadata suffix on combo names, the MCP auto-emit (the v2 host owns MCP), and the omni-sync command plus its background timer (the TTL and a content fingerprint drive catalog.reload instead). A refresh never downgrades what is already published: the previous overlay is carried forward until the new one lands, so names, pricing and the usable filter no longer drop out for the length of every TTL window. The disk snapshot is read after the credential is resolved, because it is keyed by that credential — reading it earlier looked up the identity the options carry rather than the one in use, and rejected a perfectly good catalog exactly when the gateway was down. The tool-schema cleaner now knows where a schema ends and a property name begins. Stripping keywords by name anywhere in the tree deleted a tool parameter called `ref` while leaving it in `required`, handing the model a schema it could not satisfy; a `$ref` it cannot resolve now forwards the tool untouched instead of widening it to accept anything. Gemini detection is anchored on the model family, so `gemini-compatible-proxy` is no longer treated as a Gemini model. A source the gateway refuses is reported on the library entry point as well, not only through the plugin, so the usable-provider filter can no longer disable itself in silence. `providerId` is bounded to a safe character set because it reaches a filesystem path, `hiddenModels` covers combos as it already covered models, the Anthropic block gets the gateway root rather than a doubled `/v1`, an unparseable tool schema forwards the tool instead of failing the request, and the package typechecks under the same settings as the v1 plugin. CI mirrors the existing plugin workflow: install, build and test on Node 22 and 24, for both packages. The plugin SDK stays pinned, and the host-shape assertions carry the risk of a contract move rather than a check against a rolling upstream tag. Co-authored-by: Max --- .github/workflows/npm-publish.yml | 89 + .github/workflows/opencode-plugin-ci.yml | 38 +- @omniroute/opencode-plugin-v2/.gitignore | 4 + @omniroute/opencode-plugin-v2/LICENSE | 21 + @omniroute/opencode-plugin-v2/README.md | 107 + @omniroute/opencode-plugin-v2/RELEASE.md | 9 + .../opencode-plugin-v2/package-lock.json | 2364 +++++++++++++++++ @omniroute/opencode-plugin-v2/package.json | 67 + @omniroute/opencode-plugin-v2/src/cache.ts | 211 ++ @omniroute/opencode-plugin-v2/src/catalog.ts | 798 ++++++ @omniroute/opencode-plugin-v2/src/compat.ts | 66 + .../opencode-plugin-v2/src/credentials.ts | 101 + .../src/enrichment-report.ts | 41 + .../opencode-plugin-v2/src/gemini-language.ts | 43 + @omniroute/opencode-plugin-v2/src/index.ts | 539 ++++ @omniroute/opencode-plugin-v2/src/options.ts | 117 + .../src/shared/auto-combos.ts | 219 ++ .../src/shared/combos-map.ts | 254 ++ .../opencode-plugin-v2/src/shared/enrich.ts | 606 +++++ .../src/shared/fingerprint.ts | 127 + .../opencode-plugin-v2/src/shared/gemini.ts | 166 ++ .../opencode-plugin-v2/src/shared/index.ts | 9 + .../opencode-plugin-v2/src/shared/logger.ts | 81 + .../src/shared/models-map.ts | 323 +++ .../opencode-plugin-v2/src/shared/naming.ts | 295 ++ .../opencode-plugin-v2/src/shared/usable.ts | 171 ++ .../tests/api-package.test.ts | 93 + .../tests/auto-combos.test.ts | 196 ++ .../tests/cache-ttl-snapshot.test.ts | 378 +++ .../opencode-plugin-v2/tests/catalog.test.ts | 330 +++ .../opencode-plugin-v2/tests/compat.test.ts | 34 + .../tests/credentials.test.ts | 128 + .../tests/enrichment-attribution.test.ts | 59 + .../tests/enrichment-render.test.ts | 64 + .../tests/enrichment-report.test.ts | 106 + .../tests/enrichment.test.ts | 124 + .../tests/fixtures/catalog.json | 68 + .../tests/fixtures/v1-parity.json | 301 +++ .../tests/gemini-language.test.ts | 226 ++ .../tests/host-contract.test.ts | 164 ++ .../opencode-plugin-v2/tests/index.test.ts | 244 ++ .../tests/management-token.test.ts | 233 ++ .../tests/nested-combos.test.ts | 236 ++ .../opencode-plugin-v2/tests/options.test.ts | 129 + .../opencode-plugin-v2/tests/parity.test.ts | 224 ++ .../tests/publish-guard.test.ts | 137 + .../tests/refresh-failopen.test.ts | 214 ++ .../tests/shared-anthropic-prefixes.test.ts | 165 ++ .../tests/shared-auto-combos.test.ts | 196 ++ .../tests/shared-combos-map.test.ts | 77 + .../tests/shared-enrich.test.ts | 47 + .../tests/shared-enrichment-fetcher.test.ts | 125 + .../shared-enrichment-source-errors.test.ts | 56 + .../tests/shared-fetch-timeout.test.ts | 49 + .../tests/shared-fingerprint.test.ts | 62 + .../tests/shared-gemini.test.ts | 144 + .../tests/shared-logger.test.ts | 130 + .../tests/shared-models-map.test.ts | 94 + .../tests/shared-naming.test.ts | 75 + .../tests/shared-usable.test.ts | 248 ++ .../tests/smoke-types.test.ts | 93 + .../tests/snapshot-stale-entries.test.ts | 219 ++ .../tests/staged-refresh.test.ts | 452 ++++ .../tests/timeouts-logger.test.ts | 161 ++ .../tests/usable-only.test.ts | 222 ++ .../tests/warm-snapshot-identity.test.ts | 88 + @omniroute/opencode-plugin-v2/tsconfig.json | 24 + @omniroute/opencode-plugin-v2/tsup.config.ts | 16 + .../features/12870-opencode-plugin-v2.md | 1 + docs/README.md | 1 + docs/guides/CLI-INTEGRATIONS.md | 10 + docs/guides/OPENCODE-V2-PLUGIN.md | 133 + docs/guides/REMOTE-MODE.md | 5 + 73 files changed, 13440 insertions(+), 7 deletions(-) create mode 100644 @omniroute/opencode-plugin-v2/.gitignore create mode 100644 @omniroute/opencode-plugin-v2/LICENSE create mode 100644 @omniroute/opencode-plugin-v2/README.md create mode 100644 @omniroute/opencode-plugin-v2/RELEASE.md create mode 100644 @omniroute/opencode-plugin-v2/package-lock.json create mode 100644 @omniroute/opencode-plugin-v2/package.json create mode 100644 @omniroute/opencode-plugin-v2/src/cache.ts create mode 100644 @omniroute/opencode-plugin-v2/src/catalog.ts create mode 100644 @omniroute/opencode-plugin-v2/src/compat.ts create mode 100644 @omniroute/opencode-plugin-v2/src/credentials.ts create mode 100644 @omniroute/opencode-plugin-v2/src/enrichment-report.ts create mode 100644 @omniroute/opencode-plugin-v2/src/gemini-language.ts create mode 100644 @omniroute/opencode-plugin-v2/src/index.ts create mode 100644 @omniroute/opencode-plugin-v2/src/options.ts create mode 100644 @omniroute/opencode-plugin-v2/src/shared/auto-combos.ts create mode 100644 @omniroute/opencode-plugin-v2/src/shared/combos-map.ts create mode 100644 @omniroute/opencode-plugin-v2/src/shared/enrich.ts create mode 100644 @omniroute/opencode-plugin-v2/src/shared/fingerprint.ts create mode 100644 @omniroute/opencode-plugin-v2/src/shared/gemini.ts create mode 100644 @omniroute/opencode-plugin-v2/src/shared/index.ts create mode 100644 @omniroute/opencode-plugin-v2/src/shared/logger.ts create mode 100644 @omniroute/opencode-plugin-v2/src/shared/models-map.ts create mode 100644 @omniroute/opencode-plugin-v2/src/shared/naming.ts create mode 100644 @omniroute/opencode-plugin-v2/src/shared/usable.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/api-package.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/auto-combos.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/cache-ttl-snapshot.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/catalog.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/compat.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/credentials.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/enrichment-attribution.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/enrichment-render.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/enrichment-report.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/enrichment.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/fixtures/catalog.json create mode 100644 @omniroute/opencode-plugin-v2/tests/fixtures/v1-parity.json create mode 100644 @omniroute/opencode-plugin-v2/tests/gemini-language.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/host-contract.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/index.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/management-token.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/nested-combos.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/options.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/parity.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/publish-guard.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/refresh-failopen.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/shared-anthropic-prefixes.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/shared-auto-combos.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/shared-combos-map.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/shared-enrich.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/shared-enrichment-fetcher.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/shared-enrichment-source-errors.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/shared-fetch-timeout.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/shared-fingerprint.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/shared-gemini.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/shared-logger.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/shared-models-map.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/shared-naming.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/shared-usable.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/smoke-types.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/snapshot-stale-entries.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/staged-refresh.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/timeouts-logger.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/usable-only.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tests/warm-snapshot-identity.test.ts create mode 100644 @omniroute/opencode-plugin-v2/tsconfig.json create mode 100644 @omniroute/opencode-plugin-v2/tsup.config.ts create mode 100644 changelog.d/features/12870-opencode-plugin-v2.md create mode 100644 docs/guides/OPENCODE-V2-PLUGIN.md diff --git a/.github/workflows/npm-publish.yml b/.github/workflows/npm-publish.yml index 5ba76f069f..fddcff49cc 100644 --- a/.github/workflows/npm-publish.yml +++ b/.github/workflows/npm-publish.yml @@ -573,3 +573,92 @@ jobs: fi npm publish --provenance --access public --ignore-scripts echo "✅ Published ${PKG_NAME}@${PKG_VERSION}" + + publish-opencode-plugin-v2: + runs-on: ubuntu-latest + permissions: + contents: read + id-token: write # npm provenance + steps: + - name: Checkout + uses: actions/checkout@v7 + with: + persist-credentials: false + fetch-depth: 0 + # Full history needed for auto-bump: git diff against previous release tag + + - name: Setup Node.js + uses: actions/setup-node@v7 + with: + node-version: ${{ env.NPM_PUBLISH_NODE_VERSION }} + registry-url: https://registry.npmjs.org + + - name: Auto-bump plugin-v2 version if plugin-v2 changed since last release + id: bump + working-directory: "@omniroute/opencode-plugin-v2" + env: + CURRENT_TAG: ${{ github.ref_name }} + run: | + set -euo pipefail + + PKG_VERSION=$(node -p "require('./package.json').version") + PKG_NAME=$(node -p "require('./package.json').name") + + # 1) Skip if current version is not yet published (no bump needed) + PUBLISHED="$(npm view "${PKG_NAME}@${PKG_VERSION}" version 2>/dev/null || true)" + if [ "$PUBLISHED" != "$PKG_VERSION" ]; then + echo "✅ ${PKG_NAME}@${PKG_VERSION} is new — no bump needed." + echo "bumped=false" >> "$GITHUB_OUTPUT" + exit 0 + fi + + # 2) Find the previous release tag (exclude the current one) + PREV_TAG=$(git tag -l 'v*' --sort=-version:refname \ + | grep -v "^${CURRENT_TAG}$" | head -1 || echo "") + if [ -z "$PREV_TAG" ]; then + echo "No previous tag to compare — skipping bump." + echo "bumped=false" >> "$GITHUB_OUTPUT" + exit 0 + fi + + # 3) Check if plugin-v2 dir actually changed since that tag + if git diff --quiet "$PREV_TAG" -- "@omniroute/opencode-plugin-v2/"; then + echo "⏭️ No plugin-v2 changes since $PREV_TAG — nothing to publish." + echo "bumped=false" >> "$GITHUB_OUTPUT" + exit 0 + fi + + # 4) Auto-bump patch version + npm version patch --no-git-tag-version --allow-same-version + NEW_VERSION=$(node -p "require('./package.json').version") + echo "bumped=true" >> "$GITHUB_OUTPUT" + echo "📦 Auto-bumped ${PKG_NAME} from ${PKG_VERSION} to ${NEW_VERSION}" + + - name: Install plugin-v2 dependencies + working-directory: "@omniroute/opencode-plugin-v2" + run: npm install --no-audit --no-fund + + - name: Build plugin-v2 + working-directory: "@omniroute/opencode-plugin-v2" + run: npm run clean && npm run build + + - name: Test plugin-v2 + working-directory: "@omniroute/opencode-plugin-v2" + run: npm test + + - name: Publish @omniroute/opencode-plugin-v2 to npm + working-directory: "@omniroute/opencode-plugin-v2" + env: + NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }} + run: | + set -euo pipefail + PKG_VERSION=$(node -p "require('./package.json').version") + PKG_NAME=$(node -p "require('./package.json').name") + # Same hardened skip-check as the main job (no --silent flag). + PUBLISHED="$(npm view "${PKG_NAME}@${PKG_VERSION}" version 2>/dev/null || true)" + if [ "$PUBLISHED" = "$PKG_VERSION" ]; then + echo "⚠️ ${PKG_NAME}@${PKG_VERSION} is already published on npm — skipping." + exit 0 + fi + npm publish --provenance --access public --ignore-scripts + echo "✅ Published ${PKG_NAME}@${PKG_VERSION}" diff --git a/.github/workflows/opencode-plugin-ci.yml b/.github/workflows/opencode-plugin-ci.yml index 0e26c0e608..9b94b688b6 100644 --- a/.github/workflows/opencode-plugin-ci.yml +++ b/.github/workflows/opencode-plugin-ci.yml @@ -5,10 +5,12 @@ on: branches: [main, "release/**"] paths: - "@omniroute/opencode-plugin/**" + - "@omniroute/opencode-plugin-v2/**" pull_request: branches: [main, "release/**"] paths: - "@omniroute/opencode-plugin/**" + - "@omniroute/opencode-plugin-v2/**" types: [opened, synchronize, reopened, ready_for_review] workflow_dispatch: @@ -44,10 +46,33 @@ jobs: - run: npm run build - run: npm test + test-v2: + name: Test v2 (Node ${{ matrix.node }}) + runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + node: ["22", "24"] + defaults: + run: + working-directory: "@omniroute/opencode-plugin-v2" + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + - uses: actions/setup-node@v7 + with: + node-version: ${{ matrix.node }} + cache: npm + cache-dependency-path: "@omniroute/opencode-plugin-v2/package-lock.json" + - run: npm ci --no-audit --no-fund + - run: npm run build + - run: npm test + build: name: Build runs-on: ubuntu-latest - needs: test + needs: [test, test-v2] steps: - uses: actions/checkout@v7 with: @@ -55,12 +80,11 @@ jobs: - uses: actions/setup-node@v7 with: node-version: "22" - cache: npm - cache-dependency-path: "@omniroute/opencode-plugin/package-lock.json" - - run: npm install --no-audit --no-fund - - run: npm run build + - name: Build plugin-v2 artifact + working-directory: "@omniroute/opencode-plugin-v2" + run: npm ci --no-audit --no-fund && npm run build - uses: actions/upload-artifact@v7 with: - name: opencode-plugin-dist - path: "@omniroute/opencode-plugin/dist" + name: opencode-plugin-v2-dist + path: "@omniroute/opencode-plugin-v2/dist" retention-days: 7 diff --git a/@omniroute/opencode-plugin-v2/.gitignore b/@omniroute/opencode-plugin-v2/.gitignore new file mode 100644 index 0000000000..7535211682 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/.gitignore @@ -0,0 +1,4 @@ +node_modules +dist +*.log +.DS_Store diff --git a/@omniroute/opencode-plugin-v2/LICENSE b/@omniroute/opencode-plugin-v2/LICENSE new file mode 100644 index 0000000000..e50b22c855 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 OmniRoute contributors + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/@omniroute/opencode-plugin-v2/README.md b/@omniroute/opencode-plugin-v2/README.md new file mode 100644 index 0000000000..873290a9d1 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/README.md @@ -0,0 +1,107 @@ +# @omniroute/opencode-plugin-v2 + +OpenCode v2 plugin (`define({ id, setup })`, Promise API) that publishes the live OmniRoute catalog — models from `/v1/models`, combos from `/api/combos` (least-common-denominator join), auto-combos from `/api/combos/auto`, enrichment (names + pricing), and usable-provider filtering — into the v2 `catalog.transform`, with `key` + `env` auth via `integration.transform`. + +Companion to `@omniroute/opencode-plugin` (OpenCode v1, same repo). The two packages are independent: this one carries its own catalog-mapping logic and the v1 plugin is left untouched. + +## Install + +```sh +npm install @omniroute/opencode-plugin-v2 +``` + +`opencode.json`: + +```json +{ + "plugins": [ + { + "package": "@omniroute/opencode-plugin-v2", + "options": { + "providerId": "omniroute", + "baseURL": "http://localhost:20128" + } + } + ] +} +``` + +## Credentials + +The plugin needs a gateway key to read the catalog, and looks for one in this +order: + +1. **The credential you connected in OpenCode.** The plugin registers an + integration, so `opencode auth` (or the Connect action in the model picker) + can store a key for it. Nothing is written to `opencode.json` — this is the + recommended route. +2. **`apiKey` in the plugin options**, when you want a per-project override. + Remember that this puts the key in a config file you may be committing. +3. **`OMNIROUTE_API_KEY` in the environment.** + +If none of the three yields a key, the catalog is empty and the plugin says so +once at startup rather than leaving you with a silent empty model list. + +### The management token is a different key + +Combos, provider health and enrichment (display names, pricing, free-tier +budgets) come from the gateway's `/api/*` endpoints, which most deployments +gate behind a **management** token rather than the inference key. Set it +explicitly: + +```json +"options": { + "baseURL": "http://localhost:20128", + "managementReadToken": "" +} +``` + +Left unset, `managementReadToken` falls back to `apiKey` for backwards +compatibility. When a gateway rejects that fallback, the catalog still +publishes — but with raw model ids instead of display names, no canonical +alias dedupe, no pricing and no combos. The plugin warns once per endpoint +when this happens, naming the endpoint and the consequence, so the degraded +catalog is never a mystery. + +## Options + +| Key | Default | Notes | +| -------------------------------- | ---------------------------------------------- | --------------------------------------------------------------------------------------------------------------------- | +| `providerId` | `"omniroute"` | Provider id and integration id; models publish under `/…` | +| `baseURL` | required | OmniRoute gateway root (no `/v1` suffix needed) | +| `apiKey` | connected credential, then `OMNIROUTE_API_KEY` | Chat key for `/v1/*` — see [Credentials](#credentials) | +| `managementReadToken` | falls back to `apiKey` | Management key for `/api/*` (combos, providers, enrichment) — usually **not** the same key | +| `displayName` | `"OmniRoute"` | Provider display name | +| `timeoutMs` | `10000` | Per-endpoint fetch timeout (auto-combos use 5s) | +| `modelCacheTtlMs` | `300000` | Catalog cache TTL; disk snapshot warms cold starts | +| `timeouts` | per-endpoint override | `{ models, combos, autoCombos, enrichment }` in ms; falls back to `timeoutMs` | +| `enrichment` | `true` | Fetch names + pricing (`/api/pricing*`, `/api/free-tier/summary`) | +| `providerTag` | `true` | Prefix a display name with the upstream provider it routes to | +| `geminiSanitization` | `true` | Strip `$schema`/`additionalProperties` from tool schemas sent to Gemini models (`$ref` tools are forwarded untouched) | +| `usableOnly` | `false` | Filter to healthy provisioned providers (`/api/providers`) | +| `visibleModels` / `hiddenModels` | `[]` | Exact-or-suffix allowlists, deny wins | +| `apiFormat.allowAnthropic` | `false` | Route allowlisted ids to the Anthropic API block | +| `apiFormat.anthropicModels` | `[]` | Full model ids routed to Anthropic | +| `apiFormat.anthropicPrefixes` | v1 defaults | Deprecated, warns once — prefer `anthropicModels` | +| `logLevel` / `startupDebug` | `warn` / `false` | Logger verbosity | + +## Tool calling on Gemini models + +Gemini answers `400 INVALID_ARGUMENT` — for the whole request, not just the +offending tool — when a tool declaration carries `$schema` or +`additionalProperties`. Anything that emits standard JSON Schema therefore +breaks tool calling as soon as the chain routes to Gemini. + +The plugin strips those keywords from tool schemas bound for a Gemini model of +this provider, and leaves every other request untouched. A tool carrying a +`$ref` is forwarded untouched instead of stripped: removing the reference +would widen the schema to "accept anything". Set +`"geminiSanitization": false` to turn it off. + +## Migrating from the v1 plugin + +The v2 plugin publishes provider id `X` bare. The v1 plugin published `opencode-X` (native-adapter gate). Sessions pinned to `opencode-X/...` must re-select the model under `X/...`. + +## License + +MIT diff --git a/@omniroute/opencode-plugin-v2/RELEASE.md b/@omniroute/opencode-plugin-v2/RELEASE.md new file mode 100644 index 0000000000..78d76c6ae3 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/RELEASE.md @@ -0,0 +1,9 @@ +# Release process — `@omniroute/opencode-plugin-v2` + +## Publishing + +One package, no ordering: bump `@omniroute/opencode-plugin-v2` (`npm version patch`) and publish it. The plugin carries its own copy of the mapping logic, so a release never has to be coordinated with another package. + +## Migration note (`opencode-X` → `X`) + +The v1 plugin published provider id `opencode-X` (native-adapter gate). The v2 plugin publishes `X` bare. Sessions pinned to `opencode-X/...` resolve `ModelUnavailableError` — users must re-select the model under `X/...`. diff --git a/@omniroute/opencode-plugin-v2/package-lock.json b/@omniroute/opencode-plugin-v2/package-lock.json new file mode 100644 index 0000000000..d709cd6779 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/package-lock.json @@ -0,0 +1,2364 @@ +{ + "name": "@omniroute/opencode-plugin-v2", + "version": "0.1.0", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "name": "@omniroute/opencode-plugin-v2", + "version": "0.1.0", + "license": "MIT", + "dependencies": { + "zod": "^4.4.3" + }, + "devDependencies": { + "@opencode-ai/plugin": "1.18.29", + "@types/node": "^22.19.19", + "tsup": "^8.5.1", + "tsx": "^4.22.3", + "typescript": "^5.9.3" + }, + "engines": { + "node": ">=22.22.3" + }, + "peerDependencies": { + "@opencode-ai/plugin": "*" + } + }, + "node_modules/@ai-sdk/provider": { + "version": "3.0.8", + "resolved": "https://registry.npmjs.org/@ai-sdk/provider/-/provider-3.0.8.tgz", + "integrity": "sha512-oGMAgGoQdBXbZqNG0Ze56CHjDZ1IDYOwGYxYjO5KLSlz5HiNQ9udIXsPZ61VWaHGZ5XW/jyjmr6t2xz2jGVwbQ==", + "dev": true, + "license": "Apache-2.0", + "dependencies": { + "json-schema": "^0.4.0" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/aix-ppc64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/aix-ppc64/-/aix-ppc64-0.28.1.tgz", + "integrity": "sha512-Svl7tq8k/08+p6CXPpRjQ1fKX+1odH/BQbb48fV6fj3CWHhsoIOoY87w1oHXm0qEpkIK3ZfVgp0hed3XBXzXMQ==", + "cpu": [ + "ppc64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "aix" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/android-arm": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/android-arm/-/android-arm-0.28.1.tgz", + "integrity": "sha512-0k2F129Xdio1TdJfzJ8sy1Q47vUD2NnwdhiAf7drUN1EBTfPf4hsFCtmMgu/6m8JSzsBrlmVjudMBQqOfG8usQ==", + "cpu": [ + "arm" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/android-arm64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/android-arm64/-/android-arm64-0.28.1.tgz", + "integrity": "sha512-34EGEbCIAgosYz6goLcopX6Mo7NyGv9tfwEM2/7Ce2VcVRk568iSvniGWcUXIy7wEDR1wzolcxcriFVrWYcwBg==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/android-x64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/android-x64/-/android-x64-0.28.1.tgz", + "integrity": "sha512-dbwY7ltSMDWsRatcRpCnES4F+im88OCUgGZjy52shC7GqHRE/cYlxNbB4Z4UpJswpcc4Qxd2oE/ufM0p61IKng==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/darwin-arm64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/darwin-arm64/-/darwin-arm64-0.28.1.tgz", + "integrity": "sha512-TZbWkQY7kvTAXbXUT7uVACR5cMHsDiSz9z7ZKAX/RTq/WJEk3QyRr0wZpNhBDX+/0CtdqUIJlOiodQcta6tY3Q==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/darwin-x64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/darwin-x64/-/darwin-x64-0.28.1.tgz", + "integrity": "sha512-zfdzgK9ACBNZLI/CyHTOx81SyNbM6YXn7rxSgX97VjyiPl9W1i4Ka4fgKECEoFCKGpvBj5qArWIGgQjOwkgskQ==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/freebsd-arm64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/freebsd-arm64/-/freebsd-arm64-0.28.1.tgz", + "integrity": "sha512-wG2EA8ENdEI0qhkSZMjfqrdY+ziCYCPMmtZjjIwOmXFjmyzEHn+UUxk5of+SYsjtfs3VpnlC7QLzSI5hY/rOAw==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "freebsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/freebsd-x64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/freebsd-x64/-/freebsd-x64-0.28.1.tgz", + "integrity": "sha512-i7dZ9vQgnvSCzi/rYCXNgtF/U+eKZNJBzu3eTQbRgHnM7tNSizLOkRFAl3qzVc/Op/u5YkHHa4pf/3DOYHthLQ==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "freebsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-arm": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-arm/-/linux-arm-0.28.1.tgz", + "integrity": "sha512-qVXBOHQS+d5Y722GwJzJUtOLlX7km3CraOaGormF1pDtPd2C/l1SHRPgjLunLGe51Sh5YYWKMFDyV4SxgMQYTQ==", + "cpu": [ + "arm" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-arm64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-arm64/-/linux-arm64-0.28.1.tgz", + "integrity": "sha512-yHs+0uc8+nvEAfAfxrWQKK5peSNzBc4PegcMO0EJ2hT71uA7vB8Ihg2e77R2P7SG5uYjPbHlLLmve4LLLRCf0g==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-ia32": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-ia32/-/linux-ia32-0.28.1.tgz", + "integrity": "sha512-d1z4ZuP0ajrfz/FhGT4vv278rX8KnPPJx8i5+AtK7TYbx9Le9F1hyzurZpkEyjkGa9dUGhQow4C1NmeGvqxN2w==", + "cpu": [ + "ia32" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-loong64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-loong64/-/linux-loong64-0.28.1.tgz", + "integrity": "sha512-M5sRjUVZrkm1OAPR3dlOYzNmN+loZKGVi1VUQGrwuqLcbR6qeAz+famMhjASeH3YVKvZz+zT1jlh/keC3Rj/lg==", + "cpu": [ + "loong64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-mips64el": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-mips64el/-/linux-mips64el-0.28.1.tgz", + "integrity": "sha512-mRObBZeHh2OxcBFPWE/FjylkRgZdYuiTR3vaTozquCGOH14iP9oN4x4Ge81CoIDYQrXmIxpFumJBu5MtZpnQJQ==", + "cpu": [ + "mips64el" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-ppc64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-ppc64/-/linux-ppc64-0.28.1.tgz", + "integrity": "sha512-slScBsMAb3GFDcdrCgLwZtPYRoH2H/youv10QiZyRjmsP48fznoveWytSgCI/R0ZcUgpc0ZhIUEx6LHts8yrfQ==", + "cpu": [ + "ppc64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-riscv64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-riscv64/-/linux-riscv64-0.28.1.tgz", + "integrity": "sha512-kw0owk1o0GFETUJyW0jc0G4Yzs0BHZn0JDZ8JRT088vjJYX777BAs1fDGxAC+q831qOs2DTC96mNsG2opdfyyQ==", + "cpu": [ + "riscv64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-s390x": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-s390x/-/linux-s390x-0.28.1.tgz", + "integrity": "sha512-/lAIjX8aYFRByhh6L5rYtPEDRqa9de/4V/juOXcta5frjvzXO4/sqEtyytse0g3zZFuWu5cDN0MkLz2qRDD2Ag==", + "cpu": [ + "s390x" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-x64": { + "version": "0.28.1", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/netbsd-arm64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/netbsd-arm64/-/netbsd-arm64-0.28.1.tgz", + "integrity": "sha512-oks0DYbLwWMmaakTsCb+zL4E+aHRVLom9IJZOAthMQEPiQmydXHkziYEsGYRx0uNV/IjEKGAV941JzH02pflqw==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "netbsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/netbsd-x64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/netbsd-x64/-/netbsd-x64-0.28.1.tgz", + "integrity": "sha512-aeL6lAnN89Hz43Mlh1G8ARasbuoYvSITDEx0tHh5b7jJnHcssqgjy9Yx430GDpmCa6OyrKoS0aNRjKundRizGg==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "netbsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/openbsd-arm64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/openbsd-arm64/-/openbsd-arm64-0.28.1.tgz", + "integrity": "sha512-MEFJe5C3R8pwXdZ5Y21oo6m7ePiS0d9pWucn99O/wvyJZChoIQKrQDxKrGeW8F5+T0okTHesAmDeiHDTIq0V/Q==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "openbsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/openbsd-x64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/openbsd-x64/-/openbsd-x64-0.28.1.tgz", + "integrity": "sha512-i/ZLIOafE0Z8cI/XANJAixoJL/uRAoS2xOA3rb0xN+KK0K177cMAsQYkzHtBrtMXAKuAc7HGgcWiZ/sRC1Nxgw==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "openbsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/openharmony-arm64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/openharmony-arm64/-/openharmony-arm64-0.28.1.tgz", + "integrity": "sha512-ge+Z7EXFNt2BO1oAMsVpiQ8EwndV9i1xXerAeTIK7AtPs3bKFXQM7nlRxDSIUIMeueR1CNXxqztLzdNeReKBJg==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "openharmony" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/sunos-x64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/sunos-x64/-/sunos-x64-0.28.1.tgz", + "integrity": "sha512-BEjgtECkL3vY+SaSQ6nzVfiALUeFxpawyp8Jmf5PtYhf1Ug40N1h/hxlhts+f1FvSvarEigdxS3BlSMI2PJLcQ==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "sunos" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/win32-arm64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/win32-arm64/-/win32-arm64-0.28.1.tgz", + "integrity": "sha512-lCv9eK/H6ZJWbE7bh2nw54CZ9M2nupBxJcTsdk/QQnWkdSjKGuxmmH8/GWrlT1eMmZfn4dGcCjRte397WqfQXA==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/win32-ia32": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/win32-ia32/-/win32-ia32-0.28.1.tgz", + "integrity": "sha512-zvb/mB2bSCoJOpoCBgYKKpX6YM6mJBlBUVUtVj41DlZJVEB6/0CKlRYxP5wWl1C1ILiCoAU5wZZ4q1P3qeS6Eg==", + "cpu": [ + "ia32" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/win32-x64": { + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/win32-x64/-/win32-x64-0.28.1.tgz", + "integrity": "sha512-bm4Mowrv+GXMlpWX++EcXw/iLyd1o3+bJkC2DkWXYVvgZCqD/bSj9ctZeAMC3cIxgjRVR2Dufaiu4YPxr5gW1A==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@jridgewell/gen-mapping": { + "version": "0.3.13", + "dev": true, + "license": "MIT", + "dependencies": { + "@jridgewell/sourcemap-codec": "^1.5.0", + "@jridgewell/trace-mapping": "^0.3.24" + } + }, + "node_modules/@jridgewell/resolve-uri": { + "version": "3.1.2", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=6.0.0" + } + }, + "node_modules/@jridgewell/sourcemap-codec": { + "version": "1.5.5", + "dev": true, + "license": "MIT" + }, + "node_modules/@jridgewell/trace-mapping": { + "version": "0.3.31", + "dev": true, + "license": "MIT", + "dependencies": { + "@jridgewell/resolve-uri": "^3.1.0", + "@jridgewell/sourcemap-codec": "^1.4.14" + } + }, + "node_modules/@msgpackr-extract/msgpackr-extract-darwin-arm64": { + "version": "3.0.4", + "resolved": "https://registry.npmjs.org/@msgpackr-extract/msgpackr-extract-darwin-arm64/-/msgpackr-extract-darwin-arm64-3.0.4.tgz", + "integrity": "sha512-LCkGo6JDfaBhgST7UpPWgNgLINpcpabaHfyz5OBx75nUYxBsaEPxjnyNjWpeb/xBup/682QnBfRBy2/LvPutZQ==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ] + }, + "node_modules/@msgpackr-extract/msgpackr-extract-darwin-x64": { + "version": "3.0.4", + "resolved": "https://registry.npmjs.org/@msgpackr-extract/msgpackr-extract-darwin-x64/-/msgpackr-extract-darwin-x64-3.0.4.tgz", + "integrity": "sha512-zExlW9zUJKZH/tOtVMttwjKa4Xm/3KcNjnE3dPN92uCktwavMxpgCA3MoJK/DOnTWsQgo224OaST27/mPNAf+w==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ] + }, + "node_modules/@msgpackr-extract/msgpackr-extract-linux-arm": { + "version": "3.0.4", + "resolved": "https://registry.npmjs.org/@msgpackr-extract/msgpackr-extract-linux-arm/-/msgpackr-extract-linux-arm-3.0.4.tgz", + "integrity": "sha512-Tg3yX65f5GbtXLkrYEHE5oibZG9epyYWas7FogTTEJeDEF9JlXJzKgXaNhT3UXlTOeA+AfZpYZYZ0uPj7Cfquw==", + "cpu": [ + "arm" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@msgpackr-extract/msgpackr-extract-linux-arm64": { + "version": "3.0.4", + "resolved": "https://registry.npmjs.org/@msgpackr-extract/msgpackr-extract-linux-arm64/-/msgpackr-extract-linux-arm64-3.0.4.tgz", + "integrity": "sha512-dgX0P/9wGPJeHFBG+ZmhgE6bmtMt7NP5CRBGyyktpopdk/mW4POnrpQsSLtKI1dwpc+pPLuXHDh6vvskyQE/sw==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@msgpackr-extract/msgpackr-extract-linux-x64": { + "version": "3.0.4", + "resolved": "https://registry.npmjs.org/@msgpackr-extract/msgpackr-extract-linux-x64/-/msgpackr-extract-linux-x64-3.0.4.tgz", + "integrity": "sha512-8TNXMEjJc3QEy7R/x1INhgiU+XakDAFUzBhaz7+Rbrs8NH5UQeHQxxmzsSBJGyV6I1jW79undiQm8tOI+D+8FQ==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@msgpackr-extract/msgpackr-extract-win32-x64": { + "version": "3.0.4", + "resolved": "https://registry.npmjs.org/@msgpackr-extract/msgpackr-extract-win32-x64/-/msgpackr-extract-win32-x64-3.0.4.tgz", + "integrity": "sha512-CmCXPQrkbwExx3j946/PtHWHbYJiCRBRDl4BlkRQcJB/YOwQxJRTpoo7aTsortjgoJ1x7opzTSxn7C+ASSLVjQ==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ] + }, + "node_modules/@opencode-ai/plugin": { + "version": "1.18.29", + "resolved": "https://registry.npmjs.org/@opencode-ai/plugin/-/plugin-1.18.29.tgz", + "integrity": "sha512-IhF83EU4I/ASgWwvm0FIh1O3a8ZVuCLqPrCbmSHdSYq7GHxIYe773i6dHqcbrzCzvlG/lP2H+dyTQ+xPAwFpbw==", + "dev": true, + "license": "MIT", + "dependencies": { + "@ai-sdk/provider": "3.0.8", + "@opencode-ai/sdk": "1.18.29", + "effect": "4.0.0-beta.83", + "zod": "4.1.8" + }, + "peerDependencies": { + "@opentui/core": ">=0.4.5", + "@opentui/keymap": ">=0.4.5", + "@opentui/solid": ">=0.4.5" + }, + "peerDependenciesMeta": { + "@opentui/core": { + "optional": true + }, + "@opentui/keymap": { + "optional": true + }, + "@opentui/solid": { + "optional": true + } + } + }, + "node_modules/@opencode-ai/plugin/node_modules/zod": { + "version": "4.1.8", + "dev": true, + "license": "MIT", + "funding": { + "url": "https://github.com/sponsors/colinhacks" + } + }, + "node_modules/@opencode-ai/sdk": { + "version": "1.18.29", + "resolved": "https://registry.npmjs.org/@opencode-ai/sdk/-/sdk-1.18.29.tgz", + "integrity": "sha512-4CS+FoLPkymTlcga8jxivGDDb2AbWMIIl3b8+myoe2wtv/1ANYCErslgz1xy5hTVHymWE6CtVNKzRuPU0ED57A==", + "dev": true, + "license": "MIT", + "dependencies": { + "cross-spawn": "7.0.6" + } + }, + "node_modules/@rollup/rollup-android-arm-eabi": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-android-arm-eabi/-/rollup-android-arm-eabi-4.60.4.tgz", + "integrity": "sha512-F5QXMSiFebS9hKZj02XhWLLnRpJ3B3AROP0tWbFBSj+6kCbg5m9j5JoHKd4mmSVy5mS/IMQloYgYxCuJC0fxEQ==", + "cpu": [ + "arm" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "android" + ] + }, + "node_modules/@rollup/rollup-android-arm64": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-android-arm64/-/rollup-android-arm64-4.60.4.tgz", + "integrity": "sha512-GxxTKApUpzRhof7poWvCJHRF51C67u1R7D6DiluBE8wKU1u5GWE8t+v81JvJYtbawoBFX1hLv5Ei4eVjkWokaw==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "android" + ] + }, + "node_modules/@rollup/rollup-darwin-arm64": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-darwin-arm64/-/rollup-darwin-arm64-4.60.4.tgz", + "integrity": "sha512-tua0TaJxMOB1R0V0RS1jFZ/RpURFDJIOR2A6jWwQeawuFyS4gBW+rntLRaQd0EQ4bd6Vp44Z2rXW+YYDBsj6IA==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ] + }, + "node_modules/@rollup/rollup-darwin-x64": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-darwin-x64/-/rollup-darwin-x64-4.60.4.tgz", + "integrity": "sha512-CSKq7MsP+5PFIcydhAiR1K0UhEI1A2jWXVKHPCBZ151yOutENwvnPocgVHkivu2kviURtCEB6zUQw0vs8RrhMg==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ] + }, + "node_modules/@rollup/rollup-freebsd-arm64": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-freebsd-arm64/-/rollup-freebsd-arm64-4.60.4.tgz", + "integrity": "sha512-+O8OkVdyvXMtJEciu2wS/pzm1IxntEEQx3z5TAVy4l32G0etZn+RsA48ARRrFm6Ri8fvqPQfgrvNxSjKAbnd3g==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "freebsd" + ] + }, + "node_modules/@rollup/rollup-freebsd-x64": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-freebsd-x64/-/rollup-freebsd-x64-4.60.4.tgz", + "integrity": "sha512-Iw3oMskH3AfNuhU0MSN7vNbdi4me/NiYo2azqPz/Le16zHSa+3RRmliCMWWQmh4lcndccU40xcJuTYJZxNo/lw==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "freebsd" + ] + }, + "node_modules/@rollup/rollup-linux-arm-gnueabihf": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-linux-arm-gnueabihf/-/rollup-linux-arm-gnueabihf-4.60.4.tgz", + "integrity": "sha512-EIPRXTVQpHyF8WOo219AD2yEltPehLTcTMz2fn6JsatLYSzQf00hj3rulF+yauOlF9/FtM2WpkT/hJh/KJFGhA==", + "cpu": [ + "arm" + ], + "dev": true, + "libc": [ + "glibc" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@rollup/rollup-linux-arm-musleabihf": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-linux-arm-musleabihf/-/rollup-linux-arm-musleabihf-4.60.4.tgz", + "integrity": "sha512-J3Yh9PzzF1Ovah2At+lHiGQdsYgArxBbXv/zHfSyaiFQEqvNv7DcW98pCrmdjCZBrqBiKrKKe2V+aaSGWuBe/w==", + "cpu": [ + "arm" + ], + "dev": true, + "libc": [ + "musl" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@rollup/rollup-linux-arm64-gnu": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-linux-arm64-gnu/-/rollup-linux-arm64-gnu-4.60.4.tgz", + "integrity": "sha512-BFDEZMYfUvLn37ONE1yMBojPxnMlTFsdyNoqncT0qFq1mAfllL+ATMMJd8TeuVMiX84s1KbcxcZbXInmcO2mRg==", + "cpu": [ + "arm64" + ], + "dev": true, + "libc": [ + "glibc" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@rollup/rollup-linux-arm64-musl": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-linux-arm64-musl/-/rollup-linux-arm64-musl-4.60.4.tgz", + "integrity": "sha512-pc9EYOSlOgdQ2uPl1o9PF6/kLSgaUosia7gOuS8mB69IxJvlclko1MECXysjs5ryez1/5zjYqx3+xYU0TU6R1A==", + "cpu": [ + "arm64" + ], + "dev": true, + "libc": [ + "musl" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@rollup/rollup-linux-loong64-gnu": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-linux-loong64-gnu/-/rollup-linux-loong64-gnu-4.60.4.tgz", + "integrity": "sha512-NxnomyxYerDh5n4iLrNa+sH+Z+U4BMEE46V2PgQ/hoB909i8gV1M5wPojWg9fk1jWpO3IQnOs20K4wyZuFLEFQ==", + "cpu": [ + "loong64" + ], + "dev": true, + "libc": [ + "glibc" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@rollup/rollup-linux-loong64-musl": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-linux-loong64-musl/-/rollup-linux-loong64-musl-4.60.4.tgz", + "integrity": "sha512-nbJnQ8a3z1mtmrwImCYhc6BGpThAyYVRQxw9uKSKG4wR6aAYno9sVjJ0zaZcW9BPJX1GbrDPf+SvdWjgTuDmnw==", + "cpu": [ + "loong64" + ], + "dev": true, + "libc": [ + "musl" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@rollup/rollup-linux-ppc64-gnu": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-linux-ppc64-gnu/-/rollup-linux-ppc64-gnu-4.60.4.tgz", + "integrity": "sha512-2EU6acNrQLd8tYvo/LXW535wupT3m6fo7HKo6lr7ktQoItxTyOL1ZCR/GfGCuXl2vR+zmfI6eRXkSemafv+iVg==", + "cpu": [ + "ppc64" + ], + "dev": true, + "libc": [ + "glibc" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@rollup/rollup-linux-ppc64-musl": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-linux-ppc64-musl/-/rollup-linux-ppc64-musl-4.60.4.tgz", + "integrity": "sha512-WeBtoMuaMxiiIrO2IYP3xs6GMWkJP2C0EoT8beTLkUPmzV1i/UcOSVw1d5r9KBODtHKilG5yFxsGRnBbK3wJ4A==", + "cpu": [ + "ppc64" + ], + "dev": true, + "libc": [ + "musl" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@rollup/rollup-linux-riscv64-gnu": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-linux-riscv64-gnu/-/rollup-linux-riscv64-gnu-4.60.4.tgz", + "integrity": "sha512-FJHFfqpKUI3A10WrWKiFbBZ7yVbGT4q4B5o1qKFFojqpaYoh9LrQgqWCmmcxQzVSXYtyB5bzkXrYzlHTs21MYA==", + "cpu": [ + "riscv64" + ], + "dev": true, + "libc": [ + "glibc" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@rollup/rollup-linux-riscv64-musl": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-linux-riscv64-musl/-/rollup-linux-riscv64-musl-4.60.4.tgz", + "integrity": "sha512-mcEl6CUT5IAUmQf1m9FYSmVqCJlpQ8r8eyftFUHG8i9OhY7BkBXSUdnLH5DOf0wCOjcP9v/QO93zpmF1SptCCw==", + "cpu": [ + "riscv64" + ], + "dev": true, + "libc": [ + "musl" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@rollup/rollup-linux-s390x-gnu": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-linux-s390x-gnu/-/rollup-linux-s390x-gnu-4.60.4.tgz", + "integrity": "sha512-ynt3JxVd2w2buzoKDWIyiV1pJW93xlQic1THVLXilz429oijRpSHivZAgp65KBu+cMcgf1eVVjdnTLvPxgCuoQ==", + "cpu": [ + "s390x" + ], + "dev": true, + "libc": [ + "glibc" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@rollup/rollup-linux-x64-gnu": { + "version": "4.60.4", + "cpu": [ + "x64" + ], + "dev": true, + "libc": [ + "glibc" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@rollup/rollup-linux-x64-musl": { + "version": "4.60.4", + "cpu": [ + "x64" + ], + "dev": true, + "libc": [ + "musl" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@rollup/rollup-openbsd-x64": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-openbsd-x64/-/rollup-openbsd-x64-4.60.4.tgz", + "integrity": "sha512-VpTfOPHgVXEBeeR8hZ2O0F3aSso+JDWqTWmTmzcQKted54IAdUVbxE+j/MVxUsKa8L20HJhv3vUezVPoquqWjA==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "openbsd" + ] + }, + "node_modules/@rollup/rollup-openharmony-arm64": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-openharmony-arm64/-/rollup-openharmony-arm64-4.60.4.tgz", + "integrity": "sha512-IPOsh5aRYuLv/nkU51X10Bf75Bsf6+gZdx1X+QP5QM6lIJFHHqbHLG0uJn/hWthzo13UAc2umiUorqZy3axoZg==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "openharmony" + ] + }, + "node_modules/@rollup/rollup-win32-arm64-msvc": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-win32-arm64-msvc/-/rollup-win32-arm64-msvc-4.60.4.tgz", + "integrity": "sha512-4QzE9E81OohJ/HKzHhsqU+zcYYojVOXlFMs1DdyMT6qXl/niOH7AVElmmEdUNHHS/oRkc++d5k6Vy85zFs0DEw==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ] + }, + "node_modules/@rollup/rollup-win32-ia32-msvc": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-win32-ia32-msvc/-/rollup-win32-ia32-msvc-4.60.4.tgz", + "integrity": "sha512-zTPgT1YuHHcd+Tmx7h8aml0FWFVelV5N54oHow9SLj+GfoDy/huQ+UV396N/C7KpMDMiPspRktzM1/0r1usYEA==", + "cpu": [ + "ia32" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ] + }, + "node_modules/@rollup/rollup-win32-x64-gnu": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-win32-x64-gnu/-/rollup-win32-x64-gnu-4.60.4.tgz", + "integrity": "sha512-DRS4G7mi9lJxqEDezIkKCaUIKCrLUUDCUaCsTPCi/rtqaC6D/jjwslMQyiDU50Ka0JKpeXeRBFBAXwArY52vBw==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ] + }, + "node_modules/@rollup/rollup-win32-x64-msvc": { + "version": "4.60.4", + "resolved": "https://registry.npmjs.org/@rollup/rollup-win32-x64-msvc/-/rollup-win32-x64-msvc-4.60.4.tgz", + "integrity": "sha512-QVTUovf40zgTqlFVrKA1uXMVvU2QWEFWfAH8Wdc48IxLvrJMQVMBRjuQyUpzZCDkakImib9eVazbWlC6ksWtJw==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ] + }, + "node_modules/@standard-schema/spec": { + "version": "1.1.0", + "resolved": "https://registry.npmjs.org/@standard-schema/spec/-/spec-1.1.0.tgz", + "integrity": "sha512-l2aFy5jALhniG5HgqrD6jXLi/rUWrKvqN/qJx6yoJsgKhblVd+iqqU4RCXavm/jPityDo5TCvKMnpjKnOriy0w==", + "dev": true, + "license": "MIT" + }, + "node_modules/@types/estree": { + "version": "1.0.8", + "dev": true, + "license": "MIT" + }, + "node_modules/@types/node": { + "version": "22.19.19", + "dev": true, + "license": "MIT", + "dependencies": { + "undici-types": "~6.21.0" + } + }, + "node_modules/acorn": { + "version": "8.16.0", + "dev": true, + "license": "MIT", + "bin": { + "acorn": "bin/acorn" + }, + "engines": { + "node": ">=0.4.0" + } + }, + "node_modules/any-promise": { + "version": "1.3.0", + "dev": true, + "license": "MIT" + }, + "node_modules/bundle-require": { + "version": "5.1.0", + "dev": true, + "license": "MIT", + "dependencies": { + "load-tsconfig": "^0.2.3" + }, + "engines": { + "node": "^12.20.0 || ^14.13.1 || >=16.0.0" + }, + "peerDependencies": { + "esbuild": ">=0.18" + } + }, + "node_modules/cac": { + "version": "6.7.14", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=8" + } + }, + "node_modules/chokidar": { + "version": "4.0.3", + "dev": true, + "license": "MIT", + "dependencies": { + "readdirp": "^4.0.1" + }, + "engines": { + "node": ">= 14.16.0" + }, + "funding": { + "url": "https://paulmillr.com/funding/" + } + }, + "node_modules/commander": { + "version": "4.1.1", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 6" + } + }, + "node_modules/confbox": { + "version": "0.1.8", + "dev": true, + "license": "MIT" + }, + "node_modules/consola": { + "version": "3.4.2", + "dev": true, + "license": "MIT", + "engines": { + "node": "^14.18.0 || >=16.10.0" + } + }, + "node_modules/cross-spawn": { + "version": "7.0.6", + "resolved": "https://registry.npmjs.org/cross-spawn/-/cross-spawn-7.0.6.tgz", + "integrity": "sha512-uV2QOWP2nWzsy2aMp8aRibhi9dlzF5Hgh5SHaB9OiTGEyDTiJJyx0uy51QXdyWbtAHNua4XJzUKca3OzKUd3vA==", + "dev": true, + "license": "MIT", + "dependencies": { + "path-key": "^3.1.0", + "shebang-command": "^2.0.0", + "which": "^2.0.1" + }, + "engines": { + "node": ">= 8" + } + }, + "node_modules/debug": { + "version": "4.4.3", + "dev": true, + "license": "MIT", + "dependencies": { + "ms": "^2.1.3" + }, + "engines": { + "node": ">=6.0" + }, + "peerDependenciesMeta": { + "supports-color": { + "optional": true + } + } + }, + "node_modules/detect-libc": { + "version": "2.1.2", + "resolved": "https://registry.npmjs.org/detect-libc/-/detect-libc-2.1.2.tgz", + "integrity": "sha512-Btj2BOOO83o3WyH59e8MgXsxEQVcarkUOpEYrubB0urwnN10yQ364rsiByU11nZlqWYZm05i/of7io4mzihBtQ==", + "dev": true, + "license": "Apache-2.0", + "optional": true, + "engines": { + "node": ">=8" + } + }, + "node_modules/effect": { + "version": "4.0.0-beta.83", + "resolved": "https://registry.npmjs.org/effect/-/effect-4.0.0-beta.83.tgz", + "integrity": "sha512-0wsak8RtgGAr9UWSbVDgJHZcUqMSvicHcvaZv1MbMM7MCGgW4Rn/137J1MHQbwYPcwYGxT/IqehFd+UbYuj78w==", + "dev": true, + "license": "MIT", + "dependencies": { + "@standard-schema/spec": "^1.1.0", + "fast-check": "^4.8.0", + "find-my-way-ts": "^0.1.6", + "ini": "^7.0.0", + "kubernetes-types": "^1.30.0", + "msgpackr": "^2.0.1", + "multipasta": "^0.2.7", + "toml": "^4.1.1", + "uuid": "^14.0.0", + "yaml": "^2.9.0" + } + }, + "node_modules/esbuild": { + "version": "0.28.1", + "dev": true, + "hasInstallScript": true, + "license": "MIT", + "bin": { + "esbuild": "bin/esbuild" + }, + "engines": { + "node": ">=18" + }, + "optionalDependencies": { + "@esbuild/aix-ppc64": "0.28.1", + "@esbuild/android-arm": "0.28.1", + "@esbuild/android-arm64": "0.28.1", + "@esbuild/android-x64": "0.28.1", + "@esbuild/darwin-arm64": "0.28.1", + "@esbuild/darwin-x64": "0.28.1", + "@esbuild/freebsd-arm64": "0.28.1", + "@esbuild/freebsd-x64": "0.28.1", + "@esbuild/linux-arm": "0.28.1", + "@esbuild/linux-arm64": "0.28.1", + "@esbuild/linux-ia32": "0.28.1", + "@esbuild/linux-loong64": "0.28.1", + "@esbuild/linux-mips64el": "0.28.1", + "@esbuild/linux-ppc64": "0.28.1", + "@esbuild/linux-riscv64": "0.28.1", + "@esbuild/linux-s390x": "0.28.1", + "@esbuild/linux-x64": "0.28.1", + "@esbuild/netbsd-arm64": "0.28.1", + "@esbuild/netbsd-x64": "0.28.1", + "@esbuild/openbsd-arm64": "0.28.1", + "@esbuild/openbsd-x64": "0.28.1", + "@esbuild/openharmony-arm64": "0.28.1", + "@esbuild/sunos-x64": "0.28.1", + "@esbuild/win32-arm64": "0.28.1", + "@esbuild/win32-ia32": "0.28.1", + "@esbuild/win32-x64": "0.28.1" + } + }, + "node_modules/fast-check": { + "version": "4.9.0", + "resolved": "https://registry.npmjs.org/fast-check/-/fast-check-4.9.0.tgz", + "integrity": "sha512-7ms6T7SybUev/PQITciI0yLM2pOSFy5zpG8Ty7tQofcVaQUvrMXp6CBwqF6fThLCLOrfBtuHAtwq6Yu4XPCllg==", + "dev": true, + "funding": [ + { + "type": "individual", + "url": "https://github.com/sponsors/dubzzz" + }, + { + "type": "opencollective", + "url": "https://opencollective.com/fast-check" + } + ], + "license": "MIT", + "dependencies": { + "pure-rand": "^8.0.0" + }, + "engines": { + "node": ">=12.17.0" + } + }, + "node_modules/fdir": { + "version": "6.5.0", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=12.0.0" + }, + "peerDependencies": { + "picomatch": "^3 || ^4" + }, + "peerDependenciesMeta": { + "picomatch": { + "optional": true + } + } + }, + "node_modules/find-my-way-ts": { + "version": "0.1.6", + "resolved": "https://registry.npmjs.org/find-my-way-ts/-/find-my-way-ts-0.1.6.tgz", + "integrity": "sha512-a85L9ZoXtNAey3Y6Z+eBWW658kO/MwR7zIafkIUPUMf3isZG0NCs2pjW2wtjxAKuJPxMAsHUIP4ZPGv0o5gyTA==", + "dev": true, + "license": "MIT" + }, + "node_modules/fix-dts-default-cjs-exports": { + "version": "1.0.1", + "dev": true, + "license": "MIT", + "dependencies": { + "magic-string": "^0.30.17", + "mlly": "^1.7.4", + "rollup": "^4.34.8" + } + }, + "node_modules/fsevents": { + "version": "2.3.3", + "resolved": "https://registry.npmjs.org/fsevents/-/fsevents-2.3.3.tgz", + "integrity": "sha512-5xoDfX+fL7faATnagmWPpbFtwh/R77WmMMqqHGS65C3vvB0YHrgF+B1YmZ3441tMj5n63k0212XNoJwzlhffQw==", + "dev": true, + "hasInstallScript": true, + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": "^8.16.0 || ^10.6.0 || >=11.0.0" + } + }, + "node_modules/ini": { + "version": "7.0.0", + "resolved": "https://registry.npmjs.org/ini/-/ini-7.0.0.tgz", + "integrity": "sha512-ifK0CgjALofS5bkrcTy4RaQ9Vx2Knf/eLeIO+NaswQEpH1UblrtTSCIvN71qQDMq0PeQ/SSPojvEJp9vvvfr+w==", + "dev": true, + "license": "ISC", + "engines": { + "node": "^22.22.2 || ^24.15.0 || >=26.0.0" + } + }, + "node_modules/isexe": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/isexe/-/isexe-2.0.0.tgz", + "integrity": "sha512-RHxMLp9lnKHGHRng9QFhRCMbYAcVpn69smSGcq3f36xjgVVWThj4qqLbTLlq7Ssj8B+fIQ1EuCEGI2lKsyQeIw==", + "dev": true, + "license": "ISC" + }, + "node_modules/joycon": { + "version": "3.1.1", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=10" + } + }, + "node_modules/json-schema": { + "version": "0.4.0", + "resolved": "https://registry.npmjs.org/json-schema/-/json-schema-0.4.0.tgz", + "integrity": "sha512-es94M3nTIfsEPisRafak+HDLfHXnKBhV3vU5eqPcS3flIWqcxJWgXHXiey3YrpaNsanY5ei1VoYEbOzijuq9BA==", + "dev": true, + "license": "(AFL-2.1 OR BSD-3-Clause)" + }, + "node_modules/kubernetes-types": { + "version": "1.30.0", + "resolved": "https://registry.npmjs.org/kubernetes-types/-/kubernetes-types-1.30.0.tgz", + "integrity": "sha512-Dew1okvhM/SQcIa2rcgujNndZwU8VnSapDgdxlYoB84ZlpAD43U6KLAFqYo17ykSFGHNPrg0qry0bP+GJd9v7Q==", + "dev": true, + "license": "Apache-2.0" + }, + "node_modules/lilconfig": { + "version": "3.1.3", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=14" + }, + "funding": { + "url": "https://github.com/sponsors/antonk52" + } + }, + "node_modules/lines-and-columns": { + "version": "1.2.4", + "dev": true, + "license": "MIT" + }, + "node_modules/load-tsconfig": { + "version": "0.2.5", + "dev": true, + "license": "MIT", + "engines": { + "node": "^12.20.0 || ^14.13.1 || >=16.0.0" + } + }, + "node_modules/magic-string": { + "version": "0.30.21", + "dev": true, + "license": "MIT", + "dependencies": { + "@jridgewell/sourcemap-codec": "^1.5.5" + } + }, + "node_modules/mlly": { + "version": "1.8.2", + "dev": true, + "license": "MIT", + "dependencies": { + "acorn": "^8.16.0", + "pathe": "^2.0.3", + "pkg-types": "^1.3.1", + "ufo": "^1.6.3" + } + }, + "node_modules/ms": { + "version": "2.1.3", + "dev": true, + "license": "MIT" + }, + "node_modules/msgpackr": { + "version": "2.1.0", + "resolved": "https://registry.npmjs.org/msgpackr/-/msgpackr-2.1.0.tgz", + "integrity": "sha512-p/pBCVO63CsvvpkomUnNNag6+n38rULuDA6HHe70o2gtC8ODI52foF/4ko2qQcp6OiErJXTmrZeXmsGGHsIQNQ==", + "dev": true, + "license": "MIT", + "optionalDependencies": { + "msgpackr-extract": "^3.0.4" + } + }, + "node_modules/msgpackr-extract": { + "version": "3.0.4", + "resolved": "https://registry.npmjs.org/msgpackr-extract/-/msgpackr-extract-3.0.4.tgz", + "integrity": "sha512-4kmO/MdyUIkLIvTPr8VHLil4AtoKIoniWPIEk5+CDy0xnWC84azhSFmuJ7PxZdsYtiP5kEeQsORAVIeMgxT+Hw==", + "dev": true, + "hasInstallScript": true, + "license": "MIT", + "optional": true, + "dependencies": { + "node-gyp-build-optional-packages": "5.2.2" + }, + "bin": { + "download-msgpackr-prebuilds": "bin/download-prebuilds.js" + }, + "optionalDependencies": { + "@msgpackr-extract/msgpackr-extract-darwin-arm64": "3.0.4", + "@msgpackr-extract/msgpackr-extract-darwin-x64": "3.0.4", + "@msgpackr-extract/msgpackr-extract-linux-arm": "3.0.4", + "@msgpackr-extract/msgpackr-extract-linux-arm64": "3.0.4", + "@msgpackr-extract/msgpackr-extract-linux-x64": "3.0.4", + "@msgpackr-extract/msgpackr-extract-win32-x64": "3.0.4" + } + }, + "node_modules/multipasta": { + "version": "0.2.8", + "resolved": "https://registry.npmjs.org/multipasta/-/multipasta-0.2.8.tgz", + "integrity": "sha512-ZPWuMKyv0cSO29f7hozp+k6+crZbQijV8ipMvxNxRf2SwtYGTX1ZX89Kd20VV4H9Znonx+EQn+iy1wGQsJ+b+Q==", + "dev": true, + "license": "MIT" + }, + "node_modules/mz": { + "version": "2.7.0", + "dev": true, + "license": "MIT", + "dependencies": { + "any-promise": "^1.0.0", + "object-assign": "^4.0.1", + "thenify-all": "^1.0.0" + } + }, + "node_modules/node-gyp-build-optional-packages": { + "version": "5.2.2", + "resolved": "https://registry.npmjs.org/node-gyp-build-optional-packages/-/node-gyp-build-optional-packages-5.2.2.tgz", + "integrity": "sha512-s+w+rBWnpTMwSFbaE0UXsRlg7hU4FjekKU4eyAih5T8nJuNZT1nNsskXpxmeqSK9UzkBl6UgRlnKc8hz8IEqOw==", + "dev": true, + "license": "MIT", + "optional": true, + "dependencies": { + "detect-libc": "^2.0.1" + }, + "bin": { + "node-gyp-build-optional-packages": "bin.js", + "node-gyp-build-optional-packages-optional": "optional.js", + "node-gyp-build-optional-packages-test": "build-test.js" + } + }, + "node_modules/object-assign": { + "version": "4.1.1", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=0.10.0" + } + }, + "node_modules/path-key": { + "version": "3.1.1", + "resolved": "https://registry.npmjs.org/path-key/-/path-key-3.1.1.tgz", + "integrity": "sha512-ojmeN0qd+y0jszEtoY48r0Peq5dwMEkIlCOu6Q5f41lfkswXuKtYrhgoTpLnyIcHm24Uhqx+5Tqm2InSwLhE6Q==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=8" + } + }, + "node_modules/pathe": { + "version": "2.0.3", + "dev": true, + "license": "MIT" + }, + "node_modules/picocolors": { + "version": "1.1.1", + "dev": true, + "license": "ISC" + }, + "node_modules/picomatch": { + "version": "4.0.4", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=12" + }, + "funding": { + "url": "https://github.com/sponsors/jonschlinkert" + } + }, + "node_modules/pirates": { + "version": "4.0.7", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 6" + } + }, + "node_modules/pkg-types": { + "version": "1.3.1", + "dev": true, + "license": "MIT", + "dependencies": { + "confbox": "^0.1.8", + "mlly": "^1.7.4", + "pathe": "^2.0.1" + } + }, + "node_modules/postcss-load-config": { + "version": "6.0.1", + "dev": true, + "funding": [ + { + "type": "opencollective", + "url": "https://opencollective.com/postcss/" + }, + { + "type": "github", + "url": "https://github.com/sponsors/ai" + } + ], + "license": "MIT", + "dependencies": { + "lilconfig": "^3.1.1" + }, + "engines": { + "node": ">= 18" + }, + "peerDependencies": { + "jiti": ">=1.21.0", + "postcss": ">=8.0.9", + "tsx": "^4.8.1", + "yaml": "^2.4.2" + }, + "peerDependenciesMeta": { + "jiti": { + "optional": true + }, + "postcss": { + "optional": true + }, + "tsx": { + "optional": true + }, + "yaml": { + "optional": true + } + } + }, + "node_modules/pure-rand": { + "version": "8.4.2", + "resolved": "https://registry.npmjs.org/pure-rand/-/pure-rand-8.4.2.tgz", + "integrity": "sha512-vvuOGgcuPJAirlHvuQw1TrOiw7ptaIXXmIbNuiNOY6lNGJJH49PQ1Kj4nd783nPdQhQdicgOjVI2yI/9BD6/Ng==", + "dev": true, + "funding": [ + { + "type": "individual", + "url": "https://github.com/sponsors/dubzzz" + }, + { + "type": "opencollective", + "url": "https://opencollective.com/fast-check" + } + ], + "license": "MIT" + }, + "node_modules/readdirp": { + "version": "4.1.2", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 14.18.0" + }, + "funding": { + "type": "individual", + "url": "https://paulmillr.com/funding/" + } + }, + "node_modules/resolve-from": { + "version": "5.0.0", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=8" + } + }, + "node_modules/rollup": { + "version": "4.60.4", + "dev": true, + "license": "MIT", + "dependencies": { + "@types/estree": "1.0.8" + }, + "bin": { + "rollup": "dist/bin/rollup" + }, + "engines": { + "node": ">=18.0.0", + "npm": ">=8.0.0" + }, + "optionalDependencies": { + "@rollup/rollup-android-arm-eabi": "4.60.4", + "@rollup/rollup-android-arm64": "4.60.4", + "@rollup/rollup-darwin-arm64": "4.60.4", + "@rollup/rollup-darwin-x64": "4.60.4", + "@rollup/rollup-freebsd-arm64": "4.60.4", + "@rollup/rollup-freebsd-x64": "4.60.4", + "@rollup/rollup-linux-arm-gnueabihf": "4.60.4", + "@rollup/rollup-linux-arm-musleabihf": "4.60.4", + "@rollup/rollup-linux-arm64-gnu": "4.60.4", + "@rollup/rollup-linux-arm64-musl": "4.60.4", + "@rollup/rollup-linux-loong64-gnu": "4.60.4", + "@rollup/rollup-linux-loong64-musl": "4.60.4", + "@rollup/rollup-linux-ppc64-gnu": "4.60.4", + "@rollup/rollup-linux-ppc64-musl": "4.60.4", + "@rollup/rollup-linux-riscv64-gnu": "4.60.4", + "@rollup/rollup-linux-riscv64-musl": "4.60.4", + "@rollup/rollup-linux-s390x-gnu": "4.60.4", + "@rollup/rollup-linux-x64-gnu": "4.60.4", + "@rollup/rollup-linux-x64-musl": "4.60.4", + "@rollup/rollup-openbsd-x64": "4.60.4", + "@rollup/rollup-openharmony-arm64": "4.60.4", + "@rollup/rollup-win32-arm64-msvc": "4.60.4", + "@rollup/rollup-win32-ia32-msvc": "4.60.4", + "@rollup/rollup-win32-x64-gnu": "4.60.4", + "@rollup/rollup-win32-x64-msvc": "4.60.4", + "fsevents": "~2.3.2" + } + }, + "node_modules/shebang-command": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/shebang-command/-/shebang-command-2.0.0.tgz", + "integrity": "sha512-kHxr2zZpYtdmrN1qDjrrX/Z1rR1kG8Dx+gkpK1G4eXmvXswmcE1hTWBWYUzlraYw1/yZp6YuDY77YtvbN0dmDA==", + "dev": true, + "license": "MIT", + "dependencies": { + "shebang-regex": "^3.0.0" + }, + "engines": { + "node": ">=8" + } + }, + "node_modules/shebang-regex": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/shebang-regex/-/shebang-regex-3.0.0.tgz", + "integrity": "sha512-7++dFhtcx3353uBaq8DDR4NuxBetBzC7ZQOhmTQInHEd6bSrXdiEyzCvG07Z44UYdLShWUyXt5M/yhz8ekcb1A==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=8" + } + }, + "node_modules/source-map": { + "version": "0.7.6", + "dev": true, + "license": "BSD-3-Clause", + "engines": { + "node": ">= 12" + } + }, + "node_modules/sucrase": { + "version": "3.35.1", + "dev": true, + "license": "MIT", + "dependencies": { + "@jridgewell/gen-mapping": "^0.3.2", + "commander": "^4.0.0", + "lines-and-columns": "^1.1.6", + "mz": "^2.7.0", + "pirates": "^4.0.1", + "tinyglobby": "^0.2.11", + "ts-interface-checker": "^0.1.9" + }, + "bin": { + "sucrase": "bin/sucrase", + "sucrase-node": "bin/sucrase-node" + }, + "engines": { + "node": ">=16 || 14 >=14.17" + } + }, + "node_modules/thenify": { + "version": "3.3.1", + "dev": true, + "license": "MIT", + "dependencies": { + "any-promise": "^1.0.0" + } + }, + "node_modules/thenify-all": { + "version": "1.6.0", + "dev": true, + "license": "MIT", + "dependencies": { + "thenify": ">= 3.1.0 < 4" + }, + "engines": { + "node": ">=0.8" + } + }, + "node_modules/tinyexec": { + "version": "0.3.2", + "dev": true, + "license": "MIT" + }, + "node_modules/tinyglobby": { + "version": "0.2.16", + "dev": true, + "license": "MIT", + "dependencies": { + "fdir": "^6.5.0", + "picomatch": "^4.0.4" + }, + "engines": { + "node": ">=12.0.0" + }, + "funding": { + "url": "https://github.com/sponsors/SuperchupuDev" + } + }, + "node_modules/toml": { + "version": "4.3.0", + "resolved": "https://registry.npmjs.org/toml/-/toml-4.3.0.tgz", + "integrity": "sha512-lVb8X9BsPVuH0M4BKeS91tXAmJvCjQ5UIyAbQFaxkKGyUFK2RPkhwaFSQH8vbpl1d23eu/IBH+dwVMHWaq9A5A==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=20" + } + }, + "node_modules/tree-kill": { + "version": "1.2.2", + "dev": true, + "license": "MIT", + "bin": { + "tree-kill": "cli.js" + } + }, + "node_modules/ts-interface-checker": { + "version": "0.1.13", + "dev": true, + "license": "Apache-2.0" + }, + "node_modules/tsup": { + "version": "8.5.1", + "dev": true, + "license": "MIT", + "dependencies": { + "bundle-require": "^5.1.0", + "cac": "^6.7.14", + "chokidar": "^4.0.3", + "consola": "^3.4.0", + "debug": "^4.4.0", + "esbuild": "^0.27.0", + "fix-dts-default-cjs-exports": "^1.0.0", + "joycon": "^3.1.1", + "picocolors": "^1.1.1", + "postcss-load-config": "^6.0.1", + "resolve-from": "^5.0.0", + "rollup": "^4.34.8", + "source-map": "^0.7.6", + "sucrase": "^3.35.0", + "tinyexec": "^0.3.2", + "tinyglobby": "^0.2.11", + "tree-kill": "^1.2.2" + }, + "bin": { + "tsup": "dist/cli-default.js", + "tsup-node": "dist/cli-node.js" + }, + "engines": { + "node": ">=18" + }, + "peerDependencies": { + "@microsoft/api-extractor": "^7.36.0", + "@swc/core": "^1", + "postcss": "^8.4.12", + "typescript": ">=4.5.0" + }, + "peerDependenciesMeta": { + "@microsoft/api-extractor": { + "optional": true + }, + "@swc/core": { + "optional": true + }, + "postcss": { + "optional": true + }, + "typescript": { + "optional": true + } + } + }, + "node_modules/tsup/node_modules/@esbuild/aix-ppc64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/aix-ppc64/-/aix-ppc64-0.27.7.tgz", + "integrity": "sha512-EKX3Qwmhz1eMdEJokhALr0YiD0lhQNwDqkPYyPhiSwKrh7/4KRjQc04sZ8db+5DVVnZ1LmbNDI1uAMPEUBnQPg==", + "cpu": [ + "ppc64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "aix" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/android-arm": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/android-arm/-/android-arm-0.27.7.tgz", + "integrity": "sha512-jbPXvB4Yj2yBV7HUfE2KHe4GJX51QplCN1pGbYjvsyCZbQmies29EoJbkEc+vYuU5o45AfQn37vZlyXy4YJ8RQ==", + "cpu": [ + "arm" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/android-arm64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/android-arm64/-/android-arm64-0.27.7.tgz", + "integrity": "sha512-62dPZHpIXzvChfvfLJow3q5dDtiNMkwiRzPylSCfriLvZeq0a1bWChrGx/BbUbPwOrsWKMn8idSllklzBy+dgQ==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/android-x64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/android-x64/-/android-x64-0.27.7.tgz", + "integrity": "sha512-x5VpMODneVDb70PYV2VQOmIUUiBtY3D3mPBG8NxVk5CogneYhkR7MmM3yR/uMdITLrC1ml/NV1rj4bMJuy9MCg==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/darwin-arm64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/darwin-arm64/-/darwin-arm64-0.27.7.tgz", + "integrity": "sha512-5lckdqeuBPlKUwvoCXIgI2D9/ABmPq3Rdp7IfL70393YgaASt7tbju3Ac+ePVi3KDH6N2RqePfHnXkaDtY9fkw==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/darwin-x64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/darwin-x64/-/darwin-x64-0.27.7.tgz", + "integrity": "sha512-rYnXrKcXuT7Z+WL5K980jVFdvVKhCHhUwid+dDYQpH+qu+TefcomiMAJpIiC2EM3Rjtq0sO3StMV/+3w3MyyqQ==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/freebsd-arm64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/freebsd-arm64/-/freebsd-arm64-0.27.7.tgz", + "integrity": "sha512-B48PqeCsEgOtzME2GbNM2roU29AMTuOIN91dsMO30t+Ydis3z/3Ngoj5hhnsOSSwNzS+6JppqWsuhTp6E82l2w==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "freebsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/freebsd-x64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/freebsd-x64/-/freebsd-x64-0.27.7.tgz", + "integrity": "sha512-jOBDK5XEjA4m5IJK3bpAQF9/Lelu/Z9ZcdhTRLf4cajlB+8VEhFFRjWgfy3M1O4rO2GQ/b2dLwCUGpiF/eATNQ==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "freebsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/linux-arm": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/linux-arm/-/linux-arm-0.27.7.tgz", + "integrity": "sha512-RkT/YXYBTSULo3+af8Ib0ykH8u2MBh57o7q/DAs3lTJlyVQkgQvlrPTnjIzzRPQyavxtPtfg0EopvDyIt0j1rA==", + "cpu": [ + "arm" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/linux-arm64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/linux-arm64/-/linux-arm64-0.27.7.tgz", + "integrity": "sha512-RZPHBoxXuNnPQO9rvjh5jdkRmVizktkT7TCDkDmQ0W2SwHInKCAV95GRuvdSvA7w4VMwfCjUiPwDi0ZO6Nfe9A==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/linux-ia32": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/linux-ia32/-/linux-ia32-0.27.7.tgz", + "integrity": "sha512-GA48aKNkyQDbd3KtkplYWT102C5sn/EZTY4XROkxONgruHPU72l+gW+FfF8tf2cFjeHaRbWpOYa/uRBz/Xq1Pg==", + "cpu": [ + "ia32" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/linux-loong64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/linux-loong64/-/linux-loong64-0.27.7.tgz", + "integrity": "sha512-a4POruNM2oWsD4WKvBSEKGIiWQF8fZOAsycHOt6JBpZ+JN2n2JH9WAv56SOyu9X5IqAjqSIPTaJkqN8F7XOQ5Q==", + "cpu": [ + "loong64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/linux-mips64el": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/linux-mips64el/-/linux-mips64el-0.27.7.tgz", + "integrity": "sha512-KabT5I6StirGfIz0FMgl1I+R1H73Gp0ofL9A3nG3i/cYFJzKHhouBV5VWK1CSgKvVaG4q1RNpCTR2LuTVB3fIw==", + "cpu": [ + "mips64el" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/linux-ppc64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/linux-ppc64/-/linux-ppc64-0.27.7.tgz", + "integrity": "sha512-gRsL4x6wsGHGRqhtI+ifpN/vpOFTQtnbsupUF5R5YTAg+y/lKelYR1hXbnBdzDjGbMYjVJLJTd2OFmMewAgwlQ==", + "cpu": [ + "ppc64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/linux-riscv64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/linux-riscv64/-/linux-riscv64-0.27.7.tgz", + "integrity": "sha512-hL25LbxO1QOngGzu2U5xeXtxXcW+/GvMN3ejANqXkxZ/opySAZMrc+9LY/WyjAan41unrR3YrmtTsUpwT66InQ==", + "cpu": [ + "riscv64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/linux-s390x": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/linux-s390x/-/linux-s390x-0.27.7.tgz", + "integrity": "sha512-2k8go8Ycu1Kb46vEelhu1vqEP+UeRVj2zY1pSuPdgvbd5ykAw82Lrro28vXUrRmzEsUV0NzCf54yARIK8r0fdw==", + "cpu": [ + "s390x" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/linux-x64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/linux-x64/-/linux-x64-0.27.7.tgz", + "integrity": "sha512-hzznmADPt+OmsYzw1EE33ccA+HPdIqiCRq7cQeL1Jlq2gb1+OyWBkMCrYGBJ+sxVzve2ZJEVeePbLM2iEIZSxA==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/netbsd-arm64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/netbsd-arm64/-/netbsd-arm64-0.27.7.tgz", + "integrity": "sha512-b6pqtrQdigZBwZxAn1UpazEisvwaIDvdbMbmrly7cDTMFnw/+3lVxxCTGOrkPVnsYIosJJXAsILG9XcQS+Yu6w==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "netbsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/netbsd-x64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/netbsd-x64/-/netbsd-x64-0.27.7.tgz", + "integrity": "sha512-OfatkLojr6U+WN5EDYuoQhtM+1xco+/6FSzJJnuWiUw5eVcicbyK3dq5EeV/QHT1uy6GoDhGbFpprUiHUYggrw==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "netbsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/openbsd-arm64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/openbsd-arm64/-/openbsd-arm64-0.27.7.tgz", + "integrity": "sha512-AFuojMQTxAz75Fo8idVcqoQWEHIXFRbOc1TrVcFSgCZtQfSdc1RXgB3tjOn/krRHENUB4j00bfGjyl2mJrU37A==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "openbsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/openbsd-x64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/openbsd-x64/-/openbsd-x64-0.27.7.tgz", + "integrity": "sha512-+A1NJmfM8WNDv5CLVQYJ5PshuRm/4cI6WMZRg1by1GwPIQPCTs1GLEUHwiiQGT5zDdyLiRM/l1G0Pv54gvtKIg==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "openbsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/openharmony-arm64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/openharmony-arm64/-/openharmony-arm64-0.27.7.tgz", + "integrity": "sha512-+KrvYb/C8zA9CU/g0sR6w2RBw7IGc5J2BPnc3dYc5VJxHCSF1yNMxTV5LQ7GuKteQXZtspjFbiuW5/dOj7H4Yw==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "openharmony" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/sunos-x64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/sunos-x64/-/sunos-x64-0.27.7.tgz", + "integrity": "sha512-ikktIhFBzQNt/QDyOL580ti9+5mL/YZeUPKU2ivGtGjdTYoqz6jObj6nOMfhASpS4GU4Q/Clh1QtxWAvcYKamA==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "sunos" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/win32-arm64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/win32-arm64/-/win32-arm64-0.27.7.tgz", + "integrity": "sha512-7yRhbHvPqSpRUV7Q20VuDwbjW5kIMwTHpptuUzV+AA46kiPze5Z7qgt6CLCK3pWFrHeNfDd1VKgyP4O+ng17CA==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/win32-ia32": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/win32-ia32/-/win32-ia32-0.27.7.tgz", + "integrity": "sha512-SmwKXe6VHIyZYbBLJrhOoCJRB/Z1tckzmgTLfFYOfpMAx63BJEaL9ExI8x7v0oAO3Zh6D/Oi1gVxEYr5oUCFhw==", + "cpu": [ + "ia32" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/@esbuild/win32-x64": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/@esbuild/win32-x64/-/win32-x64-0.27.7.tgz", + "integrity": "sha512-56hiAJPhwQ1R4i+21FVF7V8kSD5zZTdHcVuRFMW0hn753vVfQN8xlx4uOPT4xoGH0Z/oVATuR82AiqSTDIpaHg==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/tsup/node_modules/esbuild": { + "version": "0.27.7", + "resolved": "https://registry.npmjs.org/esbuild/-/esbuild-0.27.7.tgz", + "integrity": "sha512-IxpibTjyVnmrIQo5aqNpCgoACA/dTKLTlhMHihVHhdkxKyPO1uBBthumT0rdHmcsk9uMonIWS0m4FljWzILh3w==", + "dev": true, + "hasInstallScript": true, + "license": "MIT", + "bin": { + "esbuild": "bin/esbuild" + }, + "engines": { + "node": ">=18" + }, + "optionalDependencies": { + "@esbuild/aix-ppc64": "0.27.7", + "@esbuild/android-arm": "0.27.7", + "@esbuild/android-arm64": "0.27.7", + "@esbuild/android-x64": "0.27.7", + "@esbuild/darwin-arm64": "0.27.7", + "@esbuild/darwin-x64": "0.27.7", + "@esbuild/freebsd-arm64": "0.27.7", + "@esbuild/freebsd-x64": "0.27.7", + "@esbuild/linux-arm": "0.27.7", + "@esbuild/linux-arm64": "0.27.7", + "@esbuild/linux-ia32": "0.27.7", + "@esbuild/linux-loong64": "0.27.7", + "@esbuild/linux-mips64el": "0.27.7", + "@esbuild/linux-ppc64": "0.27.7", + "@esbuild/linux-riscv64": "0.27.7", + "@esbuild/linux-s390x": "0.27.7", + "@esbuild/linux-x64": "0.27.7", + "@esbuild/netbsd-arm64": "0.27.7", + "@esbuild/netbsd-x64": "0.27.7", + "@esbuild/openbsd-arm64": "0.27.7", + "@esbuild/openbsd-x64": "0.27.7", + "@esbuild/openharmony-arm64": "0.27.7", + "@esbuild/sunos-x64": "0.27.7", + "@esbuild/win32-arm64": "0.27.7", + "@esbuild/win32-ia32": "0.27.7", + "@esbuild/win32-x64": "0.27.7" + } + }, + "node_modules/tsx": { + "version": "4.22.3", + "dev": true, + "license": "MIT", + "dependencies": { + "esbuild": "~0.28.0" + }, + "bin": { + "tsx": "dist/cli.mjs" + }, + "engines": { + "node": ">=18.0.0" + }, + "optionalDependencies": { + "fsevents": "~2.3.3" + } + }, + "node_modules/typescript": { + "version": "5.9.3", + "dev": true, + "license": "Apache-2.0", + "bin": { + "tsc": "bin/tsc", + "tsserver": "bin/tsserver" + }, + "engines": { + "node": ">=14.17" + } + }, + "node_modules/ufo": { + "version": "1.6.4", + "dev": true, + "license": "MIT" + }, + "node_modules/undici-types": { + "version": "6.21.0", + "dev": true, + "license": "MIT" + }, + "node_modules/uuid": { + "version": "14.0.2", + "resolved": "https://registry.npmjs.org/uuid/-/uuid-14.0.2.tgz", + "integrity": "sha512-xZe/16rV4aa+HGSOCiY2YeLT1OybRLrrkL/Rqaq7p7GMVXjFh+6wN4oMYgjFmnSnhY8t6Xpdl2l9qmnHYuMHwQ==", + "dev": true, + "funding": [ + "https://github.com/sponsors/broofa", + "https://github.com/sponsors/ctavan" + ], + "license": "MIT", + "bin": { + "uuid": "dist-node/bin/uuid" + } + }, + "node_modules/which": { + "version": "2.0.2", + "resolved": "https://registry.npmjs.org/which/-/which-2.0.2.tgz", + "integrity": "sha512-BLI3Tl1TW3Pvl70l3yq3Y64i+awpwXqsGBYWkkqMtnbXgrMD+yj7rhW0kuEDxzJaYXGjEW5ogapKNMEKNMjibA==", + "dev": true, + "license": "ISC", + "dependencies": { + "isexe": "^2.0.0" + }, + "bin": { + "node-which": "bin/node-which" + }, + "engines": { + "node": ">= 8" + } + }, + "node_modules/yaml": { + "version": "2.9.0", + "dev": true, + "license": "ISC", + "bin": { + "yaml": "bin.mjs" + }, + "engines": { + "node": ">= 14.6" + }, + "funding": { + "url": "https://github.com/sponsors/eemeli" + } + }, + "node_modules/zod": { + "version": "4.4.3", + "license": "MIT", + "funding": { + "url": "https://github.com/sponsors/colinhacks" + } + } + } +} diff --git a/@omniroute/opencode-plugin-v2/package.json b/@omniroute/opencode-plugin-v2/package.json new file mode 100644 index 0000000000..da8bc134d6 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/package.json @@ -0,0 +1,67 @@ +{ + "name": "@omniroute/opencode-plugin-v2", + "version": "0.1.0", + "description": "OmniRoute OpenCode plugin (v2 Promise API): catalog transform with models, combos, enrichment, and naming.", + "type": "module", + "main": "./dist/index.js", + "types": "./dist/index.d.ts", + "exports": { + ".": { + "types": "./dist/index.d.ts", + "import": "./dist/index.js" + } + }, + "files": [ + "dist", + "README.md", + "LICENSE" + ], + "scripts": { + "build": "tsup", + "clean": "rm -rf dist", + "test": "node --import tsx/esm --test tests/*.test.ts", + "prepublishOnly": "npm run clean && npm run build && npm test" + }, + "dependencies": { + "zod": "^4.4.3" + }, + "devDependencies": { + "@opencode-ai/plugin": "1.18.29", + "@types/node": "^22.19.19", + "tsup": "^8.5.1", + "tsx": "^4.22.3", + "typescript": "^5.9.3" + }, + "engines": { + "node": ">=22.22.3" + }, + "license": "MIT", + "author": "OmniRoute contributors", + "repository": { + "type": "git", + "url": "https://github.com/diegosouzapw/OmniRoute.git", + "directory": "@omniroute/opencode-plugin-v2" + }, + "homepage": "https://github.com/diegosouzapw/OmniRoute/tree/main/%40omniroute/opencode-plugin-v2#readme", + "bugs": { + "url": "https://github.com/diegosouzapw/OmniRoute/issues" + }, + "keywords": [ + "omniroute", + "opencode", + "opencode-plugin", + "opencode-v2", + "ai-sdk", + "openai-compatible", + "provider", + "catalog", + "combos", + "gemini" + ], + "publishConfig": { + "access": "public" + }, + "peerDependencies": { + "@opencode-ai/plugin": ">=1.18.29 <2" + } +} diff --git a/@omniroute/opencode-plugin-v2/src/cache.ts b/@omniroute/opencode-plugin-v2/src/cache.ts new file mode 100644 index 0000000000..58aca429e4 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/cache.ts @@ -0,0 +1,211 @@ +import { createHash } from "node:crypto"; +import { homedir } from "node:os"; +import { mkdir, readFile, unlink, writeFile } from "node:fs/promises"; +import { dirname, join } from "node:path"; +import type { + OmniRouteEnrichmentEntry, + OmniRouteEnrichmentMap, + OmniRouteProviderConnection, + OmniRouteRawAutoCombo, + OmniRouteRawCombo, + OmniRouteRawModelEntry, +} from "./shared/index.js"; + +export const DEFAULT_MODEL_CACHE_TTL_MS = 300_000 as const; + +/** + * Breather after a refresh whose models fetch came back empty (gateway down + * or refusing). Transforms inside the window serve last-known-good without + * re-firing the fetch suite. Short on purpose: it only guards the + * pathological case, normal TTL expiry still refetches every window. + */ +export const UNREACHABLE_COOLDOWN_MS = 15_000 as const; + +export interface CatalogSnapshot { + models: OmniRouteRawModelEntry[]; + combos: OmniRouteRawCombo[]; + autoCombos: OmniRouteRawAutoCombo[]; + providers?: OmniRouteProviderConnection[]; + enrichment?: OmniRouteEnrichmentMap; + fetchedAt: number; +} + +export const SNAPSHOT_FORMAT_VERSION = 2 as const; + +/** + * A raw snapshot entry is stale when it cannot be mapped to a publishable + * model: no string `id` (unroutable) or a pre-mapped `api` block without a + * valid `npm` package (the runner would reject it as `Unsupported package`). + * Plain `/v1/models` entries carry no `api` block -- it is synthesized at + * publish time -- so only a present-but-invalid block drops the entry. + */ +export function isStaleSnapshotModel(entry: unknown): boolean { + if (!entry || typeof entry !== "object") return true; + const id = (entry as { id?: unknown }).id; + if (typeof id !== "string" || id.length === 0) return true; + const api = (entry as { api?: unknown }).api; + if (api === undefined) return false; + if (!api || typeof api !== "object") return true; + const npm = (api as { npm?: unknown }).npm; + return typeof npm !== "string" || npm.length === 0; +} + +interface DiskSnapshotV2 { + v: 2; + identityFingerprint: string; + models: OmniRouteRawModelEntry[]; + combos: OmniRouteRawCombo[]; + autoCombos?: OmniRouteRawAutoCombo[]; + providers?: OmniRouteProviderConnection[]; + /** + * Display names, provider labels, pricing and free-tier budgets, as + * `[key, entry]` pairs (a Map does not survive JSON). Persisted because a + * cold start otherwise publishes raw model ids until the first refresh + * completes — which is the moment the snapshot exists to cover. + */ + enrichment?: [string, OmniRouteEnrichmentEntry][]; + writtenAt: number; +} + +/** + * Ceiling on what one snapshot may occupy on disk. A gateway with thousands of + * models makes this file grow without bound otherwise; past the cap the + * enrichment overlay is dropped first (it is rebuilt on the next refresh) + * rather than losing the catalog itself. + */ +const MAX_SNAPSHOT_BYTES = 32 * 1024 * 1024; + +function trimTrailingSlashes(value: string): string { + let i = value.length; + while (i > 0 && value.charCodeAt(i - 1) === 0x2f) i -= 1; + return i === value.length ? value : value.slice(0, i); +} + +function normalizeBaseURL(baseURL: string): string { + try { + const parsed = new URL(baseURL); + parsed.hash = ""; + parsed.pathname = trimTrailingSlashes(parsed.pathname) || "/"; + return parsed.toString(); + } catch { + return trimTrailingSlashes(baseURL); + } +} + +export function memoryCacheKey(baseURL: string, credentialId: string): string { + return `${baseURL}::${createHash("sha256").update(credentialId).digest("hex")}`; +} + +export function snapshotIdentityFingerprint( + baseURL: string, + apiKey: string, + managementReadToken: string +): string { + return createHash("sha256") + .update(JSON.stringify([normalizeBaseURL(baseURL), apiKey, managementReadToken])) + .digest("hex"); +} + +export function diskSnapshotPath(providerId: string): string { + // OPENCODE_DATA_DIR is honoured verbatim when set: whoever controls the + // process environment already chooses where the process writes, so + // resolving it further would only surprise. The providerId segment stays + // bounded by the options schema (letters, digits, '.', '_' and '-'; never + // "." or ".."), keeping the file inside /plugins/. + const dir = process.env.OPENCODE_DATA_DIR ?? join(homedir(), ".local", "share", "opencode"); + return join(dir, "plugins", `omniroute-${providerId}.json`); +} + +export async function readDiskSnapshot( + providerId: string, + identityFingerprint: string, + logger?: { warn: (message: string) => void } +): Promise { + try { + const body = await readFile(diskSnapshotPath(providerId), "utf8"); + const parsed = JSON.parse(body) as Partial; + if ( + !parsed || + typeof parsed.v !== "number" || + parsed.v < SNAPSHOT_FORMAT_VERSION || + typeof parsed.identityFingerprint !== "string" || + parsed.identityFingerprint !== identityFingerprint + ) { + return undefined; + } + if ( + !Array.isArray(parsed.models) || + parsed.models.length === 0 || + !Array.isArray(parsed.combos) + ) { + return undefined; + } + const stale = (parsed.models as unknown[]).filter(isStaleSnapshotModel).length; + const models = (parsed.models as OmniRouteRawModelEntry[]).filter( + (entry) => !isStaleSnapshotModel(entry) + ); + if (stale > 0) { + logger?.warn(`[omniroute-v2] dropping ${stale} stale snapshot entries without api block`); + } + if (models.length === 0) return undefined; + return { + models, + combos: parsed.combos as OmniRouteRawCombo[], + autoCombos: Array.isArray(parsed.autoCombos) + ? (parsed.autoCombos as OmniRouteRawAutoCombo[]) + : [], + providers: Array.isArray(parsed.providers) + ? (parsed.providers as OmniRouteProviderConnection[]) + : [], + // A snapshot written before this field existed, or one whose overlay was + // dropped for size, simply starts unenriched and recovers on the first + // refresh — the same state as before it was persisted at all. + enrichment: Array.isArray(parsed.enrichment) + ? new Map(parsed.enrichment as [string, OmniRouteEnrichmentEntry][]) + : undefined, + fetchedAt: typeof parsed.writtenAt === "number" ? parsed.writtenAt : Date.now(), + }; + } catch { + return undefined; + } +} + +export async function writeDiskSnapshot( + providerId: string, + snapshot: CatalogSnapshot, + identityFingerprint: string +): Promise { + try { + if (snapshot.models.length === 0) return; + const file = diskSnapshotPath(providerId); + await mkdir(dirname(file), { recursive: true, mode: 0o700 }); + const envelope: DiskSnapshotV2 = { + v: 2, + identityFingerprint, + models: snapshot.models, + combos: snapshot.combos, + autoCombos: snapshot.autoCombos, + providers: snapshot.providers ?? [], + enrichment: snapshot.enrichment ? [...snapshot.enrichment.entries()] : undefined, + writtenAt: Date.now(), + }; + let payload = JSON.stringify(envelope); + if (payload.length > MAX_SNAPSHOT_BYTES && envelope.enrichment !== undefined) { + delete envelope.enrichment; + payload = JSON.stringify(envelope); + } + if (payload.length > MAX_SNAPSHOT_BYTES) return; + await writeFile(file, payload, { encoding: "utf8", mode: 0o600 }); + } catch { + // Best-effort: callers already hold the in-memory entry. + } +} + +export async function clearDiskSnapshot(providerId: string): Promise { + try { + await unlink(diskSnapshotPath(providerId)); + return true; + } catch { + return false; + } +} diff --git a/@omniroute/opencode-plugin-v2/src/catalog.ts b/@omniroute/opencode-plugin-v2/src/catalog.ts new file mode 100644 index 0000000000..73c3f4ab70 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/catalog.ts @@ -0,0 +1,798 @@ +import type { CatalogDraft } from "@opencode-ai/plugin/v2/promise"; +import { type HostContract, detectHostContract, emitsLegacyFields } from "./compat.js"; +import type { Model as LegacyModelV2 } from "@opencode-ai/sdk/v2"; +import type { ModelV2Info, ProviderV2Info } from "@opencode-ai/sdk/v2/types"; +import { + type ApiFormatV2, + type LogLevel, + type Logger, + type OmniRouteAutoCombosFetcher, + type OmniRouteCombosFetcher, + type OmniRouteEnrichmentFetcher, + type OmniRouteEnrichmentMap, + type OmniRouteModelsFetcher, + type OmniRouteProviderConnection, + type OmniRouteProvidersFetcher, + type OmniRouteRawAutoCombo, + type OmniRouteRawCombo, + type OmniRouteRawModelEntry, + applyEnrichment, + buildCanonicalToAliasMap, + canonicalDedupSet, + createLogger, + defaultOmniRouteEnrichmentFetcher, + defaultOmniRouteProvidersFetcher, + ensureV1Suffix, + isUsableCombo, + isUsableRawModelId, + lookupEnrichment, + mapAutoComboToModelV2, + mapComboToModelV2, + mapRawModelToModelV2, + usableProviderAliasSet, +} from "./shared/index.js"; + +export type ModelsFetcher = OmniRouteModelsFetcher; +export type CombosFetcher = OmniRouteCombosFetcher; +export type AutoCombosFetcher = OmniRouteAutoCombosFetcher; +export type ProvidersFetcher = OmniRouteProvidersFetcher; +export type EnrichmentFetcher = OmniRouteEnrichmentFetcher; + +export interface EndpointTimeouts { + models?: number; + combos?: number; + autoCombos?: number; + enrichment?: number; +} + +export interface ResolvedOptions { + providerId: string; + baseURL: string; + apiKey: string; + managementReadToken?: string; + timeoutMs: number; + timeouts?: EndpointTimeouts; + logger?: Logger; + logLevel?: LogLevel; + startupDebug?: boolean; + modelCacheTtlMs: number; + /** v1 parity: prefix the display name with the upstream provider label. */ + providerTag?: boolean; + displayName?: string; + apiFormat?: ApiFormatV2; + visibleModels?: string[]; + hiddenModels?: string[]; + usableOnly: boolean; + enrichment?: OmniRouteEnrichmentMap | boolean; + /** + * Shared collision-warning dedupe set keyed `cacheKey::comboKey`. When + * omitted a fresh per-publish set is used. index.ts passes one setup-wide + * set so a repeated publish (stale replay + refresh) warns once per key. + */ + collisionWarned?: Set; +} + +export interface CatalogFetchers { + fetcher?: ModelsFetcher; + combosFetcher?: CombosFetcher; + autoCombosFetcher?: AutoCombosFetcher; + providersFetcher?: ProvidersFetcher; + enrichmentFetcher?: EnrichmentFetcher; + models?: ModelsFetcher; + combos?: CombosFetcher; + autoCombos?: AutoCombosFetcher; + providers?: ProvidersFetcher; + enrichment?: EnrichmentFetcher; + /** + * Called when a gateway source cannot be read. Without it this function + * degrades silently — the catalog publishes with raw ids and no combos and + * nothing says why, which is the failure the plugin path reports. + */ + onSourceError?: (endpoint: string, reason: string) => void; +} + +// The shared mappers speak the legacy (`Provider.models[id]`) `Model` shape +// (imported from `@opencode-ai/sdk/v2`, also re-exported by the plugin root +// as `ModelV2`); the real v2 `CatalogDraft` carries `ModelV2Info` instead. +// Convert the fields 1:1 at the draft boundary -- NEVER `as unknown as` the +// whole model. +// +// Binary-compat note: the prod binary (beta-17823) reads a top-level +// `package` field on both Model and Provider structs (`package:a.Package`, +// gated by `isAISDK = startsWith("aisdk:")`), with a model-to-provider +// fallback (`package: u.package ?? s.package`). The pinned SDK types +// (1.18.29) only know the `api` block, so the binary field is published via +// the typed extensions below (spread/Object.assign, never `any`). +export const BINARY_AISDK_PREFIX = "aisdk:"; + +/** Top-level `package` as the legacy contract expects it (`aisdk:`). */ +export interface BinaryCompatPackage { + package: string; +} + +/** + * The legacy contract keeps on the model/provider itself what the `api` block + * carries in the pinned types: the aisdk package, the endpoint (as + * `settings.baseURL`) and the per-request headers. None of these keys collide + * with a key of `ModelV2Info`/`ProviderV2Info`, so both field sets can be + * published on the same object. + */ +export interface BinaryCompatFields extends BinaryCompatPackage { + settings: Record; + headers: Record; +} + +/** Legacy variants read their options from `settings`, not `headers`/`body`. */ +export type BinaryCompatVariant = ModelV2Info["variants"][number] & { + settings: Record; +}; + +export type BinaryCompatModel = ModelV2Info & BinaryCompatFields; +export type BinaryCompatProvider = ProviderV2Info & + BinaryCompatPackage & { + settings: Record; + }; + +export function toBinaryPackage(npm: string): string { + return npm.startsWith(BINARY_AISDK_PREFIX) ? npm : `${BINARY_AISDK_PREFIX}${npm}`; +} +export function legacyApiToInfoApi(api: LegacyModelV2["api"]): ModelV2Info["api"] { + if (!api || typeof api.npm !== "string" || api.npm.length === 0) { + throw new Error( + "[omniroute-v2] refusing to publish a model without an api block (missing api.npm)" + ); + } + return { id: api.id, type: "aisdk", package: api.npm, url: api.url }; +} + +function legacyCostToInfoCost(cost: LegacyModelV2["cost"]): ModelV2Info["cost"] { + return [{ input: cost.input, output: cost.output, cache: cost.cache }]; +} + +function legacyCapabilitiesToInfoCapabilities( + caps: LegacyModelV2["capabilities"] +): ModelV2Info["capabilities"] { + const input: string[] = []; + if (caps.input.text) input.push("text"); + if (caps.input.audio) input.push("audio"); + if (caps.input.image) input.push("image"); + if (caps.input.video) input.push("video"); + if (caps.input.pdf) input.push("pdf"); + const output: string[] = []; + if (caps.output.text) output.push("text"); + if (caps.output.audio) output.push("audio"); + if (caps.output.image) output.push("image"); + if (caps.output.video) output.push("video"); + if (caps.output.pdf) output.push("pdf"); + return { tools: caps.toolcall, input, output }; +} + +function legacyToInfo(providerID: string, modelID: string, m: LegacyModelV2): ModelV2Info { + const variants = Object.entries(m.variants ?? {}).map(([id, body]) => ({ + id, + headers: {}, + body: body as Record, + })); + const parsed = Date.parse(m.release_date); + return { + id: modelID, + providerID, + ...(m.family !== undefined ? { family: m.family } : {}), + name: m.name, + api: legacyApiToInfoApi(m.api), + capabilities: legacyCapabilitiesToInfoCapabilities(m.capabilities), + request: { headers: { ...m.headers }, body: { ...m.options } }, + variants, + time: { released: Number.isNaN(parsed) ? 0 : parsed }, + cost: legacyCostToInfoCost(m.cost), + status: m.status, + enabled: true, + limit: { ...m.limit }, + }; +} + +export interface PublishCounts { + models: number; + combos: number; + autoCombos: number; +} + +export interface ModelListFilter { + exact: Set; + suffixes: Set; +} + +export function compileModelListFilter(list?: string[]): ModelListFilter | undefined { + if (!list || list.length === 0) return undefined; + const exact = new Set(); + const suffixes = new Set(); + for (const id of list) { + if (id.includes("/")) { + exact.add(id); + } else { + suffixes.add(id); + } + } + if (exact.size === 0 && suffixes.size === 0) return undefined; + return { exact, suffixes }; +} + +function matchesSuffix(id: string, suffixes: Set): boolean { + if (suffixes.size === 0) return false; + const slash = id.indexOf("/"); + const suffix = slash > 0 ? id.slice(slash + 1) : id; + return suffixes.has(suffix); +} + +export function passesModelAllowlist( + id: string, + visible?: ModelListFilter, + hidden?: ModelListFilter +): boolean { + if (hidden) { + if (hidden.exact.has(id) || matchesSuffix(id, hidden.suffixes)) return false; + } + if (visible) { + if (!visible.exact.has(id) && !matchesSuffix(id, visible.suffixes)) return false; + } + return true; +} + +export function passesComboAllowlist(combo: OmniRouteRawCombo, visible?: ModelListFilter): boolean { + if (!visible) return true; + const steps = Array.isArray(combo.models) ? combo.models : []; + if (steps.length === 0) return true; + let sawResolvableMember = false; + for (const step of steps) { + if (step?.kind === "combo-ref") continue; + const modelId = typeof step?.model === "string" ? step.model : ""; + if (modelId.length === 0) continue; + sawResolvableMember = true; + if (visible.exact.has(modelId) || matchesSuffix(modelId, visible.suffixes)) return true; + } + if (!sawResolvableMember) return true; + return false; +} + +/** + * Project the `api` block onto the legacy top-level fields. Only the `aisdk` + * variant of `ModelApi`/`ProviderApi` carries a package, so the caller narrows + * before calling; a `native` api has no legacy equivalent and publishes + * nothing (the legacy contract has no native models). + */ +function legacyModelFields(info: ModelV2Info): BinaryCompatFields | undefined { + if (info.api.type !== "aisdk") return undefined; + const settings: Record = { + ...(info.api.settings ?? {}), + ...info.request.body, + }; + if (info.api.url !== undefined) settings.baseURL = info.api.url; + return { + package: toBinaryPackage(info.api.package), + settings, + headers: { ...info.request.headers }, + }; +} + +/** `{id, headers, body}` (pinned types) plus `{settings}` (legacy contract). */ +function legacyVariants(variants: ModelV2Info["variants"]): BinaryCompatVariant[] { + return variants.map((variant) => ({ ...variant, settings: { ...variant.body } })); +} + +function assignModelFields( + target: ModelV2Info, + source: LegacyModelV2, + contract: HostContract +): void { + const info = legacyToInfo(target.providerID || source.providerID, target.id || source.id, source); + target.name = info.name; + target.api = info.api; + target.capabilities = info.capabilities; + target.request = info.request; + target.variants = info.variants; + target.time = info.time; + target.cost = info.cost; + target.status = info.status; + target.enabled = info.enabled; + target.limit = info.limit; + if (info.family !== undefined) { + target.family = info.family; + } + if (!emitsLegacyFields(contract)) return; + const legacy = legacyModelFields(info); + if (legacy !== undefined) { + Object.assign(target, legacy); + target.variants = legacyVariants(info.variants); + } +} + +function assignProviderFields( + target: ProviderV2Info, + source: { name: string; api: ProviderV2Info["api"]; integrationID: string }, + contract: HostContract +): void { + target.name = source.name; + target.api = source.api; + target.integrationID = source.integrationID; + if (!emitsLegacyFields(contract)) return; + // The legacy contract defaults `Provider.Info.package` to `""` and model + // resolution falls back to it (`package: model.package ?? provider.package`), + // so the provider carries the same `aisdk:` value as its models, and + // the endpoint as `settings.baseURL`. + if (source.api.type !== "aisdk") return; + const settings: Record = { ...(source.api.settings ?? {}) }; + if (source.api.url !== undefined) settings.baseURL = source.api.url; + Object.assign(target, { package: toBinaryPackage(source.api.package), settings }); +} + +/** A widened capability flag (`boolean | { field }`) read back as a plain flag. */ +function isCapabilityEnabled(value: boolean | { field: string }): boolean { + return value !== false; +} + +/** + * Combo steps reach us from the gateway with a shape the SDK types do not + * describe (`kind`, `comboName`, `model` appear per step kind). One reader + * keeps that single untyped boundary in one place instead of scattering casts. + */ +function readStepField(step: unknown, key: "kind" | "comboName" | "model"): unknown { + return (step as Record | null | undefined)?.[key]; +} + +/** + * Resolve the display-name + pricing overlay. A caller may hand over a + * ready-made map (tests, pre-resolved overlays) or turn the fetch off; a + * failed fetch soft-fails to an empty map so the catalog still publishes, + * with mapper-default names and zeroed pricing rather than nothing at all. + */ +async function resolveEnrichmentOverlay( + opts: ResolvedOptions, + fetchers: CatalogFetchers | undefined, + log: Logger +): Promise { + if (opts.enrichment instanceof Map) return opts.enrichment; + if (opts.enrichment === false) return new Map(); + const fetchEnrichment = + fetchers?.enrichmentFetcher ?? fetchers?.enrichment ?? defaultOmniRouteEnrichmentFetcher; + try { + return await fetchEnrichment( + opts.baseURL, + opts.managementReadToken ?? opts.apiKey, + opts.timeouts?.enrichment ?? opts.timeoutMs, + fetchers?.onSourceError + ); + } catch (err) { + log.warn( + `[omniroute-v2] enrichment fetch failed, continuing without names/pricing: ${err instanceof Error ? err.message : String(err)}` + ); + return new Map(); + } +} + +/** + * Resolve the provider aliases worth publishing when `usableOnly` is on. + * Gated on the flag, so the default configuration issues no request at all. + * The filter subtracts: a failed or empty connections fetch yields + * `undefined` and keeps the whole catalog, because only a prefix proven not + * provisioned may be dropped. + */ +async function resolveUsableAliases( + opts: ResolvedOptions, + providersFetcher: OmniRouteProvidersFetcher | undefined, + onSourceError: ((endpoint: string, reason: string) => void) | undefined, + enrichment: OmniRouteEnrichmentMap, + timeoutMs: number, + log: Logger +): Promise | undefined> { + if (!opts.usableOnly) return undefined; + let rawConnections: OmniRouteProviderConnection[]; + try { + const fetchProviders = providersFetcher ?? defaultOmniRouteProvidersFetcher; + rawConnections = await fetchProviders( + opts.baseURL, + opts.managementReadToken ?? opts.apiKey, + timeoutMs, + onSourceError + ); + } catch (err) { + log.warn( + `[omniroute-v2] providers fetch failed, usableOnly filter disabled for this refresh: ${err instanceof Error ? err.message : String(err)}` + ); + rawConnections = []; + } + return rawConnections.length > 0 ? usableProviderAliasSet(rawConnections, enrichment) : undefined; +} + +/** Everything the combo publishing pass reads, passed as one value. */ +interface PublishContext { + draft: CatalogDraft; + opts: ResolvedOptions; + log: Logger; + providerId: string; + hostContract: HostContract; + enrichment: OmniRouteEnrichmentMap; + rawModelById: Map; + publishedKeys: Set; + publishedModelIds: Map; + visibleFilter: ReturnType; + hiddenFilter: ReturnType; + usable: ReturnType | undefined; + canonicalToAlias: ReturnType; + combosFetcher: CatalogFetchers["combos"] | undefined; + combosTimeout: number; + /** Shared with the auto-combos pass: one collision warning per key, per run. */ + warnedCombos: Set; + cacheKey: string; +} + +/** + * Fetch the gateway's combos and publish them, resolving nested combo-refs to + * a fixpoint first: a combo whose members are themselves combos only knows its + * lowest common denominator once those are known. Combos that never resolve + * are dropped rather than published with a fabricated capability set, and + * reported once. + * + * Returns the number published, or `undefined` when the combos fetch failed — + * the caller then publishes a models-only catalog instead of an empty one. + */ +async function publishCombos(ctx: PublishContext): Promise { + const { + draft, + opts, + log, + providerId: X, + hostContract, + enrichment, + rawModelById, + publishedKeys, + publishedModelIds, + visibleFilter, + hiddenFilter, + usable, + canonicalToAlias, + combosFetcher, + combosTimeout, + warnedCombos, + cacheKey, + } = ctx; + let rawCombos: OmniRouteRawCombo[]; + try { + rawCombos = combosFetcher + ? await combosFetcher(opts.baseURL, opts.managementReadToken ?? opts.apiKey, combosTimeout) + : []; + } catch (err) { + log.warn( + `[omniroute-v2] combos fetch failed, falling back to models-only catalog: ${err instanceof Error ? err.message : String(err)}` + ); + return undefined; + } + + let comboCount = 0; + // Ported from v1 (fixpoint 8 passes + warn once per (cacheKey, comboKey) + // + intentional-dedup exception). Nested combo-refs resolve against the + // friendly combo name; unresolvable combos are dropped (never published + // with a fabricated empty LCD) and reported once. + const MAX_COMBO_PASSES = 8; + const pending = rawCombos.filter((combo) => { + if (!combo || !combo.id) return false; + if (combo.isHidden === true) return false; + if (usable && !isUsableCombo(combo, usable)) return false; + if (visibleFilter && !passesComboAllowlist(combo, visibleFilter)) return false; + // Deny wins for combos too: a user who hides an id expects it gone from + // the picker whether it is a model or a combo built on it. + if (hiddenFilter && passesComboAllowlist(combo, hiddenFilter)) return false; + return true; + }); + const resolvedByName = new Map(); + let unresolved: typeof pending = []; + + for (let pass = 0; pass < MAX_COMBO_PASSES && pending.length > 0; pass++) { + const stillPending: typeof pending = []; + for (const combo of pending) { + const memberSteps = Array.isArray(combo.models) ? combo.models : []; + const memberEntries: OmniRouteRawModelEntry[] = []; + let deferred = false; + for (const step of memberSteps) { + const kind = readStepField(step, "kind"); + if (kind === "combo-ref") { + const comboName = readStepField(step, "comboName"); + if (typeof comboName !== "string" || comboName.length === 0) continue; + const nested = resolvedByName.get(comboName); + if (!nested) { + deferred = true; + break; + } + memberEntries.push(synthesizeNestedMember(comboName, nested)); + continue; + } + const modelId = readStepField(step, "model"); + if (typeof modelId !== "string" || modelId.length === 0) continue; + const member = rawModelById.get(modelId); + if (member) memberEntries.push(member); + } + if (deferred) { + stillPending.push(combo); + continue; + } + const mapped = mapComboToModelV2(combo, memberEntries, X, opts.baseURL, opts.apiFormat); + applyEnrichment(mapped, lookupEnrichment(combo.id, enrichment, canonicalToAlias), { + isCombo: true, + }); + const mid = mapped.id.startsWith(X + "/") ? mapped.id.slice(X.length + 1) : mapped.id; + const key = X + "/" + mid; + if (publishedKeys.has(key)) { + // Intentional dedup (v1 parity): `/v1/models` pre-mirrors combos as + // raw entries, so the combo's friendly NAME matches the overwritten + // entry's model id (bare or provider-prefixed, endsWith to cover + // both). Only warn on a genuine accidental collision (name differs + // from the entry it overwrites). + const existingId = publishedModelIds.get(key) ?? ""; + const friendly = + typeof combo.name === "string" && combo.name.trim().length > 0 + ? combo.name.trim() + : combo.id; + const isIntentionalDedup = + existingId === friendly || + existingId === X + "/" + friendly || + existingId.endsWith("/" + friendly); + if (!isIntentionalDedup) { + const dedupeKey = `${cacheKey}::${key}`; + if (!warnedCombos.has(dedupeKey)) { + warnedCombos.add(dedupeKey); + log.warn(`[omniroute-v2] combo key "${key}" collides with a model id; combo wins.`); + } + } + } + draft.model.update(X, mid, (m) => { + assignModelFields(m, mapped, hostContract); + }); + publishedKeys.add(key); + publishedModelIds.set(key, mapped.id); + comboCount += 1; + const lookupName = + typeof combo.name === "string" && combo.name.trim().length > 0 + ? combo.name.trim() + : combo.id; + if (!resolvedByName.has(lookupName)) resolvedByName.set(lookupName, mapped); + } + if (stillPending.length === pending.length) { + unresolved = stillPending; + break; + } + unresolved = stillPending; + pending.length = 0; + pending.push(...stillPending); + } + + if (unresolved.length > 0) { + log.warn( + `[omniroute-v2] ${unresolved.length} combo(s) could not resolve all nested combo-refs after ${MAX_COMBO_PASSES} passes; dropped to avoid over-claiming.` + ); + } + return comboCount; +} + +/** + * Synthesize a raw-model entry from an already-resolved nested combo so a + * parent combo's LCD folds the whole nested capability vector (context, + * output, modalities, capabilities) instead of only direct raw members. + * v1 parity (combo member synthesis at nested resolution time). + */ +function synthesizeNestedMember(name: string, nested: LegacyModelV2): OmniRouteRawModelEntry { + const inputModalities: string[] = []; + if (nested.capabilities.input.text) inputModalities.push("text"); + if (nested.capabilities.input.audio) inputModalities.push("audio"); + if (nested.capabilities.input.image) inputModalities.push("image"); + if (nested.capabilities.input.video) inputModalities.push("video"); + if (nested.capabilities.input.pdf) inputModalities.push("pdf"); + const outputModalities: string[] = []; + if (nested.capabilities.output.text) outputModalities.push("text"); + if (nested.capabilities.output.audio) outputModalities.push("audio"); + if (nested.capabilities.output.image) outputModalities.push("image"); + if (nested.capabilities.output.video) outputModalities.push("video"); + if (nested.capabilities.output.pdf) outputModalities.push("pdf"); + return { + id: `combo-ref:${name}`, + context_length: nested.limit.context, + max_output_tokens: nested.limit.output, + ...(nested.limit.input !== undefined ? { max_input_tokens: nested.limit.input } : {}), + owned_by: "combo", + input_modalities: inputModalities, + output_modalities: outputModalities, + capabilities: { + temperature: nested.capabilities.temperature, + // A raw entry carries plain flags; the mapped model widens them to + // `boolean | { field }` (custom reasoning/thinking field). Every + // non-false form means the capability is present, which is all the + // LCD fold reads. + reasoning: isCapabilityEnabled(nested.capabilities.reasoning), + thinking: isCapabilityEnabled(nested.capabilities.interleaved), + attachment: nested.capabilities.attachment, + tool_calling: nested.capabilities.toolcall, + }, + }; +} + +export async function publishCatalog( + draft: CatalogDraft, + opts: ResolvedOptions, + fetchers?: CatalogFetchers +): Promise { + const X = opts.providerId; + const log = opts.logger ?? createLogger(opts.startupDebug ? "debug" : (opts.logLevel ?? "warn")); + const modelsTimeout = opts.timeouts?.models ?? opts.timeoutMs; + const combosTimeout = opts.timeouts?.combos ?? opts.timeoutMs; + // v1 parity keeps the 5s auto-combos budget when no per-endpoint value is + // set (P2 resolves it in index.ts; direct publishCatalog callers may only + // pass timeoutMs). + const autoCombosTimeout = opts.timeouts?.autoCombos ?? 5_000; + // The contract is discovered from the object the host seeds into the + // provider draft, which the host fills before any model is published. The + // verdict is then reused for every model: the model seed carries no + // discriminating key, and a single provider/model pair always speaks one + // contract. + let hostContract: HostContract = "unknown"; + draft.provider.update(X, (p) => { + hostContract = detectHostContract(p); + assignProviderFields( + p, + { + name: opts.displayName ?? "OmniRoute", + api: { + type: "aisdk", + package: "@ai-sdk/openai-compatible", + url: ensureV1Suffix(opts.baseURL), + }, + integrationID: X, + }, + hostContract + ); + }); + log.debug(`[omniroute-v2] host catalog contract detected: ${hostContract}`); + + const modelsFetcher = fetchers?.fetcher ?? fetchers?.models; + const combosFetcher = fetchers?.combosFetcher ?? fetchers?.combos; + const autoCombosFetcher = fetchers?.autoCombosFetcher ?? fetchers?.autoCombos; + const providersFetcher = fetchers?.providersFetcher ?? fetchers?.providers; + + let rawModels: OmniRouteRawModelEntry[]; + try { + rawModels = modelsFetcher ? await modelsFetcher(opts.baseURL, opts.apiKey, modelsTimeout) : []; + } catch (err) { + log.warn( + `[omniroute-v2] models fetch failed, publishing empty catalog: ${err instanceof Error ? err.message : String(err)}` + ); + return { models: 0, combos: 0, autoCombos: 0 }; + } + + const visibleFilter = compileModelListFilter(opts.visibleModels); + const hiddenFilter = compileModelListFilter(opts.hiddenModels); + + const enrichment = await resolveEnrichmentOverlay(opts, fetchers, log); + const canonicalToAlias = buildCanonicalToAliasMap(enrichment); + const canonicalDedup = canonicalDedupSet(rawModels, canonicalToAlias); + + const usable = await resolveUsableAliases( + opts, + providersFetcher, + fetchers?.onSourceError, + enrichment, + modelsTimeout, + log + ); + + const rawModelById = new Map(); + for (const entry of rawModels) { + if (entry.id) rawModelById.set(entry.id, entry); + } + + const publishedKeys = new Set(); + // Mapped model id per published key (models and combos alike). Mirrors + // v1's `models[comboKey]` lookup so the intentional-dedup check sees the + // overwritten entry's id, not just key presence. + const publishedModelIds = new Map(); + let modelCount = 0; + for (const entry of rawModels) { + if (!entry.id) continue; + if (canonicalDedup.has(entry.id)) continue; + if (usable && !isUsableRawModelId(entry.id, usable)) continue; + if (!passesModelAllowlist(entry.id, visibleFilter, hiddenFilter)) continue; + const mapped = mapRawModelToModelV2(entry, { + providerId: X, + baseURL: opts.baseURL, + apiFormat: opts.apiFormat, + }); + applyEnrichment(mapped, lookupEnrichment(entry.id, enrichment, canonicalToAlias), { + providerTag: opts.providerTag !== false, + }); + const mid = mapped.id.startsWith(X + "/") ? mapped.id.slice(X.length + 1) : mapped.id; + draft.model.update(X, mid, (m) => { + assignModelFields(m, mapped, hostContract); + }); + publishedKeys.add(X + "/" + mid); + publishedModelIds.set(X + "/" + mid, mapped.id); + modelCount += 1; + } + + const warnedCombos = opts.collisionWarned ?? new Set(); + const cacheKey = `${opts.baseURL}::${opts.providerId}`; + const comboCount = await publishCombos({ + draft, + opts, + log, + providerId: X, + hostContract, + enrichment, + rawModelById, + publishedKeys, + publishedModelIds, + visibleFilter, + hiddenFilter, + usable, + canonicalToAlias, + combosFetcher, + combosTimeout, + warnedCombos, + cacheKey, + }); + if (comboCount === undefined) return { models: modelCount, combos: 0, autoCombos: 0 }; + + // Migration: v1 published opencode-X; v2 publishes X bare. Sessions pinned + // opencode-X resolve ModelUnavailableError -- see RELEASE.md migration note. + // Re-publishing under "opencode-"+X here is FORBIDDEN: a double + // publish would double chat entries in the picker. + + // Auto combos: virtual server-side entries from /api/combos/auto, keyed + // "auto" / "auto/" (v1 parity). Fail-open: a fetcher throw keeps + // models + combos and only warns - old gateways may not serve the + // endpoint at all (the default fetcher maps 404 to [] itself). + let rawAutoCombos: OmniRouteRawAutoCombo[]; + try { + rawAutoCombos = autoCombosFetcher + ? await autoCombosFetcher( + opts.baseURL, + opts.managementReadToken ?? opts.apiKey, + autoCombosTimeout + ) + : []; + } catch (err) { + log.warn( + `[omniroute-v2] auto combos fetch failed, falling back to models+combos catalog: ${err instanceof Error ? err.message : String(err)}` + ); + return { models: modelCount, combos: comboCount, autoCombos: 0 }; + } + + let autoComboCount = 0; + for (const autoCombo of rawAutoCombos) { + if (!autoCombo || !autoCombo.id) continue; + if (autoCombo.isHidden === true) continue; + // Auto combos are catalog entries like any other: an id a user asked to + // hide must stay hidden, and an allowlist that excludes it must exclude + // it. They used to skip both filters entirely. + if (!passesModelAllowlist(autoCombo.id, visibleFilter, hiddenFilter)) continue; + if (usable && !isUsableRawModelId(autoCombo.id, usable)) continue; + const mapped = mapAutoComboToModelV2(autoCombo, X, opts.baseURL, opts.apiFormat); + applyEnrichment(mapped, lookupEnrichment(autoCombo.id, enrichment, canonicalToAlias), { + isCombo: true, + isAutoCombo: true, + }); + const key = X + "/" + mapped.id; + if (publishedKeys.has(key)) { + const dedupeKey = `${cacheKey}::${key}`; + if (!warnedCombos.has(dedupeKey)) { + warnedCombos.add(dedupeKey); + log.warn( + `[omniroute-v2] auto combo key "${key}" collides with a model id; auto combo wins.` + ); + } + } + draft.model.update(X, mapped.id, (m) => { + assignModelFields(m, mapped, hostContract); + }); + publishedKeys.add(key); + publishedModelIds.set(key, mapped.id); + autoComboCount += 1; + } + + return { models: modelCount, combos: comboCount, autoCombos: autoComboCount }; +} diff --git a/@omniroute/opencode-plugin-v2/src/compat.ts b/@omniroute/opencode-plugin-v2/src/compat.ts new file mode 100644 index 0000000000..1c4aa9fe14 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/compat.ts @@ -0,0 +1,66 @@ +function isObject(value: unknown): value is Record { + return typeof value === "object" && value !== null; +} + +function isTransformHolder(value: unknown): value is { transform: unknown } { + return isObject(value) && "transform" in value; +} + +/** + * The catalog domain is the one this plugin cannot work without. The + * integration domain carries the credential flow and the `aisdk` domain the + * tool-schema cleaning: a host missing either still gets its catalog, so + * neither is asserted here — each is probed where it is used. + */ +export function assertContext(ctx: unknown): void { + if (!isObject(ctx)) { + throw new Error("[omniroute-v2] contract breach: ctx must be an object"); + } + if (!isTransformHolder(ctx.catalog) || typeof ctx.catalog.transform !== "function") { + throw new Error("[omniroute-v2] contract breach: ctx.catalog.transform must be a function"); + } + if (!isObject(ctx.options)) { + throw new Error("[omniroute-v2] contract breach: ctx.options must be an object"); + } +} + +/** + * Catalog contract spoken by the running host. + * + * opencode v2 is a moving target: the catalog contract changed between the + * binary that ships today and the SDK types this package pins. Rather than + * keying off a version list (which goes stale on the next release), the + * contract is discovered at runtime from the object the host seeds into the + * draft. + * + * - `legacy-package` — the seed carries a top-level `package` and no `api` + * block. Observed on `@opencode-ai/cli` 0.0.0-beta-17823, whose + * `Provider.Info.empty` is `{id, name, activation, package}`. + * - `sdk-api` — the seed carries an `api` block. This is the contract of the + * pinned `@opencode-ai/plugin`/`@opencode-ai/sdk` types. + * - `unknown` — neither or both. The caller publishes the superset. + */ +export type HostContract = "legacy-package" | "sdk-api" | "unknown"; + +export function detectHostContract(seed: unknown): HostContract { + if (!isObject(seed)) return "unknown"; + const hasApi = "api" in seed; + const hasPackage = "package" in seed; + if (hasApi && !hasPackage) return "sdk-api"; + if (hasPackage && !hasApi) return "legacy-package"; + return "unknown"; +} + +/** + * Whether to publish the legacy top-level fields (`package`, `settings`, + * `headers`, `variants[].settings`) next to the `api`-block fields. + * + * A host proven to speak the legacy contract gets them because it needs them; + * an unrecognised host gets them because the superset is the safer default + * (both field sets have been observed to survive an unknown-key write). A host + * that speaks the `api` contract does not, so a future strict schema cannot + * reject the write on an excess property. + */ +export function emitsLegacyFields(contract: HostContract): boolean { + return contract !== "sdk-api"; +} diff --git a/@omniroute/opencode-plugin-v2/src/credentials.ts b/@omniroute/opencode-plugin-v2/src/credentials.ts new file mode 100644 index 0000000000..9cf2b8536b --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/credentials.ts @@ -0,0 +1,101 @@ +import type { PluginContext } from "@opencode-ai/plugin/v2/promise"; +import type { Logger } from "./shared/index.js"; + +/** Where a resolved key came from, so the failure message can name the fix. */ +export type ApiKeyOrigin = "connection" | "option" | "env" | "missing"; + +export interface ResolvedApiKey { + key: string; + origin: ApiKeyOrigin; +} + +const ENV_VAR = "OMNIROUTE_API_KEY"; + +/** + * `ctx.integration.connection` is newer than the `key`/`env` methods this + * plugin registers, so a host that predates it exposes `integration` without + * it. Probing the shape keeps the plugin loadable on both. + */ +function connectionApi(ctx: PluginContext): PluginContext["integration"]["connection"] | undefined { + const connection = (ctx.integration as Partial).connection; + if ( + connection === undefined || + typeof connection.active !== "function" || + typeof connection.resolve !== "function" + ) { + return undefined; + } + return connection; +} + +/** + * Read the credential the user stored through the host's own auth flow. + * + * The plugin advertises `key` and `env` methods on its integration, so a user + * can connect it from the UI; without this lookup that connection would only + * feed inference and the catalog fetches would still need a key pasted into + * the config file. + * + * Returns `undefined` (never throws) when there is no connection, when the + * host is too old to expose one, or when the stored credential is an OAuth + * grant — this plugin authenticates the gateway with a bearer key, and an + * access token from an unrelated grant is not one. + */ +async function keyFromConnection( + ctx: PluginContext, + integrationID: string, + log: Logger +): Promise { + const connection = connectionApi(ctx); + if (connection === undefined) return undefined; + try { + const active = await connection.active(integrationID); + if (active === undefined) return undefined; + const credential = await connection.resolve(active); + if (credential === undefined) return undefined; + if (credential.type !== "key") { + log.warn( + `[omniroute-v2] ignoring the stored ${credential.type} credential: this plugin authenticates with an API key` + ); + return undefined; + } + return credential.key.length > 0 ? credential.key : undefined; + } catch (err) { + log.warn( + `[omniroute-v2] could not read the stored credential: ${err instanceof Error ? err.message : String(err)}` + ); + return undefined; + } +} + +/** + * Resolve the gateway key, preferring the credential the host holds over one + * written in config. A key in `opencode.json` still wins over the environment + * so an explicit per-project override keeps working. + */ +export async function resolveApiKey( + ctx: PluginContext, + integrationID: string, + optionKey: string | undefined, + log: Logger +): Promise { + const stored = await keyFromConnection(ctx, integrationID, log); + if (stored !== undefined) return { key: stored, origin: "connection" }; + if (optionKey !== undefined && optionKey.length > 0) return { key: optionKey, origin: "option" }; + const fromEnv = process.env[ENV_VAR]; + if (fromEnv !== undefined && fromEnv.length > 0) return { key: fromEnv, origin: "env" }; + return { key: "", origin: "missing" }; +} + +/** + * A missing key produces an empty catalog and no error the user can see, so + * say it once, and name the three ways to supply one. + */ +export function warnIfMissing(resolved: ResolvedApiKey, integrationID: string, log: Logger): void { + if (resolved.origin !== "missing") return; + log.warn( + `[omniroute-v2] no API key for "${integrationID}": the catalog will be empty. ` + + `Connect the integration from opencode, set "apiKey" in the plugin options, ` + + `or export ${ENV_VAR}.` + ); +} diff --git a/@omniroute/opencode-plugin-v2/src/enrichment-report.ts b/@omniroute/opencode-plugin-v2/src/enrichment-report.ts new file mode 100644 index 0000000000..d2ab275ae8 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/enrichment-report.ts @@ -0,0 +1,41 @@ +import type { Logger } from "./shared/index.js"; + +/** What the catalog loses when a given gateway source cannot be read. */ +function consequenceOf(endpoint: string): string { + if (endpoint.includes("/api/providers")) { + return "the usable-provider filter is disabled for this refresh, so unprovisioned providers stay listed"; + } + return "model names, provider tags, canonical dedupe and pricing are degraded"; +} + +/** + * A source the gateway refuses is not fatal — the catalog still publishes — + * but staying quiet about it is: the picker then shows raw ids, or lists + * providers that cannot serve, with nothing telling the user why. Say it once + * per endpoint so a refresh loop cannot spam the log. + * + * `usingFallbackToken` is true when no `managementReadToken` was configured and + * the inference key stands in for it, which is the usual reason a gateway + * answers 401/403 on `/api/*` — the advice differs from a token that was set + * and still got rejected. + */ +export function createSourceErrorReporter( + log: Logger, + usingFallbackToken: boolean +): (endpoint: string, reason: string) => void { + const warned = new Set(); + return (endpoint, reason) => { + if (warned.has(endpoint)) return; + warned.add(endpoint); + const unauthorized = reason.includes("401") || reason.includes("403"); + const hint = !unauthorized + ? "" + : usingFallbackToken + ? ` These endpoints need a management token: set "managementReadToken" in the plugin options ` + + `(it currently falls back to "apiKey", which a gateway usually rejects here).` + : ` The configured "managementReadToken" was rejected — check it grants read access to /api/*.`; + log.warn( + `[omniroute-v2] gateway source ${endpoint} unavailable (${reason}): ${consequenceOf(endpoint)}.${hint}` + ); + }; +} diff --git a/@omniroute/opencode-plugin-v2/src/gemini-language.ts b/@omniroute/opencode-plugin-v2/src/gemini-language.ts new file mode 100644 index 0000000000..c571367166 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/gemini-language.ts @@ -0,0 +1,43 @@ +import type { LanguageModelV3 } from "@ai-sdk/provider"; +import { type Logger, isGeminiModelId, sanitizeToolInputSchemas } from "./shared/index.js"; + +type CallOptions = Parameters[0]; + +/** + * Gemini answers `400 INVALID_ARGUMENT` — for the entire request, not just the + * offending tool — when a tool declaration carries `$schema` or + * `additionalProperties`. Anything upstream that emits standard JSON Schema + * therefore breaks tool calling as soon as the chain routes to Gemini. A + * `$ref` is forwarded untouched instead: stripping it would widen the schema + * to "accept anything", which is worse than letting the gateway answer. The + * v1 plugin dealt with this by wrapping `fetch` and rewriting the JSON body; the + * v2 home for it is the language model, where the tools are still structured + * data and no re-parsing is needed. + * + * Returns the model untouched when it is not bound for Gemini, so the wrapper + * costs nothing on every other chain. + */ +export function sanitizeToolSchemasFor( + language: T, + modelId: string, + log: Logger +): T { + if (language === undefined) return language; + if (!isGeminiModelId(modelId)) return language; + + const clean = (options: CallOptions): CallOptions => { + const tools = sanitizeToolInputSchemas(options.tools); + if (tools === undefined) return options; + log.debug( + `[omniroute-v2] stripped Gemini-incompatible schema keywords from ${tools.length} tool declaration(s) for ${modelId}` + ); + return { ...options, tools } as CallOptions; + }; + + // Prototype-linked so every other member of the model — including accessors + // and anything a future SDK version adds — keeps working untouched. + const wrapped: LanguageModelV3 = Object.create(language as object) as LanguageModelV3; + wrapped.doGenerate = (options) => language.doGenerate(clean(options)); + wrapped.doStream = (options) => language.doStream(clean(options)); + return wrapped as T; +} diff --git a/@omniroute/opencode-plugin-v2/src/index.ts b/@omniroute/opencode-plugin-v2/src/index.ts new file mode 100644 index 0000000000..f6c0471baf --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/index.ts @@ -0,0 +1,539 @@ +import { define, type PluginContext } from "@opencode-ai/plugin/v2/promise"; +import { + optionalTierFingerprint, + catalogContentFingerprint, + createLogger, + defaultOmniRouteAutoCombosFetcher, + defaultOmniRouteCombosFetcher, + defaultOmniRouteEnrichmentFetcher, + defaultOmniRouteModelsFetcher, + defaultOmniRouteProvidersFetcher, + type OmniRouteEnrichmentMap, + type OmniRouteProviderConnection, +} from "./shared/index.js"; +import type { + OmniRouteRawAutoCombo, + OmniRouteRawCombo, + OmniRouteRawModelEntry, +} from "./shared/index.js"; +import type { ResolvedOptions } from "./catalog.js"; +import { publishCatalog } from "./catalog.js"; +import { + DEFAULT_MODEL_CACHE_TTL_MS, + UNREACHABLE_COOLDOWN_MS, + memoryCacheKey, + readDiskSnapshot, + snapshotIdentityFingerprint, + writeDiskSnapshot, + type CatalogSnapshot, +} from "./cache.js"; +import { assertContext } from "./compat.js"; +import { type ApiKeyOrigin, resolveApiKey, warnIfMissing } from "./credentials.js"; +import { createSourceErrorReporter } from "./enrichment-report.js"; +import { sanitizeToolSchemasFor } from "./gemini-language.js"; +import { PLUGIN_ID, parsePluginOptions, resolveTimeouts, type PluginOptions } from "./options.js"; + +/** + * A fetch result that says whether it succeeded. Returning a bare `[]` on + * failure makes an outage indistinguishable from a gateway that legitimately + * has no combos — and the difference decides whether the last known value + * should be kept or dropped. + */ +type SourceResult = { ok: true; value: T } | { ok: false }; + +interface RefreshState { + entries: Map; + inFlight: Map>; + fingerprint: string | undefined; + /** Digest of the optional tier, so a reload only follows a real change. */ + optionalFingerprint: string | undefined; + /** + * When the last refresh found the gateway unreachable, skip the network + * until this timestamp and serve last-known-good instead. Without it every + * transform past TTL re-fires the full fetch suite against a gateway that + * just proved it cannot answer — a self-inflicted retry storm. + */ + unreachableUntil: number; +} + +function toResolvedOptions(parsed: PluginOptions): ResolvedOptions { + return { + providerId: parsed.providerId, + baseURL: parsed.baseURL, + apiKey: parsed.apiKey ?? process.env.OMNIROUTE_API_KEY ?? "", + managementReadToken: parsed.managementReadToken, + timeoutMs: parsed.timeoutMs, + timeouts: parsed.timeouts, + logLevel: parsed.logLevel, + startupDebug: parsed.startupDebug, + providerTag: parsed.providerTag, + modelCacheTtlMs: + typeof parsed.modelCacheTtlMs === "number" && parsed.modelCacheTtlMs > 0 + ? parsed.modelCacheTtlMs + : DEFAULT_MODEL_CACHE_TTL_MS, + displayName: parsed.displayName, + apiFormat: parsed.apiFormat, + visibleModels: parsed.visibleModels, + hiddenModels: parsed.hiddenModels, + usableOnly: parsed.usableOnly, + enrichment: parsed.enrichment, + }; +} + +export default define({ + id: PLUGIN_ID, + setup: async (ctx: PluginContext) => { + assertContext(ctx); + const parsed = parsePluginOptions(ctx.options); + const X = parsed.providerId; + const resolved = toResolvedOptions(parsed); + const timeouts = resolveTimeouts(parsed); + const log = createLogger(parsed.startupDebug ? "debug" : (parsed.logLevel ?? "warn")); + resolved.logger = log; + resolved.logLevel = parsed.logLevel; + resolved.startupDebug = parsed.startupDebug; + log.info(`[omniroute-v2] init providerId=${X}`); + + // v1 parity port: in-memory TTL + disk snapshot. The memory key + // `baseURL::sha256(creds)` isolates credential tuples (prod vs + // staging); the TTL is checked in the transform before any fetch; + // concurrent calls share the refresh promise in the setup closure keyed + // by (providerId, baseURL); the disk snapshot feeds warm-startup and + // the offline fallback. The existing in-memory keep-last-good is kept. + const state: RefreshState = { + entries: new Map(), + inFlight: new Map(), + fingerprint: undefined, + optionalFingerprint: undefined, + unreachableUntil: 0, + }; + + // The credential the host holds wins over one written in config, so a + // user who connected the integration from the UI never has to paste a + // key into `opencode.json`. Reading it is async and the transforms must + // register synchronously, so the lookup happens on the first publish; + // until then the option/env key resolved above stands in. + const credentialsOf = (): { cacheKey: string; identityFingerprint: string } => ({ + cacheKey: memoryCacheKey( + resolved.baseURL, + `${resolved.apiKey}\0${resolved.managementReadToken ?? resolved.apiKey}` + ), + identityFingerprint: snapshotIdentityFingerprint( + resolved.baseURL, + resolved.apiKey, + resolved.managementReadToken ?? resolved.apiKey + ), + }); + let { cacheKey, identityFingerprint } = credentialsOf(); + + // Both keys are derived from the credential: two credentials must never + // share a snapshot, so they are recomputed whenever the key moves. + let credentialChecked = false; + let apiKeyOrigin: ApiKeyOrigin = resolved.apiKey.length > 0 ? "option" : "missing"; + const ensureCredential = async (): Promise => { + // Settled once a key is in hand: re-reading on every refresh would let + // a mid-session change silently repoint the snapshot keys. + if (credentialChecked && apiKeyOrigin !== "missing") return; + const next = await resolveApiKey(ctx, X, parsed.apiKey, log); + const moved = next.key !== resolved.apiKey; + resolved.apiKey = next.key; + apiKeyOrigin = next.origin; + if (moved) ({ cacheKey, identityFingerprint } = credentialsOf()); + if (!credentialChecked) warnIfMissing(next, X, log); + else if (moved) log.info(`[omniroute-v2] API key picked up from the ${next.origin} source`); + credentialChecked = true; + }; + + const fetchModelsSafe = async (): Promise => { + try { + return await defaultOmniRouteModelsFetcher( + resolved.baseURL, + resolved.apiKey, + timeouts.models + ); + } catch (err) { + log.warn( + `[omniroute-v2] models fetch failed, publishing empty catalog: ${err instanceof Error ? err.message : String(err)}` + ); + return []; + } + }; + // Failures are reported once per endpoint (with the management-token hint + // when the inference key stands in), so a gated `/api/*` degrades loudly + // rather than silently. Declared before the wrappers that use it. + const reportSourceError = createSourceErrorReporter( + log, + resolved.managementReadToken === undefined + ); + const fetchCombosSafe = async (): Promise> => { + try { + return { + ok: true, + value: await defaultOmniRouteCombosFetcher( + resolved.baseURL, + resolved.managementReadToken ?? resolved.apiKey, + timeouts.combos + ), + }; + } catch (err) { + const reason = err instanceof Error ? err.message : String(err); + reportSourceError("/api/combos", reason); + log.warn(`[omniroute-v2] combos fetch failed, keeping the last known combos: ${reason}`); + return { ok: false }; + } + }; + // Providers connections follow the same rule: gated on usableOnly (no + // request when false, v1 parity), soft-fail to [] so the filter degrades + // to keep-all instead of hiding the catalog. + const fetchProvidersSafe = async (): Promise> => { + if (!resolved.usableOnly) return { ok: true, value: [] }; + try { + return { + ok: true, + value: await defaultOmniRouteProvidersFetcher( + resolved.baseURL, + resolved.managementReadToken ?? resolved.apiKey, + timeouts.models, + reportSourceError + ), + }; + } catch (err) { + log.warn( + `[omniroute-v2] providers fetch failed, keeping the last known provider list: ${err instanceof Error ? err.message : String(err)}` + ); + return { ok: false }; + } + }; + // Enrichment follows the same rule: gated on the option (default on, + // v1 parity), soft-fail to an empty map so names/pricing degrade to + // mapper defaults instead of hiding the catalog. + const fetchEnrichmentSafe = async (): Promise> => { + if (resolved.enrichment === false) return { ok: true, value: new Map() }; + try { + return { + ok: true, + value: await defaultOmniRouteEnrichmentFetcher( + resolved.baseURL, + resolved.managementReadToken ?? resolved.apiKey, + timeouts.enrichment, + reportSourceError + ), + }; + } catch (err) { + log.warn( + `[omniroute-v2] enrichment fetch failed, keeping the last known names/pricing: ${err instanceof Error ? err.message : String(err)}` + ); + return { ok: false }; + } + }; + const fetchAutoCombosSafe = async (): Promise> => { + try { + return { + ok: true, + value: await defaultOmniRouteAutoCombosFetcher( + resolved.baseURL, + resolved.managementReadToken ?? resolved.apiKey, + timeouts.autoCombos, + log, + reportSourceError + ), + }; + } catch (err) { + // The default fetcher reports the refusal itself (with the + // management-token hint); this warn is the fallback for injected + // stubs that throw without reporting. + const reason = err instanceof Error ? err.message : String(err); + log.warn(`[omniroute-v2] auto combos fetch failed, keeping the last known ones: ${reason}`); + return { ok: false }; + } + }; + + /** + * Fetch in two tiers. Models are what a catalog *is*: without them there + * is nothing to publish. Everything else — combos, auto-combos, the + * provider list, the enrichment overlay — improves an already usable + * catalog, so awaiting any of them before publishing makes the catalog + * hostage to the slowest source: a gateway that accepts the connection + * and never answers one endpoint kept everything unpublished until that + * fetch's own timeout fired, which is longer than some hosts stay alive. + * + * The optional tier therefore keeps running after the publish and upgrades + * the stored snapshot when it lands, so the next transform serves the + * complete catalog. + */ + async function refreshSnapshot(): Promise { + // Models are what a catalog *is*; everything else improves one that + // already works. Combos used to sit here too, so a gateway slow to + // answer /api/combos held the whole picker back — the very thing the + // staged publish exists to prevent. + const essential = fetchModelsSafe(); + const optional = Promise.all([ + fetchCombosSafe(), + fetchAutoCombosSafe(), + fetchProvidersSafe(), + fetchEnrichmentSafe(), + ]); + const models = await essential; + const previous = state.entries.get(cacheKey); + // A gateway that just failed everything gets a short breather: serving + // last-known-good for a few seconds beats hammering it on every + // transform while it is down. Arms whenever the models fetch comes back + // empty — with or without a prior entry to serve — so a totally dead + // gateway stops getting hit every window. Partial degradation (models + // healthy, an optional tier failed) still retries normally next window. + if (models.length === 0) { + state.unreachableUntil = Date.now() + UNREACHABLE_COOLDOWN_MS; + } + // Carry every source forward until its replacement lands, and keep the + // old value when a fetch FAILED — but honour a gateway that legitimately + // returns nothing, which is a different answer from "I could not ask". + const snapshot: CatalogSnapshot = { + models, + combos: previous?.combos ?? [], + autoCombos: previous?.autoCombos ?? [], + providers: previous?.providers ?? [], + enrichment: previous?.enrichment ?? new Map(), + fetchedAt: Date.now(), + }; + if (models.length > 0) { + state.entries.set(cacheKey, snapshot); + await writeDiskSnapshot(X, snapshot, identityFingerprint); + } + void optional.then( + (parts) => upgradeWithOptional(snapshot, parts), + (err) => { + // The wrappers never reject; a throw here would be a bug in them, and + // an unhandled rejection is a worse way to learn about it. + log.warn( + `[omniroute-v2] optional catalog sources failed unexpectedly: ${err instanceof Error ? err.message : String(err)}` + ); + } + ); + return snapshot; + } + + /** + * Fold late optional data into the snapshot that was published without it. + * Skipped when a newer refresh has already replaced that snapshot, so a + * slow tier can never resurrect a stale catalog. + */ + async function upgradeWithOptional( + base: CatalogSnapshot, + [combos, autoCombos, providers, enrichment]: [ + SourceResult, + SourceResult, + SourceResult, + SourceResult, + ] + ): Promise { + if (state.entries.get(cacheKey) !== base) return; + // Per source: a success replaces (even with an empty answer — that is + // the gateway's answer), a failure keeps what we had. + const upgraded: CatalogSnapshot = { + ...base, + combos: combos.ok ? combos.value : base.combos, + autoCombos: autoCombos.ok ? autoCombos.value : base.autoCombos, + providers: providers.ok ? providers.value : base.providers, + enrichment: enrichment.ok ? enrichment.value : base.enrichment, + }; + const unchanged = + upgraded.combos === base.combos && + upgraded.autoCombos === base.autoCombos && + upgraded.providers === base.providers && + upgraded.enrichment === base.enrichment; + if (unchanged) return; + state.entries.set(cacheKey, upgraded); + if (upgraded.models.length > 0) { + await writeDiskSnapshot(X, upgraded, identityFingerprint); + } + // Reload only when the optional tier actually moved: the catalog + // fingerprint covers ids alone, so without this the host would rebuild + // its catalog once per TTL window for an identical result. + const optionalFingerprint = optionalTierFingerprint( + upgraded.autoCombos ?? [], + upgraded.providers ?? [], + upgraded.enrichment, + upgraded.combos + ); + const optionalChanged = state.optionalFingerprint !== optionalFingerprint; + state.optionalFingerprint = optionalFingerprint; + if (optionalChanged && typeof ctx.catalog.reload === "function") { + try { + await ctx.catalog.reload(); + } catch (err) { + log.warn( + `[omniroute-v2] catalog reload after late sources failed, keeping current catalog: ${err instanceof Error ? err.message : String(err)}` + ); + } + } + } + + function loadSnapshot(): Promise { + const now = Date.now(); + const hit = state.entries.get(cacheKey); + if (hit && hit.fetchedAt + resolved.modelCacheTtlMs > now) return Promise.resolve(hit); + // Cooldown after a total models failure: skip the network until it + // lapses. Serves last-known-good when one exists; otherwise the refresh + // below still runs (nothing to serve, no point pretending). + if (now < state.unreachableUntil && hit) return Promise.resolve(hit); + if (now >= state.unreachableUntil) state.unreachableUntil = 0; + const inflight = state.inFlight.get(cacheKey); + if (inflight) return inflight; + const snapshot = refreshSnapshot(); + state.inFlight.set(cacheKey, snapshot); + const clear = () => { + if (state.inFlight.get(cacheKey) === snapshot) state.inFlight.delete(cacheKey); + }; + snapshot.then(clear, clear); + return snapshot; + } + + // Warm-startup: the disk snapshot is read at boot (without blocking + // the synchronous transform registration) to publish the last-known + // catalog before the first successful fetch. + /** + * Warm start: publish the last known catalog from disk before the first + * fetch returns. Deliberately read *after* the credential is resolved — + * the snapshot is keyed by the credential tuple, and resolving the host + * credential changes that key, so reading at setup time would look up the + * wrong identity and reject a perfectly good snapshot. + */ + let warmLoadedFor: string | undefined; + const ensureWarmSnapshot = async (): Promise => { + if (warmLoadedFor === identityFingerprint) return; + warmLoadedFor = identityFingerprint; + const warm = await readDiskSnapshot(X, identityFingerprint, log); + if (warm && !state.entries.has(cacheKey)) state.entries.set(cacheKey, warm); + }; + + // Fail-closed models (keep-last-good, validated): an empty models fetch + // (transient 500/timeout) must not wipe a known catalog. The latest + // non-empty entry (fresh fetch or warm disk snapshot) is replayed + // instead of publishing the empty set. `refreshSnapshot` never overwrites + // the memory entry on failure, so `entries` stays the last-known-good + // source — including cross-setup via the disk snapshot. + // Fail-open one level down, in the wrappers (never reject) and the + // `publishCatalog` catches — so no try/catch here. + const catalogRegistration = ctx.catalog.transform(async (draft) => { + await ensureCredential(); + await ensureWarmSnapshot(); + const snapshot = await loadSnapshot(); + let effective = snapshot; + if (snapshot.models.length === 0) { + const stale = state.entries.get(cacheKey); + if (stale !== undefined && stale.models.length > 0) { + log.warn( + `[omniroute-v2] models fetch returned empty, keeping last-known catalog (${stale.models.length} models, ${stale.combos.length} combos)` + ); + effective = stale; + } + } + const counts = await (async (): Promise<{ + models: number; + combos: number; + autoCombos: number; + }> => { + // fetcher-level fail-open covers fetches; this guard covers mapper/draft throws. + try { + return await publishCatalog(draft, resolved, { + onSourceError: reportSourceError, + models: async () => effective.models, + combos: async () => effective.combos, + autoCombos: async () => effective.autoCombos, + providers: async () => effective.providers ?? [], + enrichment: async () => effective.enrichment ?? new Map(), + }); + } catch (err) { + log.warn( + `[omniroute-v2] catalog publish failed, keeping current catalog: ${err instanceof Error ? err.message : String(err)}` + ); + return { models: 0, combos: 0, autoCombos: 0 }; + } + })(); + void counts; + const fingerprint = catalogContentFingerprint( + effective.models, + effective.combos, + effective.autoCombos + ); + const changed = state.fingerprint !== undefined && state.fingerprint !== fingerprint; + state.fingerprint = fingerprint; + if (changed && typeof ctx.catalog.reload === "function") { + await Promise.resolve(); + try { + await ctx.catalog.reload(); + } catch (err) { + log.warn( + `[omniroute-v2] catalog reload failed, keeping current catalog: ${err instanceof Error ? err.message : String(err)}` + ); + } + } + }); + const integrationHook = (ctx.integration as Partial | undefined) + ?.transform; + // A host that exposes the hook but throws while registering it must cost + // the plugin nothing but the connect action: the throw happens OUTSIDE + // any await, so only a call-site guard catches it (an await-guard alone + // would let a synchronous throw escape setup and kill the catalog). + let integrationRegistration: unknown; + if (typeof integrationHook === "function") { + try { + integrationRegistration = integrationHook((draft) => { + draft.update(X, (integration) => { + integration.name = parsed.displayName ?? "OmniRoute"; + }); + draft.method.update({ integrationID: X, method: { type: "key", label: "API key" } }); + draft.method.update({ + integrationID: X, + method: { type: "env", names: ["OMNIROUTE_API_KEY"] }, + }); + }); + } catch (err) { + log.warn( + `[omniroute-v2] host refused the integration hook, the connect action will be missing: ${err instanceof Error ? err.message : String(err)}` + ); + integrationRegistration = undefined; + } + } + /** + * `aisdk.language` is newer than the catalog domain, so a host may not + * expose it; the plugin must stay loadable there, minus the sanitising. + */ + const languageHook = (ctx.aisdk as Partial | undefined)?.language; + // A host that rejects this registration must cost the catalog nothing: the + // plugin is a catalog first, and tool-schema cleaning is an extra. + let languageRegistration: Promise<{ dispose: () => Promise }> | undefined; + if (parsed.geminiSanitization !== false && typeof languageHook === "function") { + try { + languageRegistration = languageHook((input) => { + if (input.model.providerID !== X) return; + input.language = sanitizeToolSchemasFor(input.language, input.model.id, log); + }); + } catch (err) { + log.warn( + `[omniroute-v2] host refused the language-model hook, Gemini tool schemas will not be cleaned: ${err instanceof Error ? err.message : String(err)}` + ); + } + } + + await catalogRegistration; + if (integrationRegistration !== undefined) { + try { + await integrationRegistration; + } catch (err) { + log.warn( + `[omniroute-v2] host refused the integration hook, the connect action will be missing: ${err instanceof Error ? err.message : String(err)}` + ); + } + } + if (languageRegistration !== undefined) { + try { + await languageRegistration; + } catch (err) { + log.warn( + `[omniroute-v2] language-model hook registration failed, Gemini tool schemas will not be cleaned: ${err instanceof Error ? err.message : String(err)}` + ); + } + } + }, +}); diff --git a/@omniroute/opencode-plugin-v2/src/options.ts b/@omniroute/opencode-plugin-v2/src/options.ts new file mode 100644 index 0000000000..9782ca74e6 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/options.ts @@ -0,0 +1,117 @@ +import { z } from "zod"; + +const apiFormatSchema = z + .object({ + allowAnthropic: z.boolean().optional(), + anthropicModels: z.array(z.string()).optional(), + // Deprecated v1 prefix list. Accepted (warn at resolve time) so copied + // v1 configs keep routing; prefer anthropicModels (full IDs). + anthropicPrefixes: z.array(z.string()).optional(), + }) + .strict(); + +const timeoutsSchema = z + .object({ + models: z.number().positive().optional(), + combos: z.number().positive().optional(), + autoCombos: z.number().positive().optional(), + enrichment: z.number().positive().optional(), + }) + .strict(); + +const pluginOptionsSchema = z + .object({ + // Reaches a filesystem path (the on-disk catalog snapshot) and the + // catalog keys, so it is bounded here rather than escaped at each use. + providerId: z + .string() + .regex(/^[A-Za-z0-9._-]+$/, "providerId may only contain letters, digits, '.', '_' and '-'") + .refine((v) => v !== "." && v !== "..", "providerId cannot be a path segment") + .default("omniroute"), + baseURL: z.string().url(), + apiKey: z.string().optional(), + displayName: z.string().optional(), + managementReadToken: z.string().optional(), + timeoutMs: z.number().positive().default(10000), + timeouts: timeoutsSchema.optional(), + logLevel: z.enum(["error", "warn", "info", "debug"]).optional(), + startupDebug: z.boolean().optional(), + modelCacheTtlMs: z.number().positive().optional(), + visibleModels: z.array(z.string()).optional(), + hiddenModels: z.array(z.string()).optional(), + usableOnly: z.boolean().default(false), + // v1 parity: enrichment overlay on by default (names + pricing). + enrichment: z.boolean().default(true), + // v1 parity: strip the JSON-Schema keywords Gemini rejects from tool + // declarations bound for a Gemini model. On by default — leaving them in + // fails the whole request with 400 INVALID_ARGUMENT. + geminiSanitization: z.boolean().default(true), + // v1 parity: prefix a model's display name with the upstream provider it + // routes to, so the same model sold through two connections is + // distinguishable in the picker. + providerTag: z.boolean().default(true), + apiFormat: apiFormatSchema.optional(), + }) + .strict(); + +export type PluginOptions = z.infer; + +/** Per-endpoint timeout defaults (v1 parity). `timeoutMs` is the global fallback. */ +export const DEFAULT_TIMEOUT_MS = 10_000 as const; +/** Auto-combos keep the v1 5s budget; the field is resolved now for the P3 port. */ +export const DEFAULT_AUTO_COMBOS_TIMEOUT_MS = 5_000 as const; + +export interface EndpointTimeouts { + models: number; + combos: number; + autoCombos: number; + enrichment: number; +} + +export function resolveTimeouts( + opts: Pick +): EndpointTimeouts { + const fallback = + typeof opts.timeoutMs === "number" && opts.timeoutMs > 0 ? opts.timeoutMs : DEFAULT_TIMEOUT_MS; + return { + models: opts.timeouts?.models ?? fallback, + combos: opts.timeouts?.combos ?? fallback, + autoCombos: opts.timeouts?.autoCombos ?? DEFAULT_AUTO_COMBOS_TIMEOUT_MS, + enrichment: opts.timeouts?.enrichment ?? fallback, + }; +} + +/** + * Parse the plugin block of `opencode.json`. + * + * A rejected option aborts the whole plugin, and the host reports that as a + * bare load failure with the validator's raw dump attached — which is how a + * single mistyped key turns into a wall of JSON and an empty model picker. The + * schema is strict on purpose (a silently ignored option is worse), so the + * least we owe the user is a first line naming what to fix. + */ +export function parsePluginOptions(raw: unknown): PluginOptions { + const result = pluginOptionsSchema.safeParse(raw); + if (result.success) return result.data; + const problems = result.error.issues.map((issue) => { + const at = issue.path.length > 0 ? issue.path.join(".") : "(root)"; + const unknown = issue.code === "unrecognized_keys" ? issue.keys.join(", ") : undefined; + return unknown !== undefined ? `unknown option "${unknown}"` : `${at}: ${issue.message}`; + }); + throw new Error(`[omniroute-v2] invalid plugin options — ${problems.join("; ")}`); +} + +/** + * The host reads the plugin id from the module, before any option is known, so + * it cannot carry the configured provider id. Publishing two gateways from one + * install is a `providerId` matter — that one does reach the catalog. + */ +export const PLUGIN_ID = "omniroute-v2"; + +export function providerIdFor(providerId: string): string { + return providerId; +} + +export function integrationIdFor(providerId: string): string { + return providerId; +} diff --git a/@omniroute/opencode-plugin-v2/src/shared/auto-combos.ts b/@omniroute/opencode-plugin-v2/src/shared/auto-combos.ts new file mode 100644 index 0000000000..04f70f7625 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/shared/auto-combos.ts @@ -0,0 +1,219 @@ +import type { Model as ModelV2 } from "@opencode-ai/sdk/v2"; +import type { ApiFormatV2 } from "./models-map.js"; +import { resolveApiBlockV2 } from "./models-map.js"; +import { autoComboModelId, formatAutoComboName, type AutoVariant } from "./naming.js"; + +export type { AutoVariant }; + +/** + * Raw shape of an auto combo entry as returned by OmniRoute's + * `/api/combos/auto` endpoint. Auto combos are virtual -- they self-manage + * provider selection via scoring/bandit exploration at runtime. + * + * Ported from the v1 plugin (`index.ts:1672-1698`); the shape is unchanged + * so old and new gateways stay wire-compatible. + */ +export interface OmniRouteRawAutoCombo { + /** Stable id (e.g. "auto", "auto/coding"). */ + id: string; + /** Human-readable name (e.g. "Auto", "Auto Coding"). */ + name?: string; + /** Variant key or undefined for the default auto. */ + variant?: AutoVariant; + /** Provider names eligible for this auto combo. */ + candidatePool?: string[]; + /** Number of candidates resolved at fetch time. */ + candidateCount?: number; + /** MAX of candidates' context windows, served by newer gateway builds. + * Absent on older servers -- the mapper falls back to a safe default. */ + context_length?: number; + /** MAX of candidates' max output tokens (same provenance as context_length). */ + max_output_tokens?: number; + /** Whether this auto combo should be hidden from the picker. */ + isHidden?: boolean; + /** Auto-combo configuration. */ + config?: { + auto?: { + candidatePool?: string[]; + explorationRate?: number; + routerStrategy?: string; + }; + }; +} + +/** Minimal warn sink so the fetcher never depends on the plugin logger. */ +export interface AutoCombosWarnSink { + warn: (message: string, ...args: unknown[]) => void; +} + +/** + * Fetcher contract for `/api/combos/auto`. Returns the list of virtual + * auto combos the server can create. Same DI shape as the other fetchers + * so unit tests can inject a stub instead of monkey-patching `fetch`. + * + * HTTP refusals (non-2xx other than 404) and network errors THROW: the caller + * distinguishes "the gateway failed" (keep last-known) from "the gateway + * answered empty" (publish empty). Only 404 stays soft — the endpoint does + * not exist yet on older gateways, and that is an answer, not a failure. + */ +export type OmniRouteAutoCombosFetcher = ( + baseURL: string, + apiKey: string, + timeoutMs?: number, + logger?: AutoCombosWarnSink, + onSourceError?: (endpoint: string, reason: string) => void +) => Promise; + +function trimTrailingSlashes(value: string): string { + let i = value.length; + while (i > 0 && value.charCodeAt(i - 1) === 0x2f /* "/" */) i--; + return i === value.length ? value : value.slice(0, i); +} + +function fallbackWarn(message: string, ...args: unknown[]): void { + console.warn(`[omniroute-plugin] [WARN] ${message}`, ...args); +} + +/** + * Default auto combos fetcher: `GET /api/combos/auto`. + * + * 404 stays soft (endpoint not deployed yet on older gateways — an answer, + * not a failure). Any other non-2xx or network error THROWS so the caller + * keeps last-known instead of publishing an empty tier: a 403 behind a + * management-token gate must not wipe the auto combos the picker had. + * v1 parity keeps the 5s timeout budget. + */ +export const defaultOmniRouteAutoCombosFetcher: OmniRouteAutoCombosFetcher = async ( + baseURL, + apiKey, + timeoutMs = 5_000, + logger?: AutoCombosWarnSink, + onSourceError?: (endpoint: string, reason: string) => void +) => { + if (!apiKey || !baseURL) return []; + const warn = logger?.warn ?? fallbackWarn; + const report = (reason: string): void => { + warn(reason); + onSourceError?.("/api/combos/auto", reason); + }; + + const trimmed = trimTrailingSlashes(baseURL); + const root = trimmed.replace(/\/v\d+$/, ""); + const url = `${root}/api/combos/auto`; + + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), timeoutMs); + try { + const res = await fetch(url, { + method: "GET", + headers: { + Authorization: `Bearer ${apiKey}`, + Accept: "application/json", + }, + signal: controller.signal, + }); + // 404 = endpoint not deployed yet -- expected during rollout + if (res.status === 404) { + warn(`/api/combos/auto not available (404) -- auto combos disabled`); + return []; + } + if (!res.ok) { + const reason = `HTTP ${res.status} ${res.statusText}`; + report(`/api/combos/auto refused (${reason}) -- keeping last-known auto combos`); + throw new Error(reason); + } + const body = (await res.json()) as unknown; + const rawList: unknown[] = Array.isArray(body) + ? body + : body && typeof body === "object" && Array.isArray((body as { combos?: unknown }).combos) + ? ((body as { combos: unknown[] }).combos as unknown[]) + : []; + const out: OmniRouteRawAutoCombo[] = []; + for (const r of rawList) { + if (r && typeof r === "object" && typeof (r as { id?: unknown }).id === "string") { + out.push(r as OmniRouteRawAutoCombo); + } + } + return out; + } catch (err) { + // Network error, timeout, abort -- keep last-known, never publish empty. + // (The 404-soft path above returns directly and never reaches this throw.) + const reason = `/api/combos/auto fetch failed: ${err instanceof Error ? err.message : String(err)} -- keeping last-known auto combos`; + report(reason); + throw err instanceof Error ? err : new Error(String(err)); + } finally { + clearTimeout(timer); + } +}; + +/** Fallbacks when the server does not advertise auto-combo limits (older + * gateway builds). MUST be positive: OpenCode's overflow guard treats + * `limit.context === 0` as "never overflow" and silently DISABLES smart + * auto-compaction, letting the session grow until the gateway's destructive + * history purge kicks in. */ +export const AUTO_COMBO_FALLBACK_CONTEXT = 128_000; +export const AUTO_COMBO_FALLBACK_OUTPUT = 8_192; + +/** + * Convert a raw auto combo into a `ModelV2` entry for the picker. + * Auto combos route to capable models, so tool_call and reasoning default + * to true. Context/output limits come from the server (MAX of the + * candidate pool's windows); a safe positive fallback applies when the + * server omits them. Never 0. + */ +export function mapAutoComboToModelV2( + autoCombo: OmniRouteRawAutoCombo, + providerId: string, + baseURL: string, + apiFormat?: ApiFormatV2 +): ModelV2 { + const name = formatAutoComboName(autoCombo.variant, autoCombo.candidateCount); + const context = + typeof autoCombo.context_length === "number" && autoCombo.context_length > 0 + ? autoCombo.context_length + : AUTO_COMBO_FALLBACK_CONTEXT; + const output = + typeof autoCombo.max_output_tokens === "number" && autoCombo.max_output_tokens > 0 + ? autoCombo.max_output_tokens + : AUTO_COMBO_FALLBACK_OUTPUT; + return { + id: autoComboModelId(autoCombo.variant), + providerID: providerId, + api: resolveApiBlockV2(autoComboModelId(autoCombo.variant), baseURL, apiFormat), + name, + capabilities: { + temperature: true, + reasoning: true, + attachment: false, + toolcall: true, + input: { + text: true, + audio: false, + image: false, + video: false, + pdf: false, + }, + output: { + text: true, + audio: false, + image: false, + video: false, + pdf: false, + }, + interleaved: false, + }, + cost: { + input: 0, + output: 0, + cache: { read: 0, write: 0 }, + }, + limit: { + context, + output, + }, + status: "active", + options: {}, + headers: {}, + release_date: "", + }; +} diff --git a/@omniroute/opencode-plugin-v2/src/shared/combos-map.ts b/@omniroute/opencode-plugin-v2/src/shared/combos-map.ts new file mode 100644 index 0000000000..74ad94117f --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/shared/combos-map.ts @@ -0,0 +1,254 @@ +import type { Model as ModelV2 } from "@opencode-ai/sdk/v2"; +import { type ApiFormatV2, type OmniRouteRawModelEntry, resolveApiBlockV2 } from "./models-map.js"; + +export interface OmniRouteRawComboMemberRef { + /** Step kind: "model" references a raw model id; "combo-ref" nests another combo. */ + kind?: "model" | "combo-ref"; + /** Full model id referenced by this step (when kind === "model"). */ + model?: string; + /** Nested combo name (when kind === "combo-ref"). */ + comboName?: string; + /** Routing weight inside the combo (0–100, advisory at LCD time). */ + weight?: number; + /** Step-local label, distinct from the parent combo's display name. */ + label?: string; +} + +export interface OmniRouteRawCombo { + id: string; + name?: string; + /** Routing strategy. Surfaced for forward-compat but not consumed by LCD. */ + strategy?: string; + /** Member step list. Only `kind: "model"` steps participate in LCD. */ + models?: OmniRouteRawComboMemberRef[]; + /** Hidden combos are excluded from the OC model picker. */ + isHidden?: boolean; + /** When OmniRoute attaches a lifecycle hint we forward it; today it doesn't. */ + release_date?: string; + /** + * Server-computed context window for this combo (aggregated from member + * models using the same logic as /v1/models). When present, the client + * uses this value directly instead of re-aggregating from member models. + * + * Added in 3.9.x — old servers do not send it. + */ + computed_context_length?: number; +} + +/** + * Fetcher contract for `/api/combos`. Same DI shape as + * `OmniRouteModelsFetcher` so unit tests can inject a stub instead of + * monkey-patching global `fetch`. + */ +export type OmniRouteCombosFetcher = ( + baseURL: string, + apiKey: string, + timeoutMs?: number +) => Promise; + +function trimTrailingSlashes(value: string): string { + let i = value.length; + while (i > 0 && value.charCodeAt(i - 1) === 0x2f /* "/" */) i--; + return i === value.length ? value : value.slice(0, i); +} + +/** + * Default fetcher: `GET /api/combos` with bearer auth + + * AbortController timeout. Accepts both the `{combos: [...]}` envelope the + * gateway emits today and a bare-array envelope (defensive — keeps the + * plugin working if a future OmniRoute build trims the wrapper). + * + * Differences from `defaultOmniRouteModelsFetcher`: + * - URL is `/api/combos`, NOT `/v1/combos`. The `/v1/...` namespace is the + * OpenAI-compatible surface (chat completions, models); combo discovery + * lives on the management plane under `/api/...`. We tolerate both + * `https://host` and `https://host/v1` baseURL forms by stripping the + * trailing `/v1` segment before appending `/api/combos`. + * - Combos endpoint requires a management-scoped API key when + * `REQUIRE_API_KEY` is enabled. We don't enforce that here; the + * gateway returns 401/403 with an actionable error which we propagate. + * + * Anything that isn't an object with a string `id` is filtered out silently. + */ +export const defaultOmniRouteCombosFetcher: OmniRouteCombosFetcher = async ( + baseURL, + apiKey, + timeoutMs = 10_000 +) => { + if (!apiKey) throw new Error("[omniroute-v2] apiKey required to fetch /api/combos"); + if (!baseURL) throw new Error("[omniroute-v2] baseURL required to fetch /api/combos"); + + // Strip trailing slashes, then strip a trailing `/v1` so we land on the + // management plane. Models live under `/v1/models`; combos live under + // `/api/combos` from the same gateway root. + const trimmed = trimTrailingSlashes(baseURL); + const root = trimmed.replace(/\/v\d+$/, ""); + const url = `${root}/api/combos`; + + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), timeoutMs); + try { + const res = await fetch(url, { + method: "GET", + headers: { + Authorization: `Bearer ${apiKey}`, + Accept: "application/json", + }, + signal: controller.signal, + }); + if (!res.ok) { + throw new Error(`[omniroute-v2] GET ${url} failed: ${res.status} ${res.statusText}`); + } + const body = (await res.json()) as unknown; + const rawList: unknown[] = Array.isArray(body) + ? body + : body && typeof body === "object" && Array.isArray((body as { combos?: unknown }).combos) + ? ((body as { combos: unknown[] }).combos as unknown[]) + : []; + const out: OmniRouteRawCombo[] = []; + for (const r of rawList) { + if (r && typeof r === "object" && typeof (r as { id?: unknown }).id === "string") { + out.push(r as OmniRouteRawCombo); + } + } + return out; + } finally { + clearTimeout(timer); + } +}; + +/** + * Map a raw combo entry → `ModelV2` by computing the lowest-common-denominator + * (LCD) of its underlying member models. The LCD policy is the only way to + * surface a single capability vector to OpenCode without lying: if any member + * lacks a capability, the combo as a whole cannot guarantee it. + * + * LCD rules: + * - `limit.context` = `min(...members.context_length)`. + * - `limit.output` = `min(...members.max_output_tokens)`. + * - `limit.input` = `min(...members.max_input_tokens)` ONLY when every + * member declares one (ModelV2.limit.input is optional — better to + * omit than to fabricate a min over partial data). + * - `capabilities.toolcall` / `reasoning` / `attachment` / `temperature`: + * `every(member ⇒ supports?)`. The `reasoning` axis ORs across + * `reasoning` and `thinking` per member before AND-ing across the + * combo (mirrors `mapRawModelToModelV2`). The `attachment` axis ORs + * across `attachment` and `vision` per member. The `temperature` axis + * uses default-true semantics: a member supports temperature unless + * it explicitly declares `temperature: false`. + * - `capabilities.input.*` / `output.*`: flattened AND across members' + * modality flags. Missing arrays default to `["text"]` (same default + * as `mapRawModelToModelV2`). + * + * Defensive: empty members array → ALL capabilities `false`, limits zero. + * That's an intentional safety posture (you can't route through an empty + * combo, so OC should grey it out in the picker). + * + * Spec mapping: `cost` zeroed; `status = "active"`; + * `release_date = combo.release_date ?? ""`; + * `api = LCD (all-anthropic else openai-compatible)`; + * `name = combo.name ?? combo.id`. + * + * @param combo Raw `/api/combos` entry. + * @param members Raw `/v1/models` entries for THIS combo's member ids. + * Caller resolves `combo.models[].model` ids; unknown ids + * are silently dropped before this call. + * @param providerId OpenCode provider id (multi-instance aware). + * @param baseURL Resolved gateway base URL for ModelV2.api.url. + */ +export function mapComboToModelV2( + combo: OmniRouteRawCombo, + members: OmniRouteRawModelEntry[], + providerId: string, + baseURL: string, + apiFormat?: ApiFormatV2 +): ModelV2 { + // `every` over an empty array returns true (would lie about an empty + // combo's capabilities) — short-circuit to all-false when no members. + const hasMembers = members.length > 0; + + const memberInMods = members.map((m) => new Set(m.input_modalities ?? ["text"])); + const memberOutMods = members.map((m) => new Set(m.output_modalities ?? ["text"])); + + const modalityAllHave = (sets: Array>, key: string): boolean => + hasMembers && sets.every((s) => s.has(key)); + + const contextValues = members + .map((m) => m.context_length) + .filter((v): v is number => typeof v === "number" && v > 0); + const outputValues = members + .map((m) => m.max_output_tokens) + .filter((v): v is number => typeof v === "number" && v > 0); + const inputValues = members + .map((m) => m.max_input_tokens) + .filter((v): v is number => typeof v === "number" && v > 0); + + const everyDeclaresInput = hasMembers && inputValues.length === members.length; + + const capabilities: ModelV2["capabilities"] = { + temperature: + hasMembers && members.every((m) => (m.capabilities?.temperature ?? true) !== false), + reasoning: + hasMembers && + members.every((m) => Boolean(m.capabilities?.reasoning || m.capabilities?.thinking)), + attachment: + hasMembers && + members.every((m) => Boolean(m.capabilities?.attachment ?? m.capabilities?.vision ?? false)), + toolcall: hasMembers && members.every((m) => Boolean(m.capabilities?.tool_calling ?? false)), + input: { + text: modalityAllHave(memberInMods, "text"), + audio: modalityAllHave(memberInMods, "audio"), + image: modalityAllHave(memberInMods, "image"), + video: modalityAllHave(memberInMods, "video"), + pdf: modalityAllHave(memberInMods, "pdf"), + }, + output: { + text: modalityAllHave(memberOutMods, "text"), + audio: modalityAllHave(memberOutMods, "audio"), + image: modalityAllHave(memberOutMods, "image"), + video: modalityAllHave(memberOutMods, "video"), + pdf: modalityAllHave(memberOutMods, "pdf"), + }, + interleaved: hasMembers && members.every((m) => Boolean(m.capabilities?.thinking)), + }; + + // Combos span multiple providers. Use Anthropic format only when ALL + // members resolve to Anthropic — otherwise fall back to OpenAI-compat + // (lowest common denominator that every upstream understands). + const comboApiBlock = (() => { + if (!hasMembers) return resolveApiBlockV2(combo.id, baseURL, apiFormat); + const allAnthropic = members.every( + (m) => resolveApiBlockV2(m.id, baseURL, apiFormat).id === "anthropic" + ); + return allAnthropic + ? resolveApiBlockV2(members[0].id, baseURL, apiFormat) + : resolveApiBlockV2(combo.id, baseURL, apiFormat); + })(); + + return { + id: combo.id, + providerID: providerId, + api: comboApiBlock, + name: combo.name && combo.name.trim().length > 0 ? combo.name : combo.id, + capabilities, + cost: { + input: 0, + output: 0, + cache: { read: 0, write: 0 }, + }, + limit: { + context: + typeof combo.computed_context_length === "number" && combo.computed_context_length > 0 + ? combo.computed_context_length + : contextValues.length > 0 + ? Math.min(...contextValues) + : 0, + ...(everyDeclaresInput ? { input: Math.min(...inputValues) } : {}), + output: outputValues.length > 0 ? Math.min(...outputValues) : 0, + }, + status: "active", + options: {}, + headers: {}, + release_date: combo.release_date ?? "", + }; +} diff --git a/@omniroute/opencode-plugin-v2/src/shared/enrich.ts b/@omniroute/opencode-plugin-v2/src/shared/enrich.ts new file mode 100644 index 0000000000..1eec8c61c7 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/shared/enrich.ts @@ -0,0 +1,606 @@ +import type { Model as ModelV2 } from "@opencode-ai/sdk/v2"; +import { buildModelDisplayName } from "./naming.js"; +import type { FreeModelFreeType } from "./naming.js"; + +export interface OmniRouteEnrichmentEntry { + /** Human-readable display name. Replaces ModelV2.name when present. */ + name?: string; + /** Per-million-token cost overlay onto ModelV2.cost. */ + pricing?: { + input?: number; + output?: number; + cacheRead?: number; + cacheWrite?: number; + }; + /** + * Provider alias prefix seen in `/v1/models` ids (e.g. `cc`, `gemini`). + * Populated by `defaultOmniRouteEnrichmentFetcher` from + * `/api/pricing/models` keys. Drives the `usableOnly` alias↔canonical + * resolution. + */ + providerAlias?: string; + /** + * Canonical provider id used by `/api/providers` connections (e.g. + * `claude`, `gemini`, `kiro`). Populated from the per-provider + * `entry.id` field inside `/api/pricing/models`. + */ + providerCanonical?: string; + /** + * Human-readable upstream provider label (e.g. `Claude`, `Kiro`, + * `Windsurf`, `GitHub Models`). Populated from the per-provider + * `entry.name` field inside `/api/pricing/models`. Used by the + * `providerTag` feature to suffix `ModelV2.name` with the routing + * destination so the OC TUI picker can differentiate the same + * model id sold through different upstream connections. + */ + providerDisplayName?: string; + /** Free-model budget type (from freeModelCatalog). */ + freeType?: FreeModelFreeType; + /** Monthly token budget for recurring free models. */ + monthlyTokens?: number; + /** Credit token budget for credit-based free models. */ + creditTokens?: number; +} + +/** Map keyed by full model id (possibly namespaced, e.g. `cc/claude-sonnet-4-6`). */ +export type OmniRouteEnrichmentMap = Map; + +/** + * Reverse-index the enrichment map from `providerCanonical → providerAlias`. + * + * OmniRoute's `/api/pricing/models` is keyed by short ALIAS (`cc`, `cx`, + * `pol`). But `/v1/models` exposes some models a SECOND time under their + * CANONICAL name (`claude/claude-opus-4-7`, `codex/gpt-5.5`, + * `pollinations/midjourney`). Without a reverse map, those canonical + * rows miss enrichment entirely and surface as raw ids in the picker. + * + * Built once per refresh from the enrichment entries themselves — no + * hardcoded registry. Only records `canonical → alias` mappings when + * both are present AND distinct (skips slots where alias === canonical + * like `kiro`). + */ +export function buildCanonicalToAliasMap( + enrichment: OmniRouteEnrichmentMap | undefined +): Map { + const out = new Map(); + if (!enrichment) return out; + for (const entry of enrichment.values()) { + const alias = typeof entry.providerAlias === "string" ? entry.providerAlias.trim() : ""; + const canonical = + typeof entry.providerCanonical === "string" ? entry.providerCanonical.trim() : ""; + if (alias.length === 0 || canonical.length === 0) continue; + if (alias === canonical) continue; + if (!out.has(canonical)) out.set(canonical, alias); + } + return out; +} + +/** + * Enrichment lookup with alias-fallback chain. + * + * Resolution order (first hit wins): + * + * 1. `enrichment.get(rawId)` — direct hit on `/` or + * bare id (the fetcher writes under both forms). + * 2. If `rawId` is `/` and `canonicalToAlias` has + * a mapping for `canonical`, try `/`. This rescues + * duplicate rows like `claude/claude-opus-4-7` (canonical) when + * enrichment only indexed under `cc/claude-opus-4-7` (alias). + * 3. Bare `` as a last resort. Already covered by step 1 in + * practice (fetcher writes bare keys), but kept defensive. + * + * Returns `undefined` when no lookup hits. + */ +export function lookupEnrichment( + rawId: string, + enrichment: OmniRouteEnrichmentMap | undefined, + canonicalToAlias: Map +): OmniRouteEnrichmentEntry | undefined { + if (!enrichment) return undefined; + const direct = enrichment.get(rawId); + if (direct) return direct; + const slash = rawId.indexOf("/"); + if (slash > 0) { + const prefix = rawId.slice(0, slash); + const modelId = rawId.slice(slash + 1); + const alias = canonicalToAlias.get(prefix); + if (alias && alias !== prefix) { + const viaAlias = enrichment.get(`${alias}/${modelId}`); + if (viaAlias) return viaAlias; + } + const bare = enrichment.get(modelId); + if (bare) return bare; + } + return undefined; +} + +/** + * Pre-pass: detect raw rows that are the CANONICAL twin of an ALIAS row + * already in the catalog. Returns the set of canonical-keyed ids to skip + * during the raw-model loop so each model surfaces exactly once under + * its enriched alias key. + * + * Example: `/v1/models` returns BOTH `cc/claude-opus-4-7` and + * `claude/claude-opus-4-7`. The former is enriched (alias `cc` exists + * in `/api/pricing/models`); the latter is raw. We keep `cc/...` and + * drop `claude/...`. + * + * Built once per refresh. Cheap — O(M) where M = raw model count. + */ +export function canonicalDedupSet( + rawModels: ReadonlyArray<{ id: string }>, + canonicalToAlias: Map +): Set { + const drop = new Set(); + if (canonicalToAlias.size === 0) return drop; + // Index every alias key present in the raw catalog. + const aliasKeys = new Set(); + for (const m of rawModels) { + if (typeof m.id === "string" && m.id.length > 0) aliasKeys.add(m.id); + } + for (const m of rawModels) { + if (typeof m.id !== "string" || m.id.length === 0) continue; + const slash = m.id.indexOf("/"); + if (slash <= 0) continue; + const prefix = m.id.slice(0, slash); + const modelId = m.id.slice(slash + 1); + const alias = canonicalToAlias.get(prefix); + if (!alias || alias === prefix) continue; + // Canonical row only gets suppressed if the alias row actually + // exists — otherwise we'd hide the model entirely. + if (aliasKeys.has(`${alias}/${modelId}`)) drop.add(m.id); + } + return drop; +} + +/** + * Build a per-alias index of enrichment metadata so we can render the + * provider prefix even for raw models that don't have their own + * curated `/api/pricing/models` entry. + * + * Real example: OmniRoute's `pricing['cohere']` slot lists 10 curated + * models but `/v1/models` also returns `cohere/rerank-multilingual-v3.0` + * and `cohere/rerank-v4.0-fast` (not in the curated 10). Without this + * index, those rows surface in the picker as `cohere/...` with no + * `Cohere - ` prefix because the per-model enrichment lookup misses. + * + * This index records the first non-empty `providerDisplayName` seen + * for each alias, plus the alias itself. Callers use it to synthesize + * a minimal `OmniRouteEnrichmentEntry` whenever the direct lookup + * misses but the raw id's prefix matches a known alias. + * + * Built once per refresh; first-wins on duplicate alias (matches + * `buildCanonicalToAliasMap` semantics). + */ +export function buildAliasIndex( + enrichment: OmniRouteEnrichmentMap | undefined +): Map { + const out = new Map(); + if (!enrichment) return out; + for (const entry of enrichment.values()) { + const alias = typeof entry.providerAlias === "string" ? entry.providerAlias.trim() : ""; + if (alias.length === 0) continue; + if (out.has(alias)) { + // First-wins, but upgrade to the first entry that carries a + // non-empty providerDisplayName so the prefix renders nicely. + const existing = out.get(alias); + if ( + existing && + (!existing.providerDisplayName || existing.providerDisplayName.trim().length === 0) && + typeof entry.providerDisplayName === "string" && + entry.providerDisplayName.trim().length > 0 + ) { + out.set(alias, entry); + } + continue; + } + out.set(alias, entry); + } + return out; +} + +/** + * Resolve a synthesised enrichment entry for `applyProviderTag` / + * `shortProviderLabel` consumption, combining two sources: + * + * 1. The direct per-model enrichment match (if present). + * 2. A per-alias fallback derived from `buildAliasIndex` — covers raw + * ids whose prefix matches a known alias but the specific model + * id wasn't curated in `/api/pricing/models`. Example: + * `cohere/rerank-multilingual-v3.0` falls back to the cohere slot's + * `providerDisplayName='Cohere'` even though that specific id + * isn't in the curated 10-model list. + * + * Returns `undefined` when neither source surfaces an alias. + * + * NOTE: this function is read-only over its inputs; it never mutates + * the underlying `direct` entry. When it falls back to the alias + * index, it constructs a fresh minimal entry exposing only the + * provider-prefix fields (`providerAlias`, `providerCanonical`, + * `providerDisplayName`). Other fields (name, pricing) are explicitly + * left undefined so `applyEnrichment` won't accidentally overwrite a + * model name with the alias-slot label. + */ +export function resolveProviderTagEntry( + rawId: string, + direct: OmniRouteEnrichmentEntry | undefined, + aliasIndex: Map, + canonicalToAlias?: Map +): OmniRouteEnrichmentEntry | undefined { + if (direct) { + const alias = typeof direct.providerAlias === "string" ? direct.providerAlias.trim() : ""; + const display = + typeof direct.providerDisplayName === "string" ? direct.providerDisplayName.trim() : ""; + if (alias.length > 0 || display.length > 0) return direct; + } + const slash = rawId.indexOf("/"); + if (slash <= 0) return direct; + const prefix = rawId.slice(0, slash); + // 1. Direct alias lookup (`cohere/...` → cohere slot keyed by alias=cohere). + let fromAlias = aliasIndex.get(prefix); + // 2. Canonical fallback (`pollinations/...` → look up via alias `pol`). + if (!fromAlias && canonicalToAlias) { + const alias = canonicalToAlias.get(prefix); + if (alias) fromAlias = aliasIndex.get(alias); + } + if (!fromAlias) return direct; + // Synthesize: borrow only the provider-prefix metadata. + return { + providerAlias: fromAlias.providerAlias, + providerCanonical: fromAlias.providerCanonical, + providerDisplayName: fromAlias.providerDisplayName, + }; +} + +/** + * Fetcher contract: resolves the enrichment overlay (display names + + * pricing + free-tier budgets) from a running OmniRoute instance. + */ +/** + * Reports a source that could not be read. Enrichment stays best-effort, but + * a caller that swallows this loses display names, provider tags, canonical + * dedupe and pricing with no way to tell why. + */ +export type OmniRouteEnrichmentSourceError = (endpoint: string, reason: string) => void; + +export type OmniRouteEnrichmentFetcher = ( + baseURL: string, + apiKey: string, + timeoutMs?: number, + onSourceError?: OmniRouteEnrichmentSourceError +) => Promise; + +function trimTrailingSlashes(value: string): string { + let i = value.length; + while (i > 0 && value.charCodeAt(i - 1) === 0x2f /* "/" */) i--; + return i === value.length ? value : value.slice(0, i); +} + +/** + * Default enrichment fetcher — pulls nice display names from + * `GET /api/pricing/models` and merges per-million-token pricing from + * `GET /api/pricing` (the actual pricing source — `/api/pricing/models` is + * a catalog endpoint whose entries are `{id, name, custom}` only). + * + * `/api/pricing/models` shape (catalog): + * - `{ [providerAlias]: { id, alias, name, models: [{ id, name, custom }] } }` + * + * `/api/pricing` shape (pricing only): + * - `{ [providerAlias]: { [modelId]: { input, output, cached, reasoning, cache_creation } } }` + * where values are USD per million tokens. + * + * The two responses are joined on `(providerAlias, modelId)` and the merged + * entries are stored under both `${providerAlias}/${modelId}` and bare + * `${modelId}` keys so downstream lookups against either form succeed. + * + * Soft-fails (returns whatever was collected) on non-2xx or parse errors; + * the two fetches are independent so one missing source still surfaces the + * other. A third best-effort fetch attaches free-tier budgets from + * `/api/free-tier/summary`. + * + * Ported from the v1 plugin (`index.ts:1906-2106`); the shared logger is + * the only intentional difference (no plugin-contract dependency here). + */ +export const defaultOmniRouteEnrichmentFetcher: OmniRouteEnrichmentFetcher = async ( + baseURL, + apiKey, + timeoutMs = 10_000, + onSourceError +) => { + const report = (endpoint: string, reason: unknown): void => { + onSourceError?.(endpoint, reason instanceof Error ? reason.message : String(reason)); + }; + const out: OmniRouteEnrichmentMap = new Map(); + if (!baseURL || !apiKey) return out; + const root = trimTrailingSlashes(baseURL.replace(/\/v1\/?$/, "")); + const headers = { + Authorization: `Bearer ${apiKey}`, + Accept: "application/json", + }; + + // 1. Catalog with nice display names. + const catalogAc = new AbortController(); + const catalogTimer = setTimeout(() => catalogAc.abort(), timeoutMs); + let catalogStatus = 0; + try { + const res = await fetch(`${root}/api/pricing/models`, { + method: "GET", + headers, + signal: catalogAc.signal, + }); + catalogStatus = res.status; + if (res.ok) { + const body = (await res.json()) as unknown; + const providers = + (body as { providers?: Record })?.providers ?? + (body as Record); + if (providers && typeof providers === "object") { + for (const [providerAlias, slot] of Object.entries(providers)) { + if (!slot || typeof slot !== "object") continue; + const models = (slot as { models?: unknown[] }).models; + if (!Array.isArray(models)) continue; + const canonicalRaw = (slot as { id?: unknown }).id; + const providerCanonical = + typeof canonicalRaw === "string" && canonicalRaw.length > 0 + ? canonicalRaw + : providerAlias; + const slotNameRaw = (slot as { name?: unknown }).name; + const providerDisplayName = + typeof slotNameRaw === "string" && slotNameRaw.trim().length > 0 + ? slotNameRaw.trim() + : undefined; + for (const m of models) { + if (!m || typeof m !== "object") continue; + const id = (m as { id?: unknown }).id; + if (typeof id !== "string" || id.length === 0) continue; + const name = (m as { name?: unknown }).name; + const entry: OmniRouteEnrichmentEntry = { + providerAlias, + providerCanonical, + }; + if (providerDisplayName) entry.providerDisplayName = providerDisplayName; + if (typeof name === "string" && name.trim().length > 0) entry.name = name; + const namespaced = `${providerAlias}/${id}`; + if (!out.has(namespaced)) out.set(namespaced, entry); + // The bare id is a fallback for ids that arrive unnamespaced. It + // gets its OWN copy: sharing the object would let a later write + // for one provider — a price, typically — land on another + // provider's entry that happens to sell the same model id. + if (!out.has(id)) out.set(id, { ...entry }); + } + } + } + } + } catch (err) { + // Network error, timeout, abort: nothing collected from THIS source, but + // the pricing fetch below may still succeed — let it try, then decide at + // the end whether the whole overlay failed (see the throw below). + report("/api/pricing/models", err); + catalogStatus = -1; + } finally { + clearTimeout(catalogTimer); + } + if ( + catalogStatus !== 0 && + catalogStatus !== -1 && + (catalogStatus < 200 || catalogStatus >= 300) + ) { + report("/api/pricing/models", `HTTP ${catalogStatus}`); + } + + // 2. Pricing values from /api/pricing. + const priceAc = new AbortController(); + const priceTimer = setTimeout(() => priceAc.abort(), timeoutMs); + let priceStatus = 0; + try { + const res = await fetch(`${root}/api/pricing`, { + method: "GET", + headers, + signal: priceAc.signal, + }); + priceStatus = res.status; + if (res.ok) { + const body = (await res.json()) as unknown; + if (body && typeof body === "object" && !Array.isArray(body)) { + for (const [providerAlias, slot] of Object.entries(body as Record)) { + if (!slot || typeof slot !== "object" || Array.isArray(slot)) continue; + for (const [modelId, raw] of Object.entries(slot as Record)) { + if (!raw || typeof raw !== "object") continue; + const p = raw as Record; + const parsed: NonNullable = {}; + if (typeof p.input === "number") parsed.input = p.input; + if (typeof p.output === "number") parsed.output = p.output; + const cacheRead = + typeof p.cached === "number" + ? p.cached + : typeof p.cacheRead === "number" + ? p.cacheRead + : undefined; + if (typeof cacheRead === "number") parsed.cacheRead = cacheRead; + const cacheWrite = + typeof p.cache_creation === "number" + ? p.cache_creation + : typeof p.cacheWrite === "number" + ? p.cacheWrite + : undefined; + if (typeof cacheWrite === "number") parsed.cacheWrite = cacheWrite; + if (Object.keys(parsed).length === 0) continue; + const namespaced = `${providerAlias}/${modelId}`; + const existingNs = out.get(namespaced); + if (existingNs) { + existingNs.pricing = { ...(existingNs.pricing ?? {}), ...parsed }; + } else { + out.set(namespaced, { pricing: parsed }); + } + const existingBare = out.get(modelId); + // Only the provider that owns the bare entry may price it. + // Otherwise the second provider selling the same model id + // overwrites the first one's price, and the picker shows a cost + // that belongs to a different connection. + const bareBelongsHere = + existingBare === undefined || existingBare.providerAlias === undefined + ? true + : existingBare.providerAlias === providerAlias; + if (bareBelongsHere) { + if (existingBare) { + existingBare.pricing = { ...(existingBare.pricing ?? {}), ...parsed }; + } else { + out.set(modelId, { pricing: parsed }); + } + } + } + } + } + } + } catch (err) { + // Same as above: report, mark this source failed, let the remaining + // sources try before deciding. + report("/api/pricing", err); + priceStatus = -1; + } finally { + clearTimeout(priceTimer); + } + if (priceStatus !== 0 && priceStatus !== -1 && (priceStatus < 200 || priceStatus >= 300)) { + report("/api/pricing", `HTTP ${priceStatus}`); + } + + // 3. Free model budgets from /api/free-tier/summary (best-effort). + const freeAc = new AbortController(); + const freeTimer = setTimeout(() => freeAc.abort(), timeoutMs); + let freeStatus = 0; + try { + const res = await fetch(`${root}/api/free-tier/summary`, { + method: "GET", + headers, + signal: freeAc.signal, + }); + freeStatus = res.status; + if (res.ok) { + const body = (await res.json()) as unknown; + const perModel: unknown[] = + body && typeof body === "object" && Array.isArray((body as { perModel?: unknown }).perModel) + ? ((body as { perModel: unknown[] }).perModel as unknown[]) + : Array.isArray(body) + ? (body as unknown[]) + : []; + for (const fm of perModel) { + if (!fm || typeof fm !== "object") continue; + const fmObj = fm as Record; + const provider = typeof fmObj.provider === "string" ? fmObj.provider : ""; + const modelId = typeof fmObj.modelId === "string" ? fmObj.modelId : ""; + const freeType = typeof fmObj.freeType === "string" ? fmObj.freeType : ""; + if (!modelId || !freeType) continue; + const monthlyTokens = + typeof fmObj.monthlyTokens === "number" ? fmObj.monthlyTokens : undefined; + const creditTokens = + typeof fmObj.creditTokens === "number" ? fmObj.creditTokens : undefined; + const displayName = typeof fmObj.displayName === "string" ? fmObj.displayName : ""; + const candidates = [ + `${provider}/${modelId}`, + modelId, + ...(displayName ? [displayName] : []), + ]; + for (const key of candidates) { + const entry = out.get(key); + if (entry) { + entry.freeType = freeType as FreeModelFreeType; + if (monthlyTokens !== undefined) entry.monthlyTokens = monthlyTokens; + if (creditTokens !== undefined) entry.creditTokens = creditTokens; + break; + } + } + } + } + } catch (err) { + report("/api/free-tier/summary", err); + // Soft-fail; free metadata is optional. + } finally { + clearTimeout(freeTimer); + } + if (freeStatus !== 0 && (freeStatus < 200 || freeStatus >= 300)) { + report("/api/free-tier/summary", `HTTP ${freeStatus}`); + } + + // A source that failed contributes nothing — but the overlay keeps its own + // memory per source: names collected while the catalog endpoint answered + // survive a later pricing outage, and prices collected while pricing + // answered survive a later catalog outage. Without this a single flapping + // source wipes the other source's good data on every refresh. So a failed + // catalog source throws (the caller keeps last-known) UNLESS the pricing + // source brought something on THIS call — then whatever was collected, + // names or prices, is the gateway's answer and ships as-is. (Status alone + // cannot decide: a 2xx pricing answer with zero priced models is still an + // answer, but it carries nothing to save the overlay with.) + const sourceFailed = (status: number): boolean => + status === -1 || (status !== 0 && (status < 200 || status >= 300)); + const catalogFailed = sourceFailed(catalogStatus); + const pricingBroughtSomething = !sourceFailed(priceStatus) && out.size > 0; + if (catalogFailed && !pricingBroughtSomething) { + throw new Error( + `enrichment catalog source failed (pricing/models: ${catalogStatus}, pricing: ${priceStatus})` + ); + } + + return out; +}; + +/** + * Apply enrichment overlay onto a ModelV2 entry. Mutates and returns the + * passed entry for convenience. + */ +/** What the caller knows about the entry that the overlay itself cannot tell. */ +export interface EnrichmentDisplayContext { + /** Combos never carry a provider tag: they route across providers. */ + isCombo?: boolean; + isAutoCombo?: boolean; + /** Set false to publish the bare display name, without the provider tag. */ + providerTag?: boolean; +} + +/** + * Fold the overlay into a mapped model: display name, provider tag, free-tier + * marker and budget, and pricing. + * + * The name is built rather than copied, because the gateway ships the parts + * separately — the pricing catalog gives a display name and an upstream + * provider label, the free-tier summary gives the budget. A picker showing + * `Claude - [Free] Sonnet 4.6 · 1M/mo` tells the user which connection serves + * the model and what it costs them; `claude-sonnet-4-6` tells them nothing. + */ +export function applyEnrichment( + model: ModelV2, + enrichment: OmniRouteEnrichmentEntry | undefined, + context: EnrichmentDisplayContext = {} +): ModelV2 { + if (!enrichment) return model; + const built = buildModelDisplayName({ + rawId: model.name && model.name.length > 0 ? model.name : model.id, + enrichmentName: enrichment.name, + providerAlias: context.providerTag === false ? undefined : enrichment.providerAlias, + providerDisplayName: context.providerTag === false ? undefined : enrichment.providerDisplayName, + isFree: enrichment.freeType !== undefined, + freeType: enrichment.freeType, + monthlyTokens: enrichment.monthlyTokens, + creditTokens: enrichment.creditTokens, + isCombo: context.isCombo, + isAutoCombo: context.isAutoCombo, + }); + if (built.trim().length > 0) { + model.name = built; + } + if (enrichment.pricing) { + if (typeof enrichment.pricing.input === "number") { + model.cost.input = enrichment.pricing.input; + } + if (typeof enrichment.pricing.output === "number") { + model.cost.output = enrichment.pricing.output; + } + if (typeof enrichment.pricing.cacheRead === "number") { + model.cost.cache.read = enrichment.pricing.cacheRead; + } + if (typeof enrichment.pricing.cacheWrite === "number") { + model.cost.cache.write = enrichment.pricing.cacheWrite; + } + } + return model; +} diff --git a/@omniroute/opencode-plugin-v2/src/shared/fingerprint.ts b/@omniroute/opencode-plugin-v2/src/shared/fingerprint.ts new file mode 100644 index 0000000000..58835956bb --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/shared/fingerprint.ts @@ -0,0 +1,127 @@ +import { createHash } from "node:crypto"; + +/** + * Fingerprint the CONTENT of a catalog snapshot (not endpoint/credential + * identity) so lazy refresh can reload-after-publish only when something + * actually changed. + * + * sha256 over sorted `id + "|" + (release_date ?? "")` lines for models + * plus sorted combo ids, joined with `\n`. Order-insensitive: two + * snapshots with the same entries in different order hash identically. + */ +export function catalogContentFingerprint( + models: { id: string; release_date?: string }[], + combos: { id: string }[], + autoCombos: { id: string }[] = [] +): string { + const modelLines = models + .map((m) => `${m.id}|${m.release_date ?? ""}`) + .sort() + .join("\n"); + const comboLines = combos + .map((c) => c.id) + .sort() + .join("\n"); + const autoLines = autoCombos + .map((c) => c.id) + .sort() + .join("\n"); + return createHash("sha256").update(`${modelLines}\n${comboLines}\n${autoLines}`).digest("hex"); +} + +/** + * Digest of the optional tier (auto-combos, provider connections, enrichment). + * The catalog fingerprint covers model and combo ids only, so an overlay that + * moves — a renamed model, a provider going unusable — leaves it unchanged. + * Reloading on every refresh instead would ask the host to rebuild its catalog + * once per TTL window for nothing. + */ +export function optionalTierFingerprint( + autoCombos: { id: string }[], + providers: { + id?: string; + name?: string; + testStatus?: string; + isActive?: boolean; + providerDisplayName?: string; + }[], + enrichment: + | Map< + string, + { + name?: string; + freeType?: string; + providerDisplayName?: string; + monthlyTokens?: number; + creditTokens?: number; + pricing?: Record; + } + > + | undefined, + combos: { id: string; name?: string; models?: unknown[] }[] = [] +): string { + const parts: string[] = []; + // Membership matters: a combo keeping its id while losing a member is a + // different combo to anyone picking it. + parts.push( + combos + .map((c) => c.id + "|" + (c.name ?? "") + "|" + String(c.models?.length ?? 0)) + .sort() + .join(",") + ); + parts.push( + autoCombos + .map((c) => c.id) + .sort() + .join(",") + ); + // A provider going quiet or getting renamed is as visible to the user as a + // price move: its activity flag and display name belong in the digest. + parts.push( + providers + .map( + (p) => + (p.id ?? p.name ?? "") + + ":" + + (p.testStatus ?? "") + + ":" + + String(p.isActive ?? "") + + ":" + + (p.providerDisplayName ?? "") + ) + .sort() + .join(",") + ); + if (enrichment !== undefined) { + const rows: string[] = []; + for (const [key, entry] of enrichment) { + // Pricing is part of what the user sees, so a price move must reach + // the picker without waiting for an id to change. + const price = entry.pricing + ? Object.entries(entry.pricing) + .map(([k, v]) => k + "=" + String(v ?? "")) + .sort() + .join(";") + : ""; + rows.push( + key + + "|" + + (entry.name ?? "") + + "|" + + (entry.freeType ?? "") + + "|" + + (entry.providerDisplayName ?? "") + + "|" + + String(entry.monthlyTokens ?? "") + + ";" + + String(entry.creditTokens ?? "") + + "|" + + price + ); + } + rows.sort(); + parts.push(String(enrichment.size)); + parts.push(rows.join("\n")); + } + return createHash("sha256").update(parts.join(" ")).digest("hex"); +} diff --git a/@omniroute/opencode-plugin-v2/src/shared/gemini.ts b/@omniroute/opencode-plugin-v2/src/shared/gemini.ts new file mode 100644 index 0000000000..2fb8b42c71 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/shared/gemini.ts @@ -0,0 +1,166 @@ +/** + * Gemini rejects several standard JSON-Schema keywords in tool declarations + * and answers `400 INVALID_ARGUMENT` for the whole request when it meets one. + * The keywords carry no meaning Gemini would honour anyway, so stripping them + * costs nothing and is what keeps a tool-calling chain alive. + */ +/** + * Keywords Gemini rejects outright. `$ref` is deliberately NOT here: it + * cannot be stripped without turning the schema into "accept anything", so + * tools carrying one are forwarded untouched (see below). `ref` is not a + * JSON Schema keyword at all, and stripping it by name destroys a legitimate + * tool parameter called `ref` — a walker that cannot tell a keyword from a + * property name mangles the schema it was meant to repair. + */ +const REJECTED_KEYWORDS = new Set(["$schema", "additionalProperties"]); + +/** Keys whose value is itself a schema. */ +const SCHEMA_VALUE_KEYS = [ + "items", + "additionalItems", + "contains", + "not", + "if", + "then", + "else", + "propertyNames", + "contentSchema", + "unevaluatedItems", + "unevaluatedProperties", +]; +/** Keys whose value maps arbitrary NAMES to schemas — never keyword space. */ +const SCHEMA_MAP_KEYS = [ + "properties", + "patternProperties", + "$defs", + "definitions", + "dependentSchemas", + "dependencies", +]; +/** Keys whose value is a list of schemas. */ +const SCHEMA_LIST_KEYS = ["allOf", "anyOf", "oneOf", "prefixItems"]; + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +/** True when any schema in the tree carries a `$ref` we cannot resolve. */ +function hasUnresolvableRef(node: unknown): boolean { + if (Array.isArray(node)) return node.some(hasUnresolvableRef); + if (!isRecord(node)) return false; + if ("$ref" in node) return true; + for (const key of SCHEMA_VALUE_KEYS) if (hasUnresolvableRef(node[key])) return true; + // (arrays are handled by the Array branch at the top of this function) + for (const key of SCHEMA_LIST_KEYS) if (hasUnresolvableRef(node[key])) return true; + for (const key of SCHEMA_MAP_KEYS) { + const map = node[key]; + if (isRecord(map) && Object.values(map).some(hasUnresolvableRef)) return true; + } + return false; +} + +/** + * Strip the rejected keywords in place, walking only the positions where a + * schema can appear. Property names are never treated as keywords, so a tool + * whose parameter happens to be called `additionalProperties` keeps it. + * Returns whether anything was removed. + */ +function stripAtSchemaPositions(node: Record): boolean { + let changed = false; + for (const keyword of REJECTED_KEYWORDS) { + if (keyword in node) { + delete node[keyword]; + changed = true; + } + } + for (const key of SCHEMA_VALUE_KEYS) { + const child = node[key]; + if (isRecord(child)) { + changed = stripAtSchemaPositions(child) || changed; + continue; + } + // `items` also takes the tuple form: an array of schemas, one per position. + if (Array.isArray(child)) { + for (const item of child) { + if (isRecord(item)) changed = stripAtSchemaPositions(item) || changed; + } + } + } + for (const key of SCHEMA_LIST_KEYS) { + const list = node[key]; + if (Array.isArray(list)) { + for (const child of list) { + if (isRecord(child)) changed = stripAtSchemaPositions(child) || changed; + } + } + } + for (const key of SCHEMA_MAP_KEYS) { + const map = node[key]; + if (!isRecord(map)) continue; + for (const child of Object.values(map)) { + if (isRecord(child)) changed = stripAtSchemaPositions(child) || changed; + } + } + return changed; +} + +/** + * Families Google actually ships, anchored on the last path segment. A plain + * substring test also claims `gemini-compatible-proxy` and `my-gemini-wrapper` + * — and since the sanitiser removes keywords, a false positive is not free. + */ +const GEMINI_MODEL_ID = + /^gemini(?:[-_.](?:\d|pro|flash|ultra|nano|exp|thinking|embedding|live|imagen)|$)/i; + +/** + * True for the routing forms a Gemini model reaches a gateway under — bare + * (`gemini-2.5-flash`), canonical (`models/gemini-1.5-pro`) and prefixed + * (`google-vertex/gemini-2.0`). + */ +export function isGeminiModelId(modelId: unknown): boolean { + if (typeof modelId !== "string") return false; + const segment = modelId.split("/").pop() ?? ""; + return GEMINI_MODEL_ID.test(segment); +} + +/** The subset of an AI SDK tool declaration this module reads. */ +export interface ToolWithInputSchema { + readonly type?: string; + readonly inputSchema?: unknown; + readonly [key: string]: unknown; +} + +/** + * Return a copy of `tools` whose input schemas are free of the keywords Gemini + * rejects, or `undefined` when there was nothing to strip — which lets the + * caller forward the original array and skip the clone entirely. + * + * Tools this module cannot read (provider-defined tools, entries without an + * object schema) are carried through unchanged rather than dropped: a tool the + * sanitiser does not understand is still a tool the model needs. + */ +export function sanitizeToolInputSchemas( + tools: readonly T[] | undefined +): T[] | undefined { + if (tools === undefined || tools.length === 0) return undefined; + let changed = false; + const out = tools.map((tool) => { + if (!isRecord(tool.inputSchema)) return tool; + // A schema carrying something uncloneable is not worth failing a request + // over: forward the tool untouched and let the model answer. + // A `$ref` cannot be stripped without turning the schema into "anything + // goes", and cannot be resolved here. Forward the tool untouched and let + // the gateway answer rather than silently widen what the model may send. + if (hasUnresolvableRef(tool.inputSchema)) return tool; + let schema: Record; + try { + schema = structuredClone(tool.inputSchema) as Record; + } catch { + return tool; + } + if (!stripAtSchemaPositions(schema)) return tool; + changed = true; + return { ...tool, inputSchema: schema }; + }); + return changed ? out : undefined; +} diff --git a/@omniroute/opencode-plugin-v2/src/shared/index.ts b/@omniroute/opencode-plugin-v2/src/shared/index.ts new file mode 100644 index 0000000000..d65d093abb --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/shared/index.ts @@ -0,0 +1,9 @@ +export * from "./models-map.js"; +export * from "./combos-map.js"; +export * from "./auto-combos.js"; +export * from "./naming.js"; +export * from "./enrich.js"; +export * from "./fingerprint.js"; +export * from "./logger.js"; +export * from "./usable.js"; +export * from "./gemini.js"; diff --git a/@omniroute/opencode-plugin-v2/src/shared/logger.ts b/@omniroute/opencode-plugin-v2/src/shared/logger.ts new file mode 100644 index 0000000000..2439ddd34c --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/shared/logger.ts @@ -0,0 +1,81 @@ +/** + * Namespaced leveled logger shared by the OmniRoute OpenCode packages. + * + * Levels: error < warn < info < debug. Default: warn. + * Ported from the v1 plugin (`logger.ts`) so both new packages share one + * sink instead of raw `console.warn` / `console.log` calls. + */ + +export type LogLevel = "error" | "warn" | "info" | "debug"; + +const LEVEL_ORDER: Record = { + error: 0, + warn: 1, + info: 2, + debug: 3, +}; + +const TAG = "[omniroute-plugin]"; + +function shouldLog(current: LogLevel, target: LogLevel): boolean { + return LEVEL_ORDER[current] >= LEVEL_ORDER[target]; +} + +let _level: LogLevel = "warn"; + +export function setLogLevel(level: LogLevel): void { + _level = level; +} + +export function getLogLevel(): LogLevel { + return _level; +} + +function fmt(level: LogLevel, msg: string, tag?: string): string { + const prefix = tag ? `${TAG}${tag}` : TAG; + return `${prefix} [${level.toUpperCase()}] ${msg}`; +} + +function buildLogger(getLevel: () => LogLevel) { + return { + error(msg: string, ...args: unknown[]): void { + if (shouldLog(getLevel(), "error")) console.error(fmt("error", msg), ...args); + }, + warn(msg: string, ...args: unknown[]): void { + if (shouldLog(getLevel(), "warn")) console.warn(fmt("warn", msg), ...args); + }, + info(msg: string, ...args: unknown[]): void { + if (shouldLog(getLevel(), "info")) console.warn(fmt("info", msg), ...args); + }, + debug(msg: string, ...args: unknown[]): void { + if (shouldLog(getLevel(), "debug")) console.warn(fmt("debug", msg), ...args); + }, + /** Always emit regardless of level (for critical init breadcrumbs). */ + always(msg: string, ...args: unknown[]): void { + console.warn(TAG, msg, ...args); + }, + + child(tag: string) { + return { + error: (msg: string, ...args: unknown[]) => + shouldLog(getLevel(), "error") && console.error(fmt("error", msg, tag), ...args), + warn: (msg: string, ...args: unknown[]) => + shouldLog(getLevel(), "warn") && console.warn(fmt("warn", msg, tag), ...args), + info: (msg: string, ...args: unknown[]) => + shouldLog(getLevel(), "info") && console.warn(fmt("info", msg, tag), ...args), + debug: (msg: string, ...args: unknown[]) => + shouldLog(getLevel(), "debug") && console.warn(fmt("debug", msg, tag), ...args), + }; + }, + }; +} + +export type Logger = ReturnType; + +/** Create an instance-scoped logger whose level cannot be changed by other instances. */ +export function createLogger(level: LogLevel): Logger { + return buildLogger(() => level); +} + +/** Backward-compatible module-global logger controlled by setLogLevel(). */ +export const logger: Logger = buildLogger(() => _level); diff --git a/@omniroute/opencode-plugin-v2/src/shared/models-map.ts b/@omniroute/opencode-plugin-v2/src/shared/models-map.ts new file mode 100644 index 0000000000..625e02f232 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/shared/models-map.ts @@ -0,0 +1,323 @@ +import type { Model as ModelV2 } from "@opencode-ai/sdk/v2"; +import { normaliseFreeLabel } from "./naming.js"; + +export interface OmniRouteRawModelEntry { + id: string; + object?: string; + owned_by?: string; + root?: string | null; + parent?: string | null; + context_length?: number; + max_input_tokens?: number; + max_output_tokens?: number; + input_modalities?: string[]; + output_modalities?: string[]; + capabilities?: { + tool_calling?: boolean; + reasoning?: boolean; + vision?: boolean; + thinking?: boolean; + attachment?: boolean; + structured_output?: boolean; + temperature?: boolean; + /** Runtime-learned or synced reasoning tiers (server-gated, blind-mapped). */ + effort_tiers?: string[]; + }; + release_date?: string; + last_updated?: string; + api_format?: string; +} + +/** + * Fetcher contract: returns the raw `/v1/models` entry list from a running + * OmniRoute instance. Surfaced as a dependency so unit tests can inject a + * stub without monkey-patching global `fetch`. + * + * Why we inline this instead of using `@omniroute/opencode-provider`'s + * `fetchLiveModels`: the sibling helper returns a stripped `{id, name, + * contextLength?}` shape that drops the `capabilities` / `*_modalities` / + * `max_*_tokens` blocks the mapping needs for ModelV2 pass-through. + */ +export type OmniRouteModelsFetcher = ( + baseURL: string, + apiKey: string, + timeoutMs?: number +) => Promise; + +/** + * Default fetcher: `GET /v1/models` with bearer auth + AbortController + * timeout. Accepts both the `{object:"list", data:[…]}` envelope OmniRoute + * emits today and a bare-array envelope (defensive — keeps the plugin + * working if a future OmniRoute build trims the wrapper). Anything that + * isn't an object with a string `id` is filtered out silently. + */ +export const defaultOmniRouteModelsFetcher: OmniRouteModelsFetcher = async ( + baseURL, + apiKey, + timeoutMs = 10_000 +) => { + if (!apiKey) throw new Error("[omniroute-v2] apiKey required to fetch /v1/models"); + if (!baseURL) throw new Error("[omniroute-v2] baseURL required to fetch /v1/models"); + + const trimmed = trimTrailingSlashes(baseURL); + // Tolerate both `https://host` and `https://host/v1` forms — the gateway + // exposes /v1/models either way; we just don't want a double `/v1/v1`. + const url = /\/v\d+$/.test(trimmed) ? `${trimmed}/models` : `${trimmed}/v1/models`; + + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), timeoutMs); + try { + const res = await fetch(url, { + method: "GET", + headers: { + Authorization: `Bearer ${apiKey}`, + Accept: "application/json", + }, + signal: controller.signal, + }); + if (!res.ok) { + throw new Error(`[omniroute-v2] GET ${url} failed: ${res.status} ${res.statusText}`); + } + const body = (await res.json()) as unknown; + const rawList: unknown[] = Array.isArray(body) + ? body + : body && typeof body === "object" && Array.isArray((body as { data?: unknown }).data) + ? ((body as { data: unknown[] }).data as unknown[]) + : []; + const out: OmniRouteRawModelEntry[] = []; + for (const r of rawList) { + if (r && typeof r === "object" && typeof (r as { id?: unknown }).id === "string") { + out.push(r as OmniRouteRawModelEntry); + } + } + return out; + } finally { + clearTimeout(timer); + } +}; + +// Manual trim helpers avoid polynomial-regex CodeQL warnings on +// user-supplied baseURL strings (string.replace(/\/+$/, "")). The same +// behaviour, no backtracking. +function trimTrailingSlashes(value: string): string { + let i = value.length; + while (i > 0 && value.charCodeAt(i - 1) === 0x2f /* "/" */) i--; + return i === value.length ? value : value.slice(0, i); +} + +/** + * Ensure a baseURL ends with `/v1` so the OpenAI-compat SDK constructs + * `/v1/chat/completions` correctly. The Anthropic SDK does NOT want `/v1` + * (it appends `/v1/messages` automatically), so callers should branch on + * format first. + */ +export function ensureV1Suffix(url: string): string { + const trimmed = trimTrailingSlashes(url); + return trimmed.endsWith("/v1") ? trimmed : `${trimmed}/v1`; +} + +export interface ApiFormatV2 { + allowAnthropic?: boolean; + anthropicModels?: string[]; + /** + * Deprecated v1 prefix list (default v1: + * `cc,claude,anthropic,kiro,kr`). Accepted for backward compatibility: + * prefix OR allowlist routes to anthropic, with a one-time deprecation + * warning pointing at `anthropicModels`. Prefer full IDs. + */ + anthropicPrefixes?: string[]; +} + +/** Default v1 prefix list, kept so copied v1 configs keep routing. */ +export const DEFAULT_ANTHROPIC_PREFIXES_V1 = ["cc", "claude", "anthropic", "kiro", "kr"]; + +const warnedPrefixLists = new Set(); + +function warnDeprecatedPrefixesOnce(prefixes: string[]): void { + const key = [...prefixes].sort().join(","); + if (warnedPrefixLists.has(key)) return; + warnedPrefixLists.add(key); + console.warn( + "[omniroute-plugin] [WARN] apiFormat.anthropicPrefixes is deprecated; convert to anthropicModels (full IDs)" + ); +} + +/** + * The Anthropic SDK block appends `/v1/messages` itself, so it needs the + * gateway root. A config carrying the `/v1` the OpenAI-compatible block wants + * would otherwise produce `/v1/v1/messages`. + */ +function stripV1Suffix(baseURL: string): string { + return baseURL.replace(/\/v1\/?$/, ""); +} + +/** + * Resolve the API block (id + url + npm package) for a given model id. + * + * v2 rule: a model routes to the Anthropic SDK block when + * `apiFormat.allowAnthropic === true` AND (its FULL id is allowlisted in + * `apiFormat.anthropicModels` OR its prefix is listed in the deprecated + * `apiFormat.anthropicPrefixes`, defaulting to the v1 list when prefixes + * are absent). The deprecated path warns once per prefix list. With + * neither allowlist nor prefix match, the model stays openai-compatible. + */ +export function resolveApiBlockV2( + modelId: string, + baseURL: string, + apiFormat?: ApiFormatV2 +): { id: string; url: string; npm: string } { + if (apiFormat?.allowAnthropic === true) { + if ((apiFormat.anthropicModels ?? []).includes(modelId)) { + return { + id: "anthropic", + url: stripV1Suffix(trimTrailingSlashes(baseURL)), + npm: "@ai-sdk/anthropic", + }; + } + const prefixes = apiFormat.anthropicPrefixes ?? DEFAULT_ANTHROPIC_PREFIXES_V1; + if (apiFormat.anthropicPrefixes !== undefined) warnDeprecatedPrefixesOnce(prefixes); + const slash = modelId.indexOf("/"); + const prefix = slash === -1 ? modelId : modelId.slice(0, slash); + if (prefixes.includes(prefix)) { + return { + id: "anthropic", + url: stripV1Suffix(trimTrailingSlashes(baseURL)), + npm: "@ai-sdk/anthropic", + }; + } + } + return { + id: "openai-compatible", + url: ensureV1Suffix(baseURL), + npm: "@ai-sdk/openai-compatible", + }; +} + +/** + * Map a raw `/v1/models` entry → `ModelV2` (the type @opencode-ai/sdk/v2 + * exports as `Model`, re-exported by @opencode-ai/plugin as `ModelV2`). + * + * ModelV2 requires a much richer shape than a flat record. Concretely it + * expects: + * - flat `id`, `name`, `providerID`, `api: {id,url,npm}` + * - nested `capabilities: { temperature, reasoning, attachment, toolcall, + * input:{text,audio,image,video,pdf}, output:{…}, interleaved }` + * - `cost: { input, output, cache:{read,write} }` (NOT optional) + * - `limit: { context, input?, output }` + * - `status: "alpha"|"beta"|"deprecated"|"active"`, `options:{}`, `headers:{}` + * - `release_date: string` + * + * Field adaptations: + * 1. Flat `tool_call` / `reasoning` / `attachment` / `modalities` + * top-level fields don't exist in ModelV2 — folded into + * `capabilities.{toolcall, reasoning, attachment, input.*, output.*}`. + * 2. `cost: undefined` is illegal (cost is required). OmniRoute doesn't + * surface pricing on /v1/models, so we emit a zeroed cost block. + * Downstream opencode reads this for display only — the live pricing + * is OmniRoute's responsibility at routing time. + * 3. `tool_call` → `toolcall` (ModelV2 field name; one word). + * 4. `attachment` maps from `capabilities.vision` per OmniRoute + * convention: vision = ability to receive image attachments. If the + * raw entry happens to expose an explicit `capabilities.attachment`, + * that wins. + * 5. `thinking` from OmniRoute has no 1:1 ModelV2 slot. We OR it into + * `reasoning` so thinking-only models still surface a non-false + * reasoning flag. + * 6. `last_updated` from OmniRoute has no ModelV2 slot — dropped. + * `release_date` lands in ModelV2.release_date with `""` fallback + * (the field is required as `string`). + * 7. `temperature: true` per OmniRoute convention (OpenAI-compat mode + * always supports the temperature knob). If a raw entry sets + * `capabilities.temperature` explicitly, that wins. + * 8. Input/output modality arrays: each known modality flips its boolean. + * Unknown strings (future OmniRoute additions) are ignored — when the + * server adds new modalities we can map them here without breaking + * existing entries. + * 9. `status: "active"` — OmniRoute doesn't tier models alpha/beta on + * /v1/models, and opencode needs a non-deprecated status to expose + * the model in the picker. If a future entry surfaces an explicit + * lifecycle hint we can map it then. + * 10. `options: {}` and `headers: {}` left empty — they're escape hatches + * for opencode users to attach per-model overrides; the provider + * plugin must not preempt them. + * 11. `limit.input` is OPTIONAL on ModelV2 (the `?` modifier). We only + * emit it when OmniRoute supplies `max_input_tokens` — keeps the + * shape clean for combo entries that only carry context_length. + */ +export function mapRawModelToModelV2( + raw: OmniRouteRawModelEntry, + ctx: { providerId: string; baseURL: string; apiFormat?: ApiFormatV2 } +): ModelV2 { + const caps = raw.capabilities ?? {}; + // effort_tiers loop: server-declared tiers become ModelV2 variants so the + // UI offers exactly the tiers OmniRoute vouches for (instead of opencode's + // invented [low, medium, high] fallback). Blind: filtering/exclusion rules + // live server-side. Absent/empty/malformed => key omitted ENTIRELY (an + // empty variants object would suppress opencode's fallback for this model). + const declaredTiers = Array.isArray(caps.effort_tiers) + ? caps.effort_tiers.filter((t): t is string => typeof t === "string" && t.length > 0) + : []; + const variants = + declaredTiers.length > 0 + ? Object.fromEntries(declaredTiers.map((tier) => [tier, { reasoningEffort: tier }])) + : undefined; + const inMods = new Set(raw.input_modalities ?? ["text"]); + const outMods = new Set(raw.output_modalities ?? ["text"]); + + return { + // OC's static-catalog reader parses the key on `/` to recover + // `(providerID, modelID)`. If the raw id is already provider-prefixed + // (e.g. `cc/claude-opus-4-7` from the `cc` Claude Code alias, or + // `nvidia/llama-3-70b` from a provider that ships prefixed ids), leave + // it as-is — double-prefixing breaks OC's lookup. Bare **combo** ids + // (`owned_by: "combo"`, e.g. `gpt-5.6-sol`) must also stay unprefixed: + // OpenCode looks up `-m /` as model id `` under + // the plugin provider. Other bare ids still prefix with + // `providerId` so credentials resolve as `(omniroute, model)`. + id: raw.id.includes("/") || raw.owned_by === "combo" ? raw.id : `${ctx.providerId}/${raw.id}`, + /** + * Display name. Falls back to raw.id when no enrichment is available; + * the caller overlays `/api/pricing/models` data via enrichment when + * the enrichment feature is enabled. + */ + name: normaliseFreeLabel(raw.id), + capabilities: { + temperature: caps.temperature ?? true, + reasoning: Boolean(caps.reasoning || caps.thinking), + attachment: Boolean(caps.attachment ?? caps.vision ?? false), + toolcall: Boolean(caps.tool_calling ?? false), + input: { + text: inMods.has("text"), + audio: inMods.has("audio"), + image: inMods.has("image"), + video: inMods.has("video"), + pdf: inMods.has("pdf"), + }, + output: { + text: outMods.has("text"), + audio: outMods.has("audio"), + image: outMods.has("image"), + video: outMods.has("video"), + pdf: outMods.has("pdf"), + }, + interleaved: Boolean(caps.thinking), + }, + cost: { + input: 0, + output: 0, + cache: { read: 0, write: 0 }, + }, + limit: { + context: typeof raw.context_length === "number" ? raw.context_length : 0, + ...(typeof raw.max_input_tokens === "number" ? { input: raw.max_input_tokens } : {}), + output: typeof raw.max_output_tokens === "number" ? raw.max_output_tokens : 0, + }, + ...(variants ? { variants } : {}), + status: "active", + options: {}, + headers: {}, + release_date: raw.release_date ?? "", + providerID: ctx.providerId, + api: resolveApiBlockV2(raw.id, ctx.baseURL, ctx.apiFormat), + }; +} diff --git a/@omniroute/opencode-plugin-v2/src/shared/naming.ts b/@omniroute/opencode-plugin-v2/src/shared/naming.ts new file mode 100644 index 0000000000..823809cf20 --- /dev/null +++ b/@omniroute/opencode-plugin-v2/src/shared/naming.ts @@ -0,0 +1,295 @@ +/** + * Universal model naming template for the OmniRoute plugin. + * + * Naming pipeline: + * [tag] + * + * [Free] - · ← free model + * Auto: (p) ← auto combo + * Combo: ← DB combo + * - ← regular model + */ + +// ── Constants ──────────────────────────────────────────────────────────── + +/** Separator between provider label and model display name. */ +export const PROVIDER_TAG_SEPARATOR = " - "; + +/** Threshold beyond which providerDisplayName is abbreviated. */ +const PROVIDER_LABEL_MAX_CHARS = 12; + +/** Aliases longer than this get title-case instead of UPPER. */ +const ALIAS_UPPER_MAX_CHARS = 5; + +// ── Auto Combo Types ───────────────────────────────────────────────────── + +export type AutoVariant = "coding" | "fast" | "cheap" | "offline" | "smart" | "lkgp"; + +export const AUTO_VARIANTS: AutoVariant[] = ["coding", "fast", "cheap", "offline", "smart", "lkgp"]; + +export const AUTO_VARIANT_DESCRIPTIONS: Record = { + default: "Best provider via scoring", + coding: "Quality-first for code tasks", + fast: "Latency-optimized routing", + cheap: "Cost-optimized routing", + offline: "Offline-friendly providers", + smart: "Quality-first with exploration", + lkgp: "Last-Known-Good-Provider routing", +}; + +// ── Free Model Types ───────────────────────────────────────────────────── + +export type FreeModelFreeType = + | "recurring-daily" + | "recurring-monthly" + | "recurring-credit" + | "one-time-initial" + | "keyless" + | "discontinued"; + +// ── Provider Label ──────────────────────────────────────────────────────── + +/** + * Title-case a long, lowercase-looking alias. + * `antigravity` → `Antigravity` + */ +function titleCaseAlias(alias: string): string { + if (alias.length === 0) return alias; + return alias.charAt(0).toUpperCase() + alias.slice(1).toLowerCase(); +} + +/** + * Pick the short label for an upstream provider. + * + * Rules: + * 1. Trim `providerDisplayName`. If ≤12 chars → use verbatim. + * 2. Alias ≤5 chars → UPPER(alias). Alias >5 → titleCase. + * 3. Neither → undefined. + */ +export function shortProviderLabel( + enrichment: { providerDisplayName?: string; providerAlias?: string } | undefined +): string | undefined { + if (!enrichment) return undefined; + const raw = + typeof enrichment.providerDisplayName === "string" ? enrichment.providerDisplayName.trim() : ""; + if (raw.length > 0 && raw.length <= PROVIDER_LABEL_MAX_CHARS) return raw; + const alias = typeof enrichment.providerAlias === "string" ? enrichment.providerAlias.trim() : ""; + if (alias.length > 0) { + return alias.length <= ALIAS_UPPER_MAX_CHARS ? alias.toUpperCase() : titleCaseAlias(alias); + } + // Long displayName with no alias to fall back on: keep the long label + // rather than dropping the provider prefix entirely. + return raw.length > 0 ? raw : undefined; +} + +// ── Free Label ──────────────────────────────────────────────────────────── + +/** + * Normalise display name so free-tier models get a consistent `[Free] ` prefix. + * + * "GPT-4.1 (Free)" → "[Free] GPT-4.1" + * "DeepSeek V4 Flash Free" → "[Free] DeepSeek V4 Flash" + * "Claude Opus 4.7" → "Claude Opus 4.7" (unchanged) + */ +export function normaliseFreeLabel(name: string): string { + // Bounded whitespace quantifiers ({0,8}/{1,8}) avoid the polynomial-ReDoS + // backtracking that unbounded \s* before an anchored \s*$ would allow on + // attacker-influenced display names. 8 covers any realistic label spacing. + const cleaned = name + .replace(/\s{0,8}\(free\)\s{0,8}$/i, "") + .replace(/[\s-]{1,8}free\s{0,8}$/i, "") + .trim(); + const wasFree = cleaned.length < name.trim().length; + if (!wasFree) return name; + return `[Free] ${cleaned}`; +} + +// ── Free Budget Formatting ──────────────────────────────────────────────── + +/** Scales, largest first, so the unit is chosen by descending magnitude. */ +const TOKEN_UNITS = [ + [1e9, "B"], + [1e6, "M"], + [1e3, "K"], +] as const; + +/** + * Format a token count as a short magnitude string: `25M`, `1.5K`, `999`. + * + * The unit has to be picked from the value that will actually be *printed*, + * not from the raw input. `toFixed(1)` rounds to the nearest tenth, so at the + * K scale 999_950 and above render as `1000.0` — and by then the M branch has + * already been skipped, producing `1000K` for a number that is `1M`. The same + * carry turns just under a billion into `1000M`. When the rounded value reaches + * the next scale, re-render at that scale instead. + */ +function fmtTokens(n: number): string { + for (let i = 0; i < TOKEN_UNITS.length; i++) { + const [scale, suffix] = TOKEN_UNITS[i]!; + if (n < scale) continue; + const value = Number((n / scale).toFixed(1)); + // `Number()` also drops a trailing `.0`, which the previous regex did. + if (value < 1000 || i === 0) return `${value}${suffix}`; + const [nextScale, nextSuffix] = TOKEN_UNITS[i - 1]!; + return `${Number((n / nextScale).toFixed(1))}${nextSuffix}`; + } + return String(n); +} + +/** + * Format a free model budget into a short human-readable suffix. + * + * recurring-daily → "25M tokens/day" + * recurring-monthly → "25M tokens/month" + * recurring-credit → "10M credits" + * one-time-initial → "1M credits (one-time)" + * keyless → "(keyless)" + * discontinued → "(discontinued)" + */ +export function formatFreeBudget(params: { + freeType: FreeModelFreeType; + monthlyTokens?: number; + creditTokens?: number; +}): string { + const { freeType, monthlyTokens = 0, creditTokens = 0 } = params; + + switch (freeType) { + case "recurring-daily": + return `${fmtTokens(monthlyTokens)} tokens/day`; + case "recurring-monthly": + return `${fmtTokens(monthlyTokens)} tokens/month`; + case "recurring-credit": + return `${fmtTokens(creditTokens)} credits`; + case "one-time-initial": + return `${fmtTokens(creditTokens)} credits (one-time)`; + case "keyless": + return "(keyless)"; + case "discontinued": + return "(discontinued)"; + default: + return ""; + } +} + +// ── Auto Combo Naming ───────────────────────────────────────────────────── + +/** + * Format auto combo display name. + * + * "Auto: Coding (4p)" + * "Auto: Default (6p)" + * "Auto" (no candidate count when unknown) + */ +export function formatAutoComboName( + variant: AutoVariant | undefined, + candidateCount?: number +): string { + const label = variant ? variant.charAt(0).toUpperCase() + variant.slice(1) : "Default"; + const count = + typeof candidateCount === "number" && candidateCount > 0 ? ` (${candidateCount}p)` : ""; + return `Auto: ${label}${count}`; +} + +/** + * Build the model ID for an auto combo entry. + * "auto/coding", "auto/fast", "auto" (default). + */ +export function autoComboModelId(variant: AutoVariant | undefined): string { + return variant ? `auto/${variant}` : "auto"; +} + +// ── Universal Display Name Builder ──────────────────────────────────────── + +export interface ModelDisplayNameParams { + /** Raw model ID (e.g. "cc/claude-sonnet-4-6"). */ + rawId: string; + /** Enrichment display name (e.g. "Claude Sonnet 4.6"). */ + enrichmentName?: string; + /** Provider tag enrichment. */ + providerAlias?: string; + /** Human-readable upstream provider label. */ + providerDisplayName?: string; + /** Whether model is free tier. */ + isFree?: boolean; + /** Free model budget info. */ + freeType?: FreeModelFreeType; + /** Monthly token budget (for recurring free models). */ + monthlyTokens?: number; + /** Credit token budget (for credit-based free models). */ + creditTokens?: number; + /** Whether this is a combo entry (skip provider tag). */ + isCombo?: boolean; + /** Whether this is an auto combo entry. */ + isAutoCombo?: boolean; + /** Auto combo variant. */ + autoVariant?: AutoVariant; + /** Auto combo candidate count. */ + autoCandidateCount?: number; +} + +/** + * Build the final display name following the universal template. + * + * Priority: + * 1. Auto combo → "Auto: (p)" + * 2. DB combo → "Combo: " + * 3. Free + enrichment + provider tag → "[Free]